From 1209a0d80b53b4b024aecb71b8eab88e1edd3075 Mon Sep 17 00:00:00 2001 From: Emil Lerch Date: Thu, 30 Jul 2026 16:03:36 -0700 Subject: [PATCH] human review - grouping.zig looks good now (I think) --- .kiro/specs/calculator/design.md | 30 +- .kiro/specs/calculator/tasks.md | 72 ++++ engine/src/Integer.zig | 563 +++++++++++++++++++++---- engine/src/bitwise.zig | 250 ++++++----- engine/src/engine.zig | 1 + engine/src/evaluator.zig | 44 +- engine/src/formatter.zig | 693 +++---------------------------- engine/src/grouping.zig | 186 +++++++++ engine/src/programmer.zig | 92 ++-- src/main.zig | 95 ++--- src/tui.zig | 113 +++-- src/tui/financial.zig | 45 +- src/tui/float_view.zig | 2 +- src/tui/programmer.zig | 47 ++- 14 files changed, 1181 insertions(+), 1052 deletions(-) create mode 100644 engine/src/grouping.zig diff --git a/.kiro/specs/calculator/design.md b/.kiro/specs/calculator/design.md index b3f62de..be00c57 100644 --- a/.kiro/specs/calculator/design.md +++ b/.kiro/specs/calculator/design.md @@ -61,7 +61,8 @@ build.zig (workspace root) | Module | Responsibility | |--------|---------------| -| `Integer.zig` | A fixed-width integer (file-as-struct), plus `IntType`, `BitWidth`, `Signedness` | +| `grouping.zig` | The thousands rule, shared by both display paths | +| `Integer.zig` | A fixed-width integer (file-as-struct): `raw`, `width`, `signedness`, the interpretations, and the notations it renders in | | `Rational.zig` | Exact rationals over big integers (file-as-struct) | | `number.zig` | The exact/inexact numeric model (section 2.7). Lowercase: `Number` is a tagged union, which a file-as-struct cannot express | | `tokenizer.zig` | Lexer, and `Base` for literals | @@ -70,7 +71,7 @@ build.zig (workspace root) | `evaluator.zig` | Walk AST, produce results | | `bitwise.zig` | The fixed-width operators (`& \| xor ~ << >> >>> rol ror`), shared by both modes | | `programmer.zig` | Programmer-mode evaluation, its `Config`, wrapping arithmetic | -| `formatter.zig` | Display and clipboard strings for every base | +| `formatter.zig` | Display and clipboard strings for floats and exact `Number`s, plus money and the display width | | `float_interp.zig` | IEEE 754 bit-level interpretation | | `units.zig` | Unit conversion tables and resolver | | `financial.zig` | CAGR, TVM, compound interest, amortization | @@ -253,22 +254,27 @@ Programmer mode's configuration belongs to programmer mode: ```zig // programmer.zig pub const Config = struct { - int_type: IntType = .{}, + width: BitWidth = .bits64, + signedness: Signedness = .signed, /// Big-endian by default so the HEX row reads as the number itself; see FR-2.8. display_endian: std.builtin.Endian = .big, }; -// Integer.zig: one concept, previously reassembled in three places -pub const IntType = struct { - width: BitWidth = .bits64, - signedness: Signedness = .signed, -}; +// Integer.zig: the file is the type +raw: u128, +width: BitWidth = .bits64, +signedness: Signedness = .signed, ``` -`Integer.zig` is TitleCase because the file is the type: its top-level fields are -`raw` and `int_type`, and `IntType`, `BitWidth` and `Signedness` are declared inside -it. Byte order is `std.builtin.Endian` rather than an engine enum of the same two -members. +`Integer.zig` is TitleCase because the file is the type. Width and signedness were +briefly a nested `IntType` struct, which existed only because `bitwise.zig` operated +on bare `u128` patterns and needed the pair passed alongside. Now that the operators +take values (`apply(op, left: Integer, right: Integer)`), nothing carries the pair +without bits to go with it, so the fields sit directly on the value and `Config` holds +them flat. That also removed a class of mistake: a pattern can no longer be operated on +at a width it did not come from. + +Byte order is `std.builtin.Endian` rather than an engine enum of the same two members. Standard mode evaluates to `Number`, the exact/inexact union of section 2.7. Programmer mode evaluates to `Integer`, which is a `u128` pattern plus the `IntType` that interprets it. Financial functions compute in `f64` and enter the expression language as inexact `Number` values (section 5.6). diff --git a/.kiro/specs/calculator/tasks.md b/.kiro/specs/calculator/tasks.md index 268ec93..7542ed1 100644 --- a/.kiro/specs/calculator/tasks.md +++ b/.kiro/specs/calculator/tasks.md @@ -807,6 +807,78 @@ instead of "overflow". Two tests updated to match. arithmetic, unit and financial errors. `engine.zig` at 95.6% (the two uncovered lines are a diagnostic branch inside a passing test). +### Task 5.16: Render through a writer, not into a buffer + +`Integer`'s renderers took a caller's buffer and indexed it directly, returning a +`display`/`raw` pair that lived at two offsets inside it. That is not how Zig formats +values, and the buffer had no bounds check: too small a buffer walked off the end. + +- A value now produces a view: `int.as(.hex, .{ .endian = .little })` returns a `View` + with a `format` method, so callers print it (`w.print("hex: {f}\n", .{...})`). The + notation is an enum and the decoration is `Options{ separators, prefix, endian }`, + with `Options.plain` as the no-separators, prefixed form. +- `View.render(buf)` exists for callers that need a slice, such as the TUI, which + measures and positions text before drawing. It is `Writer.fixed` underneath, so a + short buffer is `error.WriteFailed`. +- `std.Io.Writer.printInt` replaced the hand-written digit loops for the prefixed + forms, including the zero padding (`.{ .fill = '0', .width = n }`). +- `digits.zig` became `grouping.zig` and holds only the thousands rule, which is the + one thing both display paths share. Its digit writers went back to `formatter.zig` + as private helpers, which is where they were before; the module had been two thirds + formatter's private helpers travelling with their siblings. +- `formatFloat` now writes through fixed writers too, so its small-buffer behaviour is + explicit and tested rather than a length precheck. + +Both frontends got shorter: the CLI prints five rows in one `print` call with no +per-base buffers, and the programmer view asks the value for each row. + +NOT DONE, and it will be obvious when you read them: `formatter.zig`'s remaining +functions (`formatFloat`, `formatCompactFloat`, `formatNumber`, `formatAmount`, +`formatMoney`) still take buffers and return `FormattedValue`/`NumberDisplay` pairs. +That conversion reaches every display call site in both frontends, including the +financial forms and the amortization table, so it is deliberately a separate pass. + +Also found while checking callers: nothing consumes the `raw` form of an integer +rendering, and there is no clipboard code in the TUI at all. The prefixed form is kept +because design 2.6 specifies a clipboard representation and FR-6.9's neighbourhood +plans a `--raw` flag, but today it is exercised only by tests. + +- Verify: 933 tests pass, fmt and zlint clean, `Integer.zig` and `grouping.zig` at + 100% line coverage, engine 99.44%. CLI output byte-identical across all five rows, + both byte orders, ASCII packing and the multi-base standard-mode view. + +### Task 5.15: A value renders itself + +`formatter.zig` had six functions taking `(buf, value: u128, bit_width, endian)`, +which is an `Integer` taken apart. Both frontends proved it: each constructed an +`Integer` and then handed the pieces over separately, so nothing stopped a pattern +being rendered at a width it did not come from. + +- `Integer` renders itself rather than being taken apart by six formatter functions. + The first version took a buffer per base and returned a display/raw pair; Task 5.16 + replaced that with views and a writer, so read that entry for the shape the code has + now. The 27 tests moved here with the renderers. +- The primitives both display paths need went into a module of their own, since + grouping arbitrary decimal text is not an `Integer` concern and the thousands rule + has to stay defined once: `formatFloat` and the decimal notations both group. That + module started as `digits.zig` and became `grouping.zig` in 5.16, once it held only + the shared rule. Leaving the primitives in `formatter.zig` and importing that from + `Integer.zig` would also have compiled (Zig allows files to import each other; lazy + evaluation only objects to a genuine dependency cycle), but it would point the + low-level type at the high-level display module. +- `formatter.zig` keeps what is not a fixed-width integer: floats, exact `Number`s, + scientific notation, money, and `displayWidthFor`. It re-exports the two grouping + functions the TUI uses on partially typed input. + +Also in this task, the nested `IntType` went away: see design 2.5. `bitwise.apply` +takes two `Integer`s and returns one, `programmer.evalExpr` threads `Integer` instead +of bare `u128`, and `Config` holds width and signedness flat. + +- Verify: 939 tests pass, fmt and zlint clean, `Integer.zig`, the primitives module, + `bitwise.zig` and `programmer.zig` all at 100% line coverage, engine 99.45%. CLI + checked across every base row, both byte orders, both signedness readings, ASCII + packing, and the shift semantics. + ### Task 5.14: File-as-struct for the types that are types `Integer.zig` and `Rational.zig` are named TitleCase and the file *is* the type: the diff --git a/engine/src/Integer.zig b/engine/src/Integer.zig index 9060993..bb8f1ed 100644 --- a/engine/src/Integer.zig +++ b/engine/src/Integer.zig @@ -1,38 +1,62 @@ -//! A fixed-width integer: a two's complement bit pattern plus the type that says -//! how to read it. +//! A fixed-width integer: a two's complement bit pattern, and the width and +//! signedness that say how to read it. //! //! Programmer mode computes on patterns of a chosen width rather than on the exact //! rationals `number.zig` provides, and standard mode drops into the same //! representation for the bitwise operators (FR-2.12). This file is that //! representation; `bitwise.zig` is the operations on it. //! -//! There used to be three shapes of the same idea: `Integer{raw, bit_width, -//! signedness}`, `ProgrammerConfig{bit_width, signedness, display_endian}`, and a -//! `Domain{bit_width, signedness}` inside `bitwise.zig`. `IntType` is the one -//! concept each was carrying a copy of: a value pairs it with bits, and a -//! configuration pairs it with a display preference. +//! The width and signedness used to be a nested `IntType` struct, which existed only +//! because `bitwise.zig` operated on bare `u128` patterns and needed them passed +//! alongside. Now that the operators take values, nothing carries the pair without +//! bits to go with it, so the fields are here directly. `programmer.Config` holds them +//! flat too. const std = @import("std"); +const Writer = std.Io.Writer; +const grouping = @import("grouping.zig"); const Integer = @This(); /// The bit pattern. May carry bits above the width; every reader masks. raw: u128, -/// How to read `raw`. -int_type: IntType, +/// How wide the value is. +width: BitWidth = .bits64, +/// Whether the top bit is a sign. +signedness: Signedness = .signed, -pub fn init(raw: u128, int_type: IntType) Integer { - return .{ .raw = raw, .int_type = int_type }; +/// All bits set within the width. +pub fn mask(self: Integer) u128 { + return self.width.mask(); } -/// Apply the width mask, truncating to the configured width. +/// The width in bits. +pub fn bits(self: Integer) u8 { + return self.width.bits(); +} + +/// The top bit's position, whether or not this value treats it as a sign. +pub fn topBit(self: Integer) u128 { + return @as(u128, 1) << @intCast(self.bits() - 1); +} + +/// The pattern, truncated to the width. pub fn masked(self: Integer) u128 { - return self.raw & self.int_type.mask(); + return self.raw & self.mask(); +} + +/// True when the pattern denotes a negative number. An unsigned value has no +/// negative range, so `>>` is a zero fill there and a large shift distance is large +/// rather than negative. +pub fn isNegative(self: Integer) bool { + return self.signedness == .signed and self.masked() & self.topBit() != 0; } /// Interpret as a signed value, sign-extended from the width. pub fn signedValue(self: Integer) i128 { - return self.int_type.signExtend(self.raw); + const m = self.masked(); + if (self.isNegative()) return @bitCast(m | ~self.mask()); + return @intCast(m); } /// Interpret as an unsigned value, which is just the mask. @@ -40,43 +64,184 @@ pub fn unsignedValue(self: Integer) u128 { return self.masked(); } -/// A fixed-width integer type: how wide, and whether the top bit is a sign. -/// -/// Everything that interprets a pattern needs exactly this pair, which is why it -/// was being reassembled in three places. The default is 64-bit signed, which is -/// also standard mode's fixed type. -pub const IntType = struct { - width: BitWidth = .bits64, - signedness: Signedness = .signed, +/// The same width and signedness, a different pattern. How operations return their +/// result without restating the type. +pub fn withRaw(self: Integer, raw: u128) Integer { + return .{ .raw = raw & self.mask(), .width = self.width, .signedness = self.signedness }; +} - pub fn mask(self: IntType) u128 { - return self.width.mask(); +/// True when two values are the same kind of integer, so an operation between them +/// is meaningful. +pub fn sameTypeAs(self: Integer, other: Integer) bool { + return self.width == other.width and self.signedness == other.signedness; +} + +// -- Display -- +// +// A value renders itself. These used to be six functions in `formatter.zig` taking +// `(buf, value: u128, bit_width, endian)`, which is this type taken apart: both +// frontends held an `Integer` and then handed the pieces over, so nothing stopped a +// pattern being rendered at a width it did not come from. +// +// `as` returns a view, which is the value plus its formatting options and a `format` +// method, so a caller prints it: `w.print("hex: {f}\n", .{int.as(.hex, .{})})`. A +// caller that wants bytes calls `render` with a buffer, which is a `Writer.fixed` +// underneath. Writing into a buffer too small for the result is +// `error.WriteFailed` rather than a walk off the end, which is what the previous +// buffer-indexing version did. + +/// The notations a fixed-width integer renders in. +pub const Notation = enum { + hex, + octal, + binary, + /// One glyph per byte, in the same byte order as hex. + ascii, + /// Decimal, reading the pattern as signed regardless of the value's own + /// signedness, because the programmer view shows both readings of one value. + decimal_signed, + /// Decimal, reading the pattern as unsigned. + decimal_unsigned, +}; + +/// How a notation is decorated. +pub const Options = struct { + /// Group digits for reading: a space per byte in hex and ASCII, per nibble in + /// binary, per three digits in octal, and thousands commas in decimal. + separators: bool = true, + /// Prefix the base ("0x", "0o", "0b"). Decimal and ASCII have no prefix. + prefix: bool = false, + /// Byte order, for the notations that show bytes. Ignored elsewhere. + /// + /// Big-endian reads as the number itself; little-endian is the x86 memory view. + /// A prefixed form is always canonical, since `0x...` names the value. + endian: std.builtin.Endian = .big, + + /// What the clipboard wants: no separators, and a prefix where one exists. + pub const plain: Options = .{ .separators = false, .prefix = true }; +}; + +/// This value in `notation`, ready to print. +pub fn as(self: Integer, notation: Notation, options: Options) View { + return .{ .value = self, .notation = notation, .options = options }; +} + +pub const View = struct { + value: Integer, + notation: Notation, + options: Options, + + /// Render into `w`. This is the `{f}` implementation, so `int.as(.hex, .{})` + /// prints directly. + pub fn format(self: View, w: *Writer) Writer.Error!void { + const value = self.value.masked(); + switch (self.notation) { + .hex => { + if (self.options.prefix) try w.writeAll("0x"); + if (self.options.prefix or !self.options.separators) { + // Canonical, zero-padded to the width: two digits per byte. + try w.printInt(value, 16, .upper, .{ .fill = '0', .width = self.value.bits() / 4 }); + } else { + // A byte at a time in `endian` order, space-separated. Only the + // byte order swaps; nibbles within a byte are always high-low. + for (0..self.byteCount()) |k| { + if (k > 0) try w.writeByte(' '); + try w.printInt(self.byteAt(k), 16, .upper, .{ .fill = '0', .width = 2 }); + } + } + }, + .octal => { + if (self.options.prefix) try w.writeAll("0o"); + const total = self.octalDigits(); + if (!self.options.separators) { + try w.printInt(value, 8, .lower, .{ .fill = '0', .width = total }); + } else { + // Groups of three from the right, so the leftmost group may be + // short. + const first_group = if (total % 3 == 0) 3 else total % 3; + for (0..total) |i| { + if (i > 0 and (i == first_group or + (i > first_group and (i - first_group) % 3 == 0))) + { + try w.writeByte(' '); + } + const shift: u7 = @intCast((total - 1 - i) * 3); + try w.printInt((value >> shift) & 0x7, 8, .lower, .{}); + } + } + }, + .binary => { + if (self.options.prefix) try w.writeAll("0b"); + const width: usize = self.value.bits(); + if (!self.options.separators) { + try w.printInt(value, 2, .lower, .{ .fill = '0', .width = width }); + } else { + for (0..width) |i| { + if (i > 0 and i % 4 == 0) try w.writeByte(' '); + const shift: u7 = @intCast(width - 1 - i); + try w.printInt((value >> shift) & 1, 2, .lower, .{}); + } + } + }, + .ascii => { + // Two columns per byte when separated, so each glyph sits under the + // right hex digit of its byte. + for (0..self.byteCount()) |k| { + if (self.options.separators) { + if (k > 0) try w.writeByte(' '); + try w.writeByte(' '); + } + try w.writeByte(asciiGlyph(self.byteAt(k))); + } + }, + .decimal_signed => try self.printDecimal(w, self.value.signedValue()), + .decimal_unsigned => try self.printDecimal(w, value), + } } - pub fn bits(self: IntType) u8 { - return self.width.bits(); + /// Render into `buf` and return what was written. For callers that need a slice, + /// such as the TUI, which measures and positions text before drawing it. + pub fn render(self: View, buf: []u8) Writer.Error![]const u8 { + var w = Writer.fixed(buf); + try self.format(&w); + return w.buffered(); } - /// The top bit's position, whether or not this type treats it as a sign. - pub fn topBit(self: IntType) u128 { - return @as(u128, 1) << @intCast(self.bits() - 1); + fn printDecimal(self: View, w: *Writer, value: anytype) Writer.Error!void { + if (!self.options.separators) return w.printInt(value, 10, .lower, .{}); + // Grouping works on the digits, so write them first. The widest value is + // minInt(i128): 39 digits and a sign. + var plain: [40]u8 = undefined; + var digits = Writer.fixed(&plain); + try digits.printInt(value, 10, .lower, .{}); + try grouping.print(w, digits.buffered()); } - /// True when the pattern denotes a negative number in this type. An unsigned - /// type has no negative values, so `>>` is a zero fill there and a large shift - /// distance is large rather than negative. - pub fn isNegative(self: IntType, value: u128) bool { - return self.signedness == .signed and (value & self.mask()) & self.topBit() != 0; + /// The byte at display position `k` (0 = leftmost), honouring endianness. + /// Big-endian shows the most significant byte first, little-endian the least. + fn byteAt(self: View, k: usize) u8 { + const count = self.byteCount(); + const msb_index = if (self.options.endian == .big) k else count - 1 - k; + const shift: u7 = @intCast((count - 1 - msb_index) * 8); + return @intCast((self.value.masked() >> shift) & 0xFF); } - /// Sign-extend the pattern to a full i128. - pub fn signExtend(self: IntType, value: u128) i128 { - const m = value & self.mask(); - if (self.isNegative(m)) return @bitCast(m | ~self.mask()); - return @intCast(m); + fn byteCount(self: View) usize { + return @as(usize, self.value.bits()) / 8; + } + + fn octalDigits(self: View) usize { + return (@as(usize, self.value.bits()) + 2) / 3; } }; +/// Map a byte to its printable glyph, or '.' if outside the printable ASCII range, +/// the convention `xxd` and similar hex dumps use. +fn asciiGlyph(byte: u8) u8 { + if (byte >= 0x20 and byte <= 0x7E) return byte; + return '.'; +} + /// Configurable integer bit width. The tag is the bit count. pub const BitWidth = enum(u8) { bits8 = 8, @@ -123,57 +288,299 @@ test "BitWidth.mask: exactly `bits` low bits are set, at every width" { } } -test "IntType: signExtend reads the top bit only when the type is signed" { - const i8_type: IntType = .{ .width = .bits8, .signedness = .signed }; - const u8_type: IntType = .{ .width = .bits8, .signedness = .unsigned }; - - try testing.expectEqual(@as(i128, -1), i8_type.signExtend(0xFF)); - try testing.expectEqual(@as(i128, 255), u8_type.signExtend(0xFF)); - try testing.expectEqual(@as(i128, -128), i8_type.signExtend(0x80)); - try testing.expectEqual(@as(i128, 127), i8_type.signExtend(0x7F)); - - try testing.expect(i8_type.isNegative(0x80)); - try testing.expect(!u8_type.isNegative(0x80)); - - // A 128-bit type has no bits above the width to fill. - const i128_type: IntType = .{ .width = .bits128, .signedness = .signed }; - try testing.expectEqual(@as(i128, -1), i128_type.signExtend(i128_type.mask())); +test "the default is 64-bit signed, which is what standard mode uses" { + const value: Integer = .{ .raw = 0xFFFF_FFFF_FFFF_FFFF }; + try testing.expectEqual(BitWidth.bits64, value.width); + try testing.expectEqual(Signedness.signed, value.signedness); + try testing.expectEqual(@as(i128, -1), value.signedValue()); } -test "IntType: the default is 64-bit signed, which standard mode uses" { - const default: IntType = .{}; - try testing.expectEqual(BitWidth.bits64, default.width); - try testing.expectEqual(Signedness.signed, default.signedness); - try testing.expectEqual(@as(i128, -1), default.signExtend(default.mask())); +test "signedValue reads the top bit only when the value is signed" { + const signed: Integer = .{ .raw = 0xFF, .width = .bits8 }; + const unsigned: Integer = .{ .raw = 0xFF, .width = .bits8, .signedness = .unsigned }; + + try testing.expectEqual(@as(i128, -1), signed.signedValue()); + try testing.expectEqual(@as(i128, 255), unsigned.signedValue()); + try testing.expect(signed.isNegative()); + try testing.expect(!unsigned.isNegative()); + + const min: Integer = .{ .raw = 0x80, .width = .bits8 }; + try testing.expectEqual(@as(i128, -128), min.signedValue()); + const max: Integer = .{ .raw = 0x7F, .width = .bits8 }; + try testing.expectEqual(@as(i128, 127), max.signedValue()); + + // A 128-bit value has no bits above the width to fill. + const wide: Integer = .{ .raw = BitWidth.bits128.mask(), .width = .bits128 }; + try testing.expectEqual(@as(i128, -1), wide.signedValue()); } test "bits above the width never reach an interpretation" { - const i8_type: IntType = .{ .width = .bits8 }; // The high bits are noise from a wider computation. - try testing.expectEqual(@as(i128, -1), i8_type.signExtend(0xDEAD_00FF)); - try testing.expectEqual(@as(u128, 0xFF), init(0xDEAD_00FF, i8_type).unsignedValue()); + const noisy: Integer = .{ .raw = 0xDEAD_00FF, .width = .bits8 }; + try testing.expectEqual(@as(i128, -1), noisy.signedValue()); + try testing.expectEqual(@as(u128, 0xFF), noisy.unsignedValue()); } -test "signedValue" { - const i8_type: IntType = .{ .width = .bits8 }; - try testing.expectEqual(@as(i128, -1), init(0xFF, i8_type).signedValue()); - try testing.expectEqual(@as(i128, 127), init(0x7F, i8_type).signedValue()); - try testing.expectEqual(@as(i128, -128), init(0x80, i8_type).signedValue()); - - const i32_type: IntType = .{ .width = .bits32 }; - try testing.expectEqual(@as(i128, -1), init(0xFFFF_FFFF, i32_type).signedValue()); +test "signedValue at 32 bits" { + const value: Integer = .{ .raw = 0xFFFF_FFFF, .width = .bits32 }; + try testing.expectEqual(@as(i128, -1), value.signedValue()); } test "unsignedValue masks correctly" { - const u8_type: IntType = .{ .width = .bits8, .signedness = .unsigned }; - try testing.expectEqual(@as(u128, 0xFF), init(0x1FF, u8_type).unsignedValue()); + const value: Integer = .{ .raw = 0x1FF, .width = .bits8, .signedness = .unsigned }; + try testing.expectEqual(@as(u128, 0xFF), value.unsignedValue()); } -test "the same bits read two ways, which is why the type travels with the value" { - const bits: u128 = 0xFF; - try testing.expectEqual(@as(i128, -1), init(bits, .{ .width = .bits8 }).signedValue()); - try testing.expectEqual( - @as(i128, 255), - init(bits, .{ .width = .bits8, .signedness = .unsigned }).signedValue(), +test "the same bits read two ways, which is why the width travels with the value" { + const signed: Integer = .{ .raw = 0xFF, .width = .bits8 }; + const unsigned: Integer = .{ .raw = 0xFF, .width = .bits8, .signedness = .unsigned }; + try testing.expectEqual(@as(i128, -1), signed.signedValue()); + try testing.expectEqual(@as(i128, 255), unsigned.signedValue()); +} + +test "withRaw keeps the type and masks the new pattern" { + const original: Integer = .{ .raw = 0x0F, .width = .bits8, .signedness = .unsigned }; + const next = original.withRaw(0x1FF); + try testing.expectEqual(@as(u128, 0xFF), next.raw); + try testing.expectEqual(BitWidth.bits8, next.width); + try testing.expectEqual(Signedness.unsigned, next.signedness); + try testing.expect(original.sameTypeAs(next)); +} + +test "sameTypeAs distinguishes width and signedness" { + const a: Integer = .{ .raw = 0, .width = .bits8 }; + try testing.expect(a.sameTypeAs(.{ .raw = 99, .width = .bits8 })); + try testing.expect(!a.sameTypeAs(.{ .raw = 0, .width = .bits16 })); + try testing.expect(!a.sameTypeAs(.{ .raw = 0, .width = .bits8, .signedness = .unsigned })); +} + +// -- Display tests -- +// +// These moved here with the renderers, from `formatter.zig`, where each one had to +// name a width and a value separately. + +/// A signed value of the given width, the shape most of these read best in. +fn at(raw: u128, width: BitWidth) Integer { + return .{ .raw = raw, .width = width }; +} + +/// Render through a writer, which is the primary path. +fn show(buf: []u8, value: Integer, notation: Notation, options: Options) ![]const u8 { + var w = Writer.fixed(buf); + try w.print("{f}", .{value.as(notation, options)}); + return w.buffered(); +} + +test "hex: display groups per byte, plain form carries the prefix" { + var buf: [256]u8 = undefined; + + try testing.expectEqualStrings("FF", try show(&buf, at(0xFF, .bits8), .hex, .{})); + try testing.expectEqualStrings("0xFF", try show(&buf, at(0xFF, .bits8), .hex, .plain)); + + try testing.expectEqualStrings("AB CD", try show(&buf, at(0xABCD, .bits16), .hex, .{})); + try testing.expectEqualStrings("0xABCD", try show(&buf, at(0xABCD, .bits16), .hex, .plain)); + + try testing.expectEqualStrings("DE AD BE EF", try show(&buf, at(0xDEADBEEF, .bits32), .hex, .{})); + try testing.expectEqualStrings("0xDEADBEEF", try show(&buf, at(0xDEADBEEF, .bits32), .hex, .plain)); + + const quad = at(0x0123456789ABCDEF, .bits64); + try testing.expectEqualStrings("01 23 45 67 89 AB CD EF", try show(&buf, quad, .hex, .{})); + try testing.expectEqualStrings("0x0123456789ABCDEF", try show(&buf, quad, .hex, .plain)); + + // Zero still fills the width. + try testing.expectEqualStrings("00 00 00 00", try show(&buf, at(0, .bits32), .hex, .{})); + try testing.expectEqualStrings("0x00000000", try show(&buf, at(0, .bits32), .hex, .plain)); +} + +test "hex: little-endian reverses the display, never the prefixed form" { + var buf: [256]u8 = undefined; + const little: Options = .{ .endian = .little }; + + try testing.expectEqualStrings("EF BE AD DE", try show(&buf, at(0xDEADBEEF, .bits32), .hex, little)); + // A prefixed form names the value, so it stays canonical whatever the byte order. + try testing.expectEqualStrings( + "0xDEADBEEF", + try show(&buf, at(0xDEADBEEF, .bits32), .hex, .{ .endian = .little, .separators = false, .prefix = true }), + ); + + const quad = at(0x0123456789ABCDEF, .bits64); + try testing.expectEqualStrings("EF CD AB 89 67 45 23 01", try show(&buf, quad, .hex, little)); + + // A single byte has no order to reverse. + try testing.expectEqualStrings("AB", try show(&buf, at(0xAB, .bits8), .hex, little)); +} + +test "binary: display groups per nibble" { + var buf: [512]u8 = undefined; + + try testing.expectEqualStrings("1111 1111", try show(&buf, at(0xFF, .bits8), .binary, .{})); + try testing.expectEqualStrings("0b11111111", try show(&buf, at(0xFF, .bits8), .binary, .plain)); + + try testing.expectEqualStrings("1010 0101", try show(&buf, at(0b1010_0101, .bits8), .binary, .{})); + try testing.expectEqualStrings( + "1111 0000 1010 1100", + try show(&buf, at(0xF0AC, .bits16), .binary, .{}), + ); + try testing.expectEqualStrings( + "0b1111000010101100", + try show(&buf, at(0xF0AC, .bits16), .binary, .plain), ); } + +test "octal: zero-padded to the width, grouped in threes from the right" { + var buf: [256]u8 = undefined; + + // 16 bits needs 6 octal digits, which groups evenly. + try testing.expectEqualStrings("000 777", try show(&buf, at(0o777, .bits16), .octal, .{})); + try testing.expectEqualStrings("0o000777", try show(&buf, at(0o777, .bits16), .octal, .plain)); + + // 32 bits needs 11, so the leftmost group is short. + try testing.expectEqualStrings( + "00 007 777 777", + try show(&buf, at(0o7777777, .bits32), .octal, .{}), + ); + try testing.expectEqualStrings( + "0o00007777777", + try show(&buf, at(0o7777777, .bits32), .octal, .plain), + ); + + try testing.expectEqualStrings("000", try show(&buf, at(0, .bits8), .octal, .{})); + try testing.expectEqualStrings("0o000", try show(&buf, at(0, .bits8), .octal, .plain)); +} + +test "decimal: the reading is chosen by the notation, not by the value's signedness" { + var buf: [256]u8 = undefined; + + // The same 8-bit pattern, both ways: the programmer view shows both rows. + const pattern = at(0xFF, .bits8); + try testing.expectEqualStrings("-1", try show(&buf, pattern, .decimal_signed, .{})); + try testing.expectEqualStrings("255", try show(&buf, pattern, .decimal_unsigned, .{})); + + // Width decides what the sign bit is: 0xFF is -1 in 8 bits, 255 in 16. + try testing.expectEqualStrings("255", try show(&buf, at(0xFF, .bits16), .decimal_signed, .{})); +} + +test "decimal: display groups in thousands, plain does not" { + var buf: [256]u8 = undefined; + + const large = at(4294967295, .bits32); + try testing.expectEqualStrings("4,294,967,295", try show(&buf, large, .decimal_unsigned, .{})); + try testing.expectEqualStrings("4294967295", try show(&buf, large, .decimal_unsigned, .plain)); + + const negative = at(@bitCast(@as(i128, -1234567)), .bits32); + try testing.expectEqualStrings("-1,234,567", try show(&buf, negative, .decimal_signed, .{})); + try testing.expectEqualStrings("-1234567", try show(&buf, negative, .decimal_signed, .plain)); + + // Three digits or fewer have nothing to group. + try testing.expectEqualStrings("255", try show(&buf, at(255, .bits8), .decimal_unsigned, .{})); +} + +test "decimal: minInt(i128) formats instead of panicking" { + // A 128-bit word with only the sign bit set. Taking its magnitude by negation + // overflows, which used to panic in Debug and ReleaseSafe. + var buf: [256]u8 = undefined; + const min = at(@as(u128, 1) << 127, .bits128); + try testing.expectEqualStrings( + "-170141183460469231731687303715884105728", + try show(&buf, min, .decimal_signed, .plain), + ); + try testing.expectEqualStrings( + "-170,141,183,460,469,231,731,687,303,715,884,105,728", + try show(&buf, min, .decimal_signed, .{}), + ); + + const max = at(~(@as(u128, 1) << 127), .bits128); + try testing.expectEqualStrings( + "170,141,183,460,469,231,731,687,303,715,884,105,727", + try show(&buf, max, .decimal_signed, .{}), + ); + + try testing.expectEqualStrings("0", try show(&buf, at(0, .bits128), .decimal_signed, .plain)); +} + +test "ascii: printable bytes render, everything else is a dot" { + var buf: [128]u8 = undefined; + + // "..asciii" packed into 64 bits: two zero bytes then the letters. + const word = at(0x0000_6173_6369_6969, .bits64); + try testing.expectEqualStrings("..asciii", try show(&buf, word, .ascii, .plain)); + // Two columns per byte so each glyph sits under its hex pair. + try testing.expectEqualStrings(" . . a s c i i i", try show(&buf, word, .ascii, .{})); + + try testing.expectEqualStrings("A", try show(&buf, at('A', .bits8), .ascii, .plain)); + try testing.expectEqualStrings(" A", try show(&buf, at('A', .bits8), .ascii, .{})); + + // Control and high bytes are dots. + try testing.expectEqualStrings(".", try show(&buf, at(0x00, .bits8), .ascii, .plain)); + try testing.expectEqualStrings(".", try show(&buf, at(0x80, .bits8), .ascii, .plain)); + + // The boundaries of the printable range, and one past it. + try testing.expectEqualStrings(" ", try show(&buf, at(0x20, .bits8), .ascii, .plain)); + try testing.expectEqualStrings("~", try show(&buf, at(0x7E, .bits8), .ascii, .plain)); + try testing.expectEqualStrings(".", try show(&buf, at(0x7F, .bits8), .ascii, .plain)); +} + +test "ascii: little-endian reverses the byte order, as hex does" { + var buf: [128]u8 = undefined; + const value = at(0x4142_4344, .bits32); + try testing.expectEqualStrings("ABCD", try show(&buf, value, .ascii, .plain)); + try testing.expectEqualStrings( + "DCBA", + try show(&buf, value, .ascii, .{ .separators = false, .endian = .little }), + ); +} + +test "display: bits above the width never appear" { + // The views mask, so noise from a wider computation cannot leak into a row. + var buf: [512]u8 = undefined; + const noisy = at(0xDEAD_BEEF_0000_00FF, .bits8); + try testing.expectEqualStrings("FF", try show(&buf, noisy, .hex, .{})); + try testing.expectEqualStrings("1111 1111", try show(&buf, noisy, .binary, .{})); + try testing.expectEqualStrings("377", try show(&buf, noisy, .octal, .{})); + try testing.expectEqualStrings("255", try show(&buf, noisy, .decimal_unsigned, .{})); + try testing.expectEqualStrings("-1", try show(&buf, noisy, .decimal_signed, .{})); +} + +test "display: every row has the width's worth of digits, at every width" { + var buf: [512]u8 = undefined; + for (std.enums.values(BitWidth)) |bw| { + const value: Integer = .{ .raw = bw.mask(), .width = bw, .signedness = .unsigned }; + const bits_count: usize = bw.bits(); + const bytes = bits_count / 8; + + // Hex: two digits per byte plus a space between bytes. + try testing.expectEqual(bytes * 3 - 1, (try show(&buf, value, .hex, .{})).len); + // Binary: one digit per bit plus a space per nibble boundary. + try testing.expectEqual( + bits_count + bits_count / 4 - 1, + (try show(&buf, value, .binary, .{})).len, + ); + // ASCII: two columns per byte plus a space between bytes. + try testing.expectEqual(bytes * 3 - 1, (try show(&buf, value, .ascii, .{})).len); + // The prefixed forms are the prefix plus one digit per unit of the base. + try testing.expectEqual(2 + bits_count / 4, (try show(&buf, value, .hex, .plain)).len); + try testing.expectEqual(2 + bits_count, (try show(&buf, value, .binary, .plain)).len); + } +} + +test "render: a buffer too small is an error, not a walk off the end" { + // The buffer-indexing version this replaced had no bounds check at all. + var tiny: [4]u8 = undefined; + try testing.expectError( + error.WriteFailed, + at(0xDEADBEEF, .bits32).as(.binary, .{}).render(&tiny), + ); + + // And the successful path returns exactly what was written. + var room: [64]u8 = undefined; + const written = try at(0xFF, .bits8).as(.hex, .{}).render(&room); + try testing.expectEqualStrings("FF", written); +} + +test "a view prints through any writer, which is how the frontends use it" { + var buf: [128]u8 = undefined; + var w = Writer.fixed(&buf); + const value = at(0xDEAD, .bits16); + try w.print("hex: {f} oct: {f}", .{ value.as(.hex, .{}), value.as(.octal, .{}) }); + try testing.expectEqualStrings("hex: DE AD oct: 157 255", w.buffered()); +} diff --git a/engine/src/bitwise.zig b/engine/src/bitwise.zig index 8313ede..fcae777 100644 --- a/engine/src/bitwise.zig +++ b/engine/src/bitwise.zig @@ -15,20 +15,19 @@ //! `~`. //! //! FR-2.12 promises that every operator means the same thing in both modes, so -//! there is one implementation, parameterised by an `IntType` (width plus -//! signedness). Standard mode is fixed at 64-bit signed; a different width is what +//! there is one implementation, over `Integer` values that carry their own width and +//! signedness. Standard mode is fixed at 64-bit signed; a different width is what //! programmer mode is for (FR-2.3). //! -//! Values are carried as a `u128` holding the two's complement bit pattern masked -//! to the width, which is the same representation `Integer` uses. +//! Operands are `Integer`, not bare patterns plus a separate type. Taking them apart +//! made it possible to hand an operation a pattern from one width and a type from +//! another, which nothing would have noticed. const std = @import("std"); const ast = @import("ast.zig"); const BinaryOp = ast.BinaryOp; const Integer = @import("Integer.zig"); const BitWidth = Integer.BitWidth; -const Signedness = Integer.Signedness; -const IntType = Integer.IntType; /// The one way a fixed-width operation can fail: a shift or rotate distance that is /// negative in the operand's type. Everything else about these operators is total. @@ -82,10 +81,10 @@ const Distance = union(enum) { /// /// A negative distance is a domain error rather than a very large one. Standard /// mode used to reduce it modulo 64, so `8 >> -1` quietly became `8 >> 63`. -fn distance(int_type: IntType, right: u128) Error!Distance { - if (int_type.isNegative(right)) return Error.DomainError; - const value = right & int_type.mask(); - if (value >= int_type.bits()) return .past_width; +fn distance(right: Integer) Error!Distance { + if (right.isNegative()) return Error.DomainError; + const value = right.masked(); + if (value >= right.bits()) return .past_width; return .{ .within = @intCast(value) }; } @@ -93,76 +92,95 @@ fn distance(int_type: IntType, right: u128) Error!Distance { /// /// Rotation is cyclic, so a distance beyond the width is reduced rather than /// saturated: rotating a 64-bit value by 65 is rotating it by 1. -fn rotation(int_type: IntType, right: u128) Error!u7 { - if (int_type.isNegative(right)) return Error.DomainError; - return @intCast((right & int_type.mask()) % int_type.bits()); +fn rotation(right: Integer) Error!u7 { + if (right.isNegative()) return Error.DomainError; + return @intCast(right.masked() % right.bits()); } -/// Apply a fixed-width operation. Operands and result are masked bit patterns. -pub fn apply(int_type: IntType, op: Op, left_in: u128, right_in: u128) Error!u128 { - const mask = int_type.mask(); - const left = left_in & mask; - const right = right_in & mask; +/// Apply a fixed-width operation to two values of the same integer type. +/// +/// Both operands must be the same width and signedness; in practice they come from +/// one evaluation with one configuration. The result takes the left operand's type. +pub fn apply(op: Op, left_in: Integer, right_in: Integer) Error!Integer { + std.debug.assert(left_in.sameTypeAs(right_in)); - return switch (op) { + const mask = left_in.mask(); + const left = left_in.masked(); + const right = right_in.masked(); + + const raw: u128 = switch (op) { .bit_and => left & right, .bit_or => left | right, .bit_xor => left ^ right, // Bits shifted past the end of the width are discarded, not wrapped: // `0b1000 << 1` is 16. Wrapping is what `rol` and `ror` are for. - .shift_left => switch (try distance(int_type, right)) { + .shift_left => switch (try distance(right_in)) { .past_width => 0, .within => |amt| (left << amt) & mask, }, - .shift_right_logical => switch (try distance(int_type, right)) { + .shift_right_logical => switch (try distance(right_in)) { .past_width => 0, - .within => |amt| (left >> amt) & mask, + .within => |amt| left >> amt, }, .shift_right => blk: { - const negative = int_type.isNegative(left); - switch (try distance(int_type, right)) { + const negative = left_in.isNegative(); + switch (try distance(right_in)) { // Shifting a negative value all the way out leaves the sign fill, // which is every bit set; a non-negative one leaves zero. .past_width => break :blk if (negative) mask else 0, .within => |amt| { - if (!negative) break :blk (left >> amt) & mask; - // Shift in the signed domain so the sign bit is the fill. - const extended = int_type.signExtend(left); + if (!negative) break :blk left >> amt; + // Shift the sign-extended value so the sign bit is the fill. + const extended = left_in.signedValue(); break :blk @as(u128, @bitCast(extended >> amt)) & mask; }, } }, .rotate_left => blk: { - const amt = try rotation(int_type, right); + const amt = try rotation(right_in); if (amt == 0) break :blk left; - const anti: u7 = @intCast(int_type.bits() - amt); + const anti: u7 = @intCast(left_in.bits() - amt); break :blk ((left << amt) | (left >> anti)) & mask; }, .rotate_right => blk: { - const amt = try rotation(int_type, right); + const amt = try rotation(right_in); if (amt == 0) break :blk left; - const anti: u7 = @intCast(int_type.bits() - amt); + const anti: u7 = @intCast(left_in.bits() - amt); break :blk ((left >> amt) | (left << anti)) & mask; }, }; + + return left_in.withRaw(raw); } /// Bitwise complement within the width. -pub fn not(int_type: IntType, value: u128) u128 { - return ~value & int_type.mask(); +pub fn not(value: Integer) Integer { + return value.withRaw(~value.masked()); } /// Two's complement negation within the width. -pub fn negate(int_type: IntType, value: u128) u128 { - return (~value +% 1) & int_type.mask(); +pub fn negate(value: Integer) Integer { + return value.withRaw(~value.masked() +% 1); } // -- Tests -- const testing = std.testing; -const i8_type: IntType = .{ .width = .bits8 }; -const u8_type: IntType = .{ .width = .bits8, .signedness = .unsigned }; +/// An 8-bit signed value, the width most of these cases are easiest to read in. +fn i8v(raw: u128) Integer { + return .{ .raw = raw, .width = .bits8 }; +} + +/// An 8-bit unsigned value. +fn u8v(raw: u128) Integer { + return .{ .raw = raw, .width = .bits8, .signedness = .unsigned }; +} + +/// A 64-bit signed value, which is what standard mode uses. +fn i64v(raw: u128) Integer { + return .{ .raw = raw }; +} test "fromBinaryOp: every fixed-width operator maps, no arithmetic one does" { // The compiler enforces the total mapping; this pins which side each lands on. @@ -178,17 +196,24 @@ test "fromBinaryOp: every fixed-width operator maps, no arithmetic one does" { try testing.expectEqual(@as(usize, @typeInfo(Op).@"enum".fields.len), fixed_width); } +test "the result carries the operands' type" { + const result = try apply(.bit_and, u8v(0xFF), u8v(0x0F)); + try testing.expectEqual(Integer.BitWidth.bits8, result.width); + try testing.expectEqual(Integer.Signedness.unsigned, result.signedness); + try testing.expectEqual(@as(u128, 0x0F), result.raw); +} + test "arithmetic right shift fills with the sign bit" { // -8 in 8 bits is 0b1111_1000; one place right is 0b1111_1100, which is -4. - const result = try apply(i8_type, .shift_right, 0b1111_1000, 1); - try testing.expectEqual(@as(u128, 0b1111_1100), result); - try testing.expectEqual(@as(i128, -4), i8_type.signExtend(result)); + const result = try apply(.shift_right, i8v(0b1111_1000), i8v(1)); + try testing.expectEqual(@as(u128, 0b1111_1100), result.raw); + try testing.expectEqual(@as(i128, -4), result.signedValue()); } test "logical right shift fills with zeros" { // The same bits, shifted the other way: 0b0111_1100 is 124. - const result = try apply(i8_type, .shift_right_logical, 0b1111_1000, 1); - try testing.expectEqual(@as(u128, 124), result); + const result = try apply(.shift_right_logical, i8v(0b1111_1000), i8v(1)); + try testing.expectEqual(@as(u128, 124), result.raw); } test "the two right shifts agree on non-negative values" { @@ -196,115 +221,120 @@ test "the two right shifts agree on non-negative values" { while (value < 0x80) : (value += 1) { var amt: u128 = 0; while (amt < 8) : (amt += 1) { - try testing.expectEqual( - try apply(i8_type, .shift_right, value, amt), - try apply(i8_type, .shift_right_logical, value, amt), - ); + const arithmetic = try apply(.shift_right, i8v(value), i8v(amt)); + const logical = try apply(.shift_right_logical, i8v(value), i8v(amt)); + try testing.expectEqual(arithmetic.raw, logical.raw); } } } -test "an unsigned domain has no sign to extend" { - // 0xFF is 255 here, not -1, so the arithmetic shift is a zero fill too. The - // old programmer-mode implementation looked at the top bit regardless of the +test "an unsigned value has no sign to extend" { + // 0xFF is 255 here, not -1, so the arithmetic shift is a zero fill too. The old + // programmer-mode implementation looked at the top bit regardless of the // configured signedness and gave 0xFF. - try testing.expectEqual(@as(u128, 0x7F), try apply(u8_type, .shift_right, 0xFF, 1)); - try testing.expectEqual(@as(u128, 0x7F), try apply(u8_type, .shift_right_logical, 0xFF, 1)); + const arithmetic = try apply(.shift_right, u8v(0xFF), u8v(1)); + const logical = try apply(.shift_right_logical, u8v(0xFF), u8v(1)); + try testing.expectEqual(@as(u128, 0x7F), arithmetic.raw); + try testing.expectEqual(@as(u128, 0x7F), logical.raw); } test "shifts run to completion instead of wrapping or clamping the distance" { // Standard mode reduced the distance modulo the width, so `1 << 64` was 1; // programmer mode clamped it to width - 1, so 8-bit `0xFF >>> 20` was 1. - const i64_type: IntType = .{}; - try testing.expectEqual(@as(u128, 0), try apply(i64_type, .shift_left, 1, 64)); - try testing.expectEqual(@as(u128, 0), try apply(i64_type, .shift_left, 1, 1000)); - try testing.expectEqual(@as(u128, 0), try apply(i8_type, .shift_right_logical, 0xFF, 20)); - try testing.expectEqual(@as(u128, 0), try apply(i8_type, .shift_left, 0xFF, 8)); + try testing.expectEqual(@as(u128, 0), (try apply(.shift_left, i64v(1), i64v(64))).raw); + try testing.expectEqual(@as(u128, 0), (try apply(.shift_left, i64v(1), i64v(1000))).raw); + try testing.expectEqual(@as(u128, 0), (try apply(.shift_right_logical, i8v(0xFF), i8v(20))).raw); + try testing.expectEqual(@as(u128, 0), (try apply(.shift_left, i8v(0xFF), i8v(8))).raw); // A negative value shifted all the way out is all sign bits, not zero. - try testing.expectEqual(@as(u128, 0xFF), try apply(i8_type, .shift_right, 0b1111_1000, 8)); - try testing.expectEqual(@as(u128, 0xFF), try apply(i8_type, .shift_right, 0b1111_1000, 100)); + try testing.expectEqual(@as(u128, 0xFF), (try apply(.shift_right, i8v(0b1111_1000), i8v(8))).raw); + try testing.expectEqual(@as(u128, 0xFF), (try apply(.shift_right, i8v(0b1111_1000), i8v(100))).raw); // A non-negative one is zero. - try testing.expectEqual(@as(u128, 0), try apply(i8_type, .shift_right, 0b0100_0000, 8)); + try testing.expectEqual(@as(u128, 0), (try apply(.shift_right, i8v(0b0100_0000), i8v(8))).raw); } test "shifting by one less than the width still keeps a bit" { // The boundary the clamping rule used to hide. - try testing.expectEqual(@as(u128, 0b1000_0000), try apply(i8_type, .shift_left, 1, 7)); - try testing.expectEqual(@as(u128, 1), try apply(i8_type, .shift_right_logical, 0b1000_0000, 7)); - try testing.expectEqual(@as(u128, 0xFF), try apply(i8_type, .shift_right, 0b1000_0000, 7)); + try testing.expectEqual(@as(u128, 0b1000_0000), (try apply(.shift_left, i8v(1), i8v(7))).raw); + try testing.expectEqual(@as(u128, 1), (try apply(.shift_right_logical, i8v(0b1000_0000), i8v(7))).raw); + try testing.expectEqual(@as(u128, 0xFF), (try apply(.shift_right, i8v(0b1000_0000), i8v(7))).raw); } test "a negative shift distance is a domain error, not a huge one" { - const neg_one: u128 = 0xFF; // -1 in 8-bit signed - try testing.expectError(Error.DomainError, apply(i8_type, .shift_left, 1, neg_one)); - try testing.expectError(Error.DomainError, apply(i8_type, .shift_right, 1, neg_one)); - try testing.expectError(Error.DomainError, apply(i8_type, .shift_right_logical, 1, neg_one)); - try testing.expectError(Error.DomainError, apply(i8_type, .rotate_left, 1, neg_one)); - try testing.expectError(Error.DomainError, apply(i8_type, .rotate_right, 1, neg_one)); - // The same pattern in an unsigned domain is 255, a distance past the width. - try testing.expectEqual(@as(u128, 0), try apply(u8_type, .shift_left, 1, neg_one)); + const neg_one = i8v(0xFF); // -1 in 8-bit signed + try testing.expectError(Error.DomainError, apply(.shift_left, i8v(1), neg_one)); + try testing.expectError(Error.DomainError, apply(.shift_right, i8v(1), neg_one)); + try testing.expectError(Error.DomainError, apply(.shift_right_logical, i8v(1), neg_one)); + try testing.expectError(Error.DomainError, apply(.rotate_left, i8v(1), neg_one)); + try testing.expectError(Error.DomainError, apply(.rotate_right, i8v(1), neg_one)); + // The same pattern in an unsigned value is 255, a distance past the width. + try testing.expectEqual(@as(u128, 0), (try apply(.shift_left, u8v(1), u8v(0xFF))).raw); } test "rotation is cyclic and reduces the distance" { - try testing.expectEqual(@as(u128, 0b0000_0011), try apply(i8_type, .rotate_left, 0b1000_0001, 1)); - try testing.expectEqual(@as(u128, 0b1100_0000), try apply(i8_type, .rotate_right, 0b1000_0001, 1)); + try testing.expectEqual( + @as(u128, 0b0000_0011), + (try apply(.rotate_left, i8v(0b1000_0001), i8v(1))).raw, + ); + try testing.expectEqual( + @as(u128, 0b1100_0000), + (try apply(.rotate_right, i8v(0b1000_0001), i8v(1))).raw, + ); // Rotating by the width is the identity, and by width + 1 is by 1. - try testing.expectEqual(@as(u128, 0b1000_0001), try apply(i8_type, .rotate_left, 0b1000_0001, 8)); - try testing.expectEqual(@as(u128, 0b0000_0011), try apply(i8_type, .rotate_left, 0b1000_0001, 9)); + try testing.expectEqual( + @as(u128, 0b1000_0001), + (try apply(.rotate_left, i8v(0b1000_0001), i8v(8))).raw, + ); + try testing.expectEqual( + @as(u128, 0b0000_0011), + (try apply(.rotate_left, i8v(0b1000_0001), i8v(9))).raw, + ); } test "rotate left and rotate right are inverses at every distance and width" { - for ([_]BitWidth{ .bits8, .bits16, .bits32, .bits64, .bits128 }) |bw| { - const int_type: IntType = .{ .width = bw, .signedness = .unsigned }; - const value: u128 = 0x1234_5678_9ABC_DEF0 & int_type.mask(); + for (std.enums.values(BitWidth)) |bw| { + const value: Integer = .{ + .raw = 0x1234_5678_9ABC_DEF0 & bw.mask(), + .width = bw, + .signedness = .unsigned, + }; var amt: u128 = 0; - while (amt < int_type.bits()) : (amt += 1) { - const there = try apply(int_type, .rotate_left, value, amt); - const back = try apply(int_type, .rotate_right, there, amt); - try testing.expectEqual(value, back); + while (amt < value.bits()) : (amt += 1) { + const there = try apply(.rotate_left, value, value.withRaw(amt)); + const back = try apply(.rotate_right, there, value.withRaw(amt)); + try testing.expectEqual(value.raw, back.raw); } } } -test "results stay inside the width" { - for ([_]BitWidth{ .bits8, .bits16, .bits32, .bits64, .bits128 }) |bw| { - const int_type: IntType = .{ .width = bw, .signedness = .signed }; - const all_ones = int_type.mask(); +test "results stay inside the width, at every width and operator" { + for (std.enums.values(BitWidth)) |bw| { + const all_ones: Integer = .{ .raw = bw.mask(), .width = bw }; inline for (@typeInfo(Op).@"enum".fields) |field| { const op = @field(Op, field.name); // 1 is a safe distance for the shifts and a legal operand for the rest. - const result = try apply(int_type, op, all_ones, 1); - try testing.expectEqual(result, result & int_type.mask()); + const result = try apply(op, all_ones, all_ones.withRaw(1)); + try testing.expectEqual(result.raw, result.masked()); + try testing.expect(result.sameTypeAs(all_ones)); } } } -test "not and negate stay inside the width" { - try testing.expectEqual(@as(u128, 0xFF), not(i8_type, 0)); - try testing.expectEqual(@as(u128, 0), not(i8_type, 0xFF)); - try testing.expectEqual(@as(u128, 0xFF), negate(i8_type, 1)); - try testing.expectEqual(@as(u128, 1), negate(i8_type, 0xFF)); +test "not and negate stay inside the width and keep the type" { + try testing.expectEqual(@as(u128, 0xFF), not(i8v(0)).raw); + try testing.expectEqual(@as(u128, 0), not(i8v(0xFF)).raw); + try testing.expectEqual(@as(u128, 0xFF), negate(i8v(1)).raw); + try testing.expectEqual(@as(u128, 1), negate(i8v(0xFF)).raw); // Negating the most negative value gives itself back, as two's complement does. - try testing.expectEqual(@as(u128, 0x80), negate(i8_type, 0x80)); + try testing.expectEqual(@as(u128, 0x80), negate(i8v(0x80)).raw); + try testing.expect(not(u8v(0)).sameTypeAs(u8v(0))); } -test "signExtend reads the top bit only when the domain is signed" { - try testing.expectEqual(@as(i128, -1), i8_type.signExtend(0xFF)); - try testing.expectEqual(@as(i128, 255), u8_type.signExtend(0xFF)); - try testing.expectEqual(@as(i128, -128), i8_type.signExtend(0x80)); - try testing.expectEqual(@as(i128, 127), i8_type.signExtend(0x7F)); - // A 128-bit domain has no bits above the width to fill. - const i128_type: IntType = .{ .width = .bits128 }; - try testing.expectEqual(@as(i128, -1), i128_type.signExtend(i128_type.mask())); -} - -test "the default type, which standard mode uses, is 64-bit signed" { - const i64_type: IntType = .{}; - try testing.expectEqual(BitWidth.bits64, i64_type.width); - try testing.expectEqual(Signedness.signed, i64_type.signedness); +test "the default value type, which standard mode uses, is 64-bit signed" { + const minus_eight = i64v(@bitCast(@as(i128, -8) & @as(i128, @bitCast(i64v(0).mask())))); + try testing.expectEqual(Integer.BitWidth.bits64, minus_eight.width); + try testing.expectEqual(Integer.Signedness.signed, minus_eight.signedness); // -8 >> 1 is -4 there, which is the case that used to differ between modes. - const minus_eight: u128 = @bitCast(@as(i128, -8) & @as(i128, @bitCast(i64_type.mask()))); - const shifted = try apply(i64_type, .shift_right, minus_eight, 1); - try testing.expectEqual(@as(i128, -4), i64_type.signExtend(shifted)); + const shifted = try apply(.shift_right, minus_eight, minus_eight.withRaw(1)); + try testing.expectEqual(@as(i128, -4), shifted.signedValue()); } diff --git a/engine/src/engine.zig b/engine/src/engine.zig index d67fe30..88ee6b4 100644 --- a/engine/src/engine.zig +++ b/engine/src/engine.zig @@ -6,6 +6,7 @@ const std = @import("std"); // Vocabulary, lowest first. +pub const grouping = @import("grouping.zig"); pub const Integer = @import("Integer.zig"); // Exact numeric model (design.md 2.7). The evaluator computes in these. pub const Rational = @import("Rational.zig"); diff --git a/engine/src/evaluator.zig b/engine/src/evaluator.zig index 3dc3521..6fc63b7 100644 --- a/engine/src/evaluator.zig +++ b/engine/src/evaluator.zig @@ -12,7 +12,6 @@ const ast = @import("ast.zig"); const Expr = ast.Expr; const BinaryOp = ast.BinaryOp; const Integer = @import("Integer.zig"); -const IntType = Integer.IntType; /// What standard-mode evaluation can fail with. /// /// Its own name and range errors, plus everything its dependencies can raise. The @@ -40,7 +39,7 @@ const financial = @import("financial.zig"); /// /// It holds no mode and no programmer configuration. It used to hold both and read /// neither: the caller chooses between `evalString` and `evalProgrammerString`, and -/// standard mode's integer type is fixed (`standard_int_type`). The TUI was writing +/// standard mode's integer type is fixed (see `standardInt`). The TUI was writing /// a mode into this on every mode change, into a field nothing consulted. pub const Environment = struct { allocator: Allocator, @@ -183,8 +182,7 @@ fn evalExact(env: *Environment, scratch: Allocator, expr: *const Expr) Error!Num // 255 in standard mode while the shifts alongside it ignored the // setting entirely. .bitwise_not => blk: { - const bits = try toFixedWidthBits(operand.toFloat(scratch)); - break :blk fromFixedWidthBits(bitwise.not(standard_int_type, bits)); + break :blk fromStandardInt(bitwise.not(try standardInt(operand.toFloat(scratch)))); }, }; }, @@ -235,35 +233,27 @@ fn evalBinaryOp(scratch: Allocator, op: BinaryOp, left: Number, right: Number) E // resolves the operator at comptime, so an operator added to `BinaryOp` // that `bitwise.fromBinaryOp` does not know is a compile error here. inline else => |fixed_op| blk: { - const l = try toFixedWidthBits(left.toFloat(scratch)); - const r = try toFixedWidthBits(right.toFloat(scratch)); - const result = try bitwise.apply( - standard_int_type, - comptime bitwise.fromBinaryOp(fixed_op).?, - l, - r, - ); - break :blk fromFixedWidthBits(result); + const l = try standardInt(left.toFloat(scratch)); + const r = try standardInt(right.toFloat(scratch)); + const result = try bitwise.apply(comptime bitwise.fromBinaryOp(fixed_op).?, l, r); + break :blk fromStandardInt(result); }, }; } -/// Standard mode's integer type: 64-bit two's complement, fixed (FR-2.3), which is -/// also `IntType`'s default. +/// Project a float onto standard mode's integer type: 64-bit two's complement, +/// fixed (FR-2.3), which is also `Integer`'s default. /// /// Standard mode does not consult the programmer-mode width. A width other than 64 /// is what programmer mode is for, and pretending otherwise is how `~` came to /// honour the setting while the shifts beside it did not. -const standard_int_type: IntType = .{}; - -/// Project a float onto the integer domain the bitwise operators work in. /// /// Every one of these operators used to do `@intFromFloat` straight onto the /// unchecked value, which is illegal behaviour out of range and aborted the /// process: `2^64 and 1` and `~1e30` both killed it, and a NaN operand produced a /// garbage answer instead. An operand that does not fit the width is a reportable /// error, not a crash. -fn toFixedWidthBits(value: f64) Error!u128 { +fn standardInt(value: f64) Error!Integer { if (!math.isFinite(value)) return Error.DomainError; // i64 covers [-2^63, 2^63); 2^63 itself is the first excluded value and is // exactly representable, so these bounds are exact. @@ -271,12 +261,12 @@ fn toFixedWidthBits(value: f64) Error!u128 { return Error.Overflow; } const bits: u64 = @bitCast(@as(i64, @intFromFloat(value))); - return @as(u128, bits); + return .{ .raw = @as(u128, bits) }; } -/// Read a result pattern back as a number, signed, since standard mode is signed. -fn fromFixedWidthBits(bits: u128) Number { - return Number.fromFloat(@floatFromInt(standard_int_type.signExtend(bits))); +/// Read a result back as a number, signed, since standard mode is signed. +fn fromStandardInt(value: Integer) Number { + return Number.fromFloat(@floatFromInt(value.signedValue())); } /// Evaluate a built-in function call. @@ -881,7 +871,8 @@ test "standard mode: the fixed-width operators use one implementation with progr for (shared) |source| { const standard = try testEval(source); const prog = try programmer_mod.evalProgrammerString(alloc, source, .{ - .int_type = .{ .width = .bits64, .signedness = .signed }, + .width = .bits64, + .signedness = .signed, }); try testing.expectEqual(@as(i128, @intFromFloat(standard)), prog.signedValue()); } @@ -945,8 +936,9 @@ test "standard mode: its integer type is fixed at 64-bit signed" { defer shifted.deinit(); try testing.expectEqual(@as(f64, 1024.0), shifted.toFloat(alloc)); - try testing.expectEqual(@as(u8, 64), standard_int_type.bits()); - try testing.expectEqual(Integer.Signedness.signed, standard_int_type.signedness); + const projected = try standardInt(1); + try testing.expectEqual(@as(u8, 64), projected.bits()); + try testing.expectEqual(Integer.Signedness.signed, projected.signedness); } test "eval rotate left in standard mode" { diff --git a/engine/src/formatter.zig b/engine/src/formatter.zig index 58bdc92..b9179e5 100644 --- a/engine/src/formatter.zig +++ b/engine/src/formatter.zig @@ -19,7 +19,7 @@ const std = @import("std"); const Integer = @import("Integer.zig"); const BitWidth = Integer.BitWidth; -const Endianness = std.builtin.Endian; +const grouping = @import("grouping.zig"); const Number = @import("number.zig").Number; /// A formatted value with both display and clipboard representations. @@ -28,6 +28,18 @@ pub const FormattedValue = struct { raw: []const u8, }; +/// Length of the grouped form of `text`, and the writer for it. Private: the TUI +/// groups partially typed input through `grouping` directly, and these are only the +/// buffer-shaped conveniences this file's own buffer-based functions need. +const groupedDecimalLen = grouping.lengthOf; + +/// Group `text` into `dest`, which must hold `groupedDecimalLen(text)` bytes. +fn writeGroupedDecimal(dest: []u8, text: []const u8) usize { + var w = std.Io.Writer.fixed(dest); + grouping.print(&w, text) catch @panic("grouping buffer too small; ask groupedDecimalLen first"); + return w.end; +} + /// Format a floating-point value for display. /// Uses comma grouping for integers, avoids scientific notation unless necessary. /// @@ -45,17 +57,18 @@ pub fn formatFloat(buf: []u8, value: f64) FormattedValue { const is_integer = value == @trunc(value) and @abs(value) < 9007199254740992.0; // 2^53 if (is_integer and @abs(value) < 1e15) { - // Format as integer with commas + // An integral value: plain digits, then the same digits grouped. const int_val: i128 = @intFromFloat(value); - const raw_len = writeSignedInt(buf, int_val); - const raw = buf[0..raw_len]; + var raw_writer = std.Io.Writer.fixed(buf); + raw_writer.printInt(int_val, 10, .lower, .{}) catch + return .{ .display = "ERR", .raw = "ERR" }; + const raw = raw_writer.buffered(); - // Now write the display version (with commas) after the raw version - const display_start = raw_len; - const display_len = writeDecimalWithCommas(buf[display_start..], int_val); - const display = buf[display_start..][0..display_len]; + var display_writer = std.Io.Writer.fixed(buf[raw.len..]); + grouping.print(&display_writer, raw) catch + return .{ .display = raw, .raw = raw }; - return .{ .display = display, .raw = raw }; + return .{ .display = display_writer.buffered(), .raw = raw }; } // Check if we should use scientific notation @@ -138,7 +151,7 @@ pub fn formatNumber(allocator: std.mem.Allocator, value: Number) !NumberDisplay // worked on the already-rendered text; two implementations of the same // notation is one more than needed, and the rational-based one handles // both ends of the range. - if (integerDigitCount(rendered.text) > max_display_integer_digits) { + if (grouping.integerDigitCount(rendered.text) > max_display_integer_digits) { const display = try r.toScientificString(allocator, scientific_significant_digits); return .{ .display = display, .raw = rendered.text, .exact = false }; } @@ -191,36 +204,6 @@ pub const max_display_integer_digits: usize = 40; /// Significant digits kept when abbreviating to scientific notation. const scientific_significant_digits: usize = 17; -/// The three parts of decimal text: an optional sign, the integer digits, and -/// everything from the decimal point onward. -/// -/// One place that knows how to take decimal text apart. `integerDigitCount` and -/// `writeGroupedDecimal` each used to work it out themselves, which is two chances -/// to disagree about where the sign ends. -const DecimalParts = struct { - /// Length of the sign, 0 or 1. - sign_len: usize, - /// Number of digits before the decimal point. - int_digits: usize, - /// The decimal point and fractional digits, empty for an integer. - tail: []const u8, -}; - -fn splitDecimalText(text: []const u8) DecimalParts { - const sign_len: usize = if (text.len > 0 and (text[0] == '-' or text[0] == '+')) 1 else 0; - const dot = std.mem.indexOfScalar(u8, text, '.') orelse text.len; - return .{ - .sign_len = sign_len, - .int_digits = dot - sign_len, - .tail = text[dot..], - }; -} - -/// Count digits before the decimal point, ignoring sign. -fn integerDigitCount(text: []const u8) usize { - return splitDecimalText(text).int_digits; -} - pub const NumberDisplay = struct { /// Human-readable form, with comma grouping. display: []const u8, @@ -283,44 +266,6 @@ pub fn formatMoney(buf: []u8, value: f64) ?[]const u8 { return formatAmount(buf, value, 2); } -/// Bytes `writeGroupedDecimal` will produce for `text`. Equal to `text.len` when -/// there is nothing to group, which callers use to skip the copy entirely. -/// -/// Public because the TUI groups partially typed input, where reformatting -/// through an f64 would discard what the user typed (trailing zeros, a lone -/// decimal point). -pub fn groupedDecimalLen(text: []const u8) usize { - // Text already in scientific notation has no long integer part to group, and - // inserting commas around an exponent would only corrupt it. - if (std.mem.indexOfAny(u8, text, "eE") != null) return text.len; - const int_digits = integerDigitCount(text); - if (int_digits <= 3) return text.len; - return text.len + (int_digits - 1) / 3; -} - -/// Copy `text` into `dest` with commas grouping the integer part. `dest` must be -/// at least `groupedDecimalLen(text)` bytes and must not overlap `text`. -pub fn writeGroupedDecimal(dest: []u8, text: []const u8) usize { - const parts = splitDecimalText(text); - const start = parts.sign_len; - const int_digits = parts.int_digits; - - @memcpy(dest[0..start], text[0..start]); - var w: usize = start; - - var i: usize = 0; - while (i < int_digits) : (i += 1) { - if (i > 0 and (int_digits - i) % 3 == 0) { - dest[w] = ','; - w += 1; - } - dest[w] = text[start + i]; - w += 1; - } - @memcpy(dest[w..][0..parts.tail.len], parts.tail); - return w + parts.tail.len; -} - /// The width to print the hex, octal and binary rows at, for a standard-mode value /// that has no configured width (FR-1.9). /// @@ -345,274 +290,26 @@ pub fn displayWidthFor(value: u128) BitWidth { }; } -/// Format an integer for programmer mode hex display. -/// Display: "FF FF FF FF" (space per byte), byte order per `endian`. -/// Raw: "0xFFFFFFFF" (no separators, canonical MSB-first value regardless of -/// endian, since raw is the number itself and copies to the clipboard as such). -pub fn formatHex(buf: []u8, value: u128, bit_width: BitWidth, endian: Endianness) FormattedValue { - const width = bit_width.bits(); - const hex_digits: usize = @as(usize, width) / 4; - const num_bytes: usize = @as(usize, width) / 8; - - // Write raw first: "0x" + hex digits, canonical MSB-first. - buf[0] = '0'; - buf[1] = 'x'; - var pos: usize = 2; - var i: usize = 0; - while (i < hex_digits) : (i += 1) { - const shift_amt: u7 = @intCast((hex_digits - 1 - i) * 4); - const nibble: u4 = @intCast((value >> shift_amt) & 0xF); - buf[pos] = hexDigit(nibble); - pos += 1; - } - const raw = buf[0..pos]; - - // Write display after raw: one byte at a time in `endian` order, - // space-separated. Only the byte order swaps; nibble order within a - // byte is always high-then-low. - const display_start = pos; - var k: usize = 0; - while (k < num_bytes) : (k += 1) { - if (k > 0) { - buf[pos] = ' '; - pos += 1; - } - const byte = byteAt(value, num_bytes, k, endian); - buf[pos] = hexDigit(@intCast(byte >> 4)); - pos += 1; - buf[pos] = hexDigit(@intCast(byte & 0xF)); - pos += 1; - } - const display = buf[display_start..pos]; - - return .{ .display = display, .raw = raw }; -} - -/// Return the byte to display at position `k` (0 = leftmost) for a value of -/// `num_bytes` width, honoring endianness. Big-endian shows the most -/// significant byte first; little-endian shows the least significant first. -fn byteAt(value: u128, num_bytes: usize, k: usize, endian: Endianness) u8 { - const msb_index: usize = if (endian == .big) k else num_bytes - 1 - k; - const shift_amt: u7 = @intCast((num_bytes - 1 - msb_index) * 8); - return @intCast((value >> shift_amt) & 0xFF); -} - -/// Format an integer for programmer mode binary display. -/// Display: "1111 0000 1010 1100" (space per nibble) -/// Raw: "0b1111000010101100" (no separators) -pub fn formatBinary(buf: []u8, value: u128, bit_width: BitWidth) FormattedValue { - const width: usize = bit_width.bits(); - - // Write raw: "0b" + binary digits - buf[0] = '0'; - buf[1] = 'b'; - var pos: usize = 2; - var i: usize = 0; - while (i < width) : (i += 1) { - const shift_amt: u7 = @intCast(width - 1 - i); - const bit: u8 = @intCast((value >> shift_amt) & 1); - buf[pos] = '0' + bit; - pos += 1; - } - const raw = buf[0..pos]; - - // Write display: binary digits with space per nibble - const display_start = pos; - i = 0; - while (i < width) : (i += 1) { - if (i > 0 and i % 4 == 0) { - buf[pos] = ' '; - pos += 1; - } - const shift_amt: u7 = @intCast(width - 1 - i); - const bit: u8 = @intCast((value >> shift_amt) & 1); - buf[pos] = '0' + bit; - pos += 1; - } - const display = buf[display_start..pos]; - - return .{ .display = display, .raw = raw }; -} - -/// Format an integer for programmer mode octal display. -/// Display: "0 000 000 000 777" (space per 3-digit group, zero-padded to full width) -/// Raw: "0o0000000000777" (no separators, with prefix) -pub fn formatOctal(buf: []u8, value: u128, bit_width: BitWidth) FormattedValue { - const total_digits: usize = (@as(usize, bit_width.bits()) + 2) / 3; - - // Write raw: "0o" + zero-padded octal digits - buf[0] = '0'; - buf[1] = 'o'; - var pos: usize = 2; - var i: usize = 0; - while (i < total_digits) : (i += 1) { - const shift_amt: u7 = @intCast((total_digits - 1 - i) * 3); - const digit: u8 = @intCast((value >> shift_amt) & 0x7); - buf[pos] = '0' + digit; - pos += 1; - } - const raw = buf[0..pos]; - - // Write display: digits with space every 3 from the right (no prefix) - const display_start = pos; - const first_group: usize = if (total_digits % 3 == 0) 3 else total_digits % 3; - - i = 0; - while (i < total_digits) : (i += 1) { - if (i > 0 and (i == first_group or (i > first_group and (i - first_group) % 3 == 0))) { - buf[pos] = ' '; - pos += 1; - } - const shift_amt: u7 = @intCast((total_digits - 1 - i) * 3); - const digit: u8 = @intCast((value >> shift_amt) & 0x7); - buf[pos] = '0' + digit; - pos += 1; - } - const display = buf[display_start..pos]; - - return .{ .display = display, .raw = raw }; -} -/// Format an integer's bytes as an ASCII representation for programmer mode. -/// One glyph per byte in `endian` order (matching the hex byte view). -/// Printable bytes (0x20-0x7E) render as themselves; every other byte renders -/// as '.', matching the convention used by `xxd` and similar hex dumps. -/// -/// Display: " . . . a s c i i" - each byte is a leading space + glyph, -/// space-separated, so the row lines up column-for-column under formatHex. -/// Raw: "...ascii" - contiguous glyphs, clipboard-friendly. -pub fn formatAscii(buf: []u8, value: u128, bit_width: BitWidth, endian: Endianness) FormattedValue { - const num_bytes: usize = @as(usize, bit_width.bits()) / 8; - - // Raw: contiguous glyphs, in display (endian) order. - var pos: usize = 0; - var k: usize = 0; - while (k < num_bytes) : (k += 1) { - buf[pos] = asciiGlyph(byteAt(value, num_bytes, k, endian)); - pos += 1; - } - const raw = buf[0..pos]; - - // Display: " c" per byte, space-separated, so glyph k sits under the - // right hex digit of byte k in the hex row. - const display_start = pos; - k = 0; - while (k < num_bytes) : (k += 1) { - if (k > 0) { - buf[pos] = ' '; - pos += 1; - } - buf[pos] = ' '; - pos += 1; - buf[pos] = asciiGlyph(byteAt(value, num_bytes, k, endian)); - pos += 1; - } - const display = buf[display_start..pos]; - - return .{ .display = display, .raw = raw }; -} - -pub fn formatDecimalUnsigned(buf: []u8, value: u128) FormattedValue { - const raw_len = writeUnsignedInt(buf, value); - const raw = buf[0..raw_len]; - - const display_start = raw_len; - const display_len = writeUnsignedWithCommas(buf[display_start..], value); - const display = buf[display_start..][0..display_len]; - - return .{ .display = display, .raw = raw }; -} - -/// Format a signed integer as decimal for programmer mode. -/// Display: "-1" or "4,294,967,295" -/// Raw: same without commas -pub fn formatDecimalSigned(buf: []u8, value: i128) FormattedValue { - const raw_len = writeSignedInt(buf, value); - const raw = buf[0..raw_len]; - - const display_start = raw_len; - const display_len = writeDecimalWithCommas(buf[display_start..], value); - const display = buf[display_start..][0..display_len]; - - return .{ .display = display, .raw = raw }; -} - -// -- Internal helpers -- - -fn hexDigit(nibble: u4) u8 { - if (nibble < 10) return '0' + @as(u8, nibble); - return 'A' + @as(u8, nibble) - 10; -} - -/// Map a byte to its printable glyph, or '.' if outside the printable ASCII -/// range (0x20-0x7E), matching the `xxd` hex-dump convention. -fn asciiGlyph(byte: u8) u8 { - if (byte >= 0x20 and byte <= 0x7E) return byte; - return '.'; -} - -fn writeUnsignedInt(buf: []u8, value: u128) usize { - if (value == 0) { - buf[0] = '0'; - return 1; - } - var digits: [39]u8 = undefined; - var count: usize = 0; - var v = value; - while (v > 0) : (v /= 10) { - digits[count] = @intCast(v % 10); - count += 1; - } - var pos: usize = 0; - var i: usize = count; - while (i > 0) { - i -= 1; - buf[pos] = '0' + digits[i]; - pos += 1; - } - return pos; -} - -fn writeSignedInt(buf: []u8, value: i128) usize { - if (value < 0) { - buf[0] = '-'; - return 1 + writeUnsignedInt(buf[1..], absoluteValue(value)); - } - return writeUnsignedInt(buf, @intCast(value)); -} - -/// Magnitude of a signed 128-bit value as an unsigned one. -/// -/// `-value` overflows for `minInt(i128)`, which panics in Debug and ReleaseSafe. -/// That value is reachable: a 128-bit programmer-mode word with only the sign bit -/// set formats through here. Negating in the unsigned domain has no such edge. -fn absoluteValue(value: i128) u128 { - const bits: u128 = @bitCast(value); - return if (value < 0) ~bits +% 1 else bits; -} - -/// Write `value` with comma grouping, reusing the text grouper. -/// -/// This used to be a second grouping implementation: it built the digits in reverse -/// and inserted separators itself, so the codebase had two places that knew what a -/// thousands group is. Writing the plain digits and then grouping them keeps one. -fn writeUnsignedWithCommas(buf: []u8, value: u128) usize { - var plain: [40]u8 = undefined; - const digits = plain[0..writeUnsignedInt(&plain, value)]; - return writeGroupedDecimal(buf, digits); -} - -fn writeDecimalWithCommas(buf: []u8, value: i128) usize { - if (value < 0) { - buf[0] = '-'; - return 1 + writeUnsignedWithCommas(buf[1..], absoluteValue(value)); - } - return writeUnsignedWithCommas(buf, @intCast(value)); -} - // -- Tests -- const testing = std.testing; +test "formatFloat: a buffer too small degrades instead of overrunning" { + // Both halves are written through a fixed writer, so a short buffer is an error + // the function handles rather than a walk off the end. + var no_room: [4]u8 = undefined; + const failed = formatFloat(&no_room, 1234567); + try testing.expectEqualStrings("ERR", failed.display); + try testing.expectEqualStrings("ERR", failed.raw); + + // Room for the digits but not for a second, grouped copy: display falls back to + // the ungrouped text rather than being truncated. + var tight: [9]u8 = undefined; + const partial = formatFloat(&tight, 1234567); + try testing.expectEqualStrings("1234567", partial.raw); + try testing.expectEqualStrings("1234567", partial.display); +} + test "formatFloat: integer value" { var buf: [256]u8 = undefined; const result = formatFloat(&buf, 42.0); @@ -666,143 +363,14 @@ test "displayWidthFor: the chosen width holds the value and sizes the rows" { for ([_]u128{ 0, 1, 255, 256, 0xFFFF, 0x1_0000, std.math.maxInt(u64), std.math.maxInt(u128) }) |value| { const bw = displayWidthFor(value); try testing.expectEqual(value, value & bw.mask()); - const hex = formatHex(&buf, value, bw, .big); + const int: Integer = .{ .raw = value, .width = bw, .signedness = .unsigned }; + const hex = try int.as(.hex, .{}).render(&buf); // Two hex digits per byte, plus one space between bytes. - const bytes = bw.bits() / 8; - try testing.expectEqual(@as(usize, bytes * 2 + bytes - 1), hex.display.len); + const bytes: usize = bw.bits() / 8; + try testing.expectEqual(bytes * 3 - 1, hex.len); } } -test "formatHex: 8-bit" { - var buf: [256]u8 = undefined; - const result = formatHex(&buf, 0xFF, .bits8, .big); - try testing.expectEqualStrings("FF", result.display); - try testing.expectEqualStrings("0xFF", result.raw); -} - -test "formatHex: 16-bit" { - var buf: [256]u8 = undefined; - const result = formatHex(&buf, 0xABCD, .bits16, .big); - try testing.expectEqualStrings("AB CD", result.display); - try testing.expectEqualStrings("0xABCD", result.raw); -} - -test "formatHex: 32-bit with grouping" { - var buf: [256]u8 = undefined; - const result = formatHex(&buf, 0xDEADBEEF, .bits32, .big); - try testing.expectEqualStrings("DE AD BE EF", result.display); - try testing.expectEqualStrings("0xDEADBEEF", result.raw); -} - -test "formatHex: 64-bit with grouping" { - var buf: [256]u8 = undefined; - const result = formatHex(&buf, 0xDEAD_BEEF_CAFE_BABE, .bits64, .big); - try testing.expectEqualStrings("DE AD BE EF CA FE BA BE", result.display); - try testing.expectEqualStrings("0xDEADBEEFCAFEBABE", result.raw); -} - -test "formatHex: zero 32-bit" { - var buf: [256]u8 = undefined; - const result = formatHex(&buf, 0, .bits32, .big); - try testing.expectEqualStrings("00 00 00 00", result.display); - try testing.expectEqualStrings("0x00000000", result.raw); -} - -test "formatHex: little-endian reverses byte display" { - var buf: [256]u8 = undefined; - const result = formatHex(&buf, 0xDEADBEEF, .bits32, .little); - // Bytes shown least-significant first (x86 memory order) - try testing.expectEqualStrings("EF BE AD DE", result.display); - // Raw stays the canonical number - try testing.expectEqualStrings("0xDEADBEEF", result.raw); -} - -test "formatHex: little-endian 64-bit" { - var buf: [256]u8 = undefined; - const result = formatHex(&buf, 0xDEAD_BEEF_CAFE_BABE, .bits64, .little); - try testing.expectEqualStrings("BE BA FE CA EF BE AD DE", result.display); - try testing.expectEqualStrings("0xDEADBEEFCAFEBABE", result.raw); -} - -test "formatHex: little-endian 8-bit is identical (single byte)" { - var buf: [256]u8 = undefined; - const result = formatHex(&buf, 0xFF, .bits8, .little); - try testing.expectEqualStrings("FF", result.display); -} - -test "formatBinary: 8-bit" { - var buf: [256]u8 = undefined; - const result = formatBinary(&buf, 0xFF, .bits8); - try testing.expectEqualStrings("1111 1111", result.display); - try testing.expectEqualStrings("0b11111111", result.raw); -} - -test "formatBinary: 8-bit mixed" { - var buf: [256]u8 = undefined; - const result = formatBinary(&buf, 0xA5, .bits8); - try testing.expectEqualStrings("1010 0101", result.display); - try testing.expectEqualStrings("0b10100101", result.raw); -} - -test "formatBinary: 16-bit" { - var buf: [256]u8 = undefined; - const result = formatBinary(&buf, 0x000F, .bits16); - try testing.expectEqualStrings("0000 0000 0000 1111", result.display); - try testing.expectEqualStrings("0b0000000000001111", result.raw); -} - -test "formatOctal: simple" { - var buf: [256]u8 = undefined; - const result = formatOctal(&buf, 511, .bits16); - // 16-bit: ceil(16/3) = 6 digits, 511 = 777 - try testing.expectEqualStrings("000 777", result.display); - try testing.expectEqualStrings("0o000777", result.raw); -} - -test "formatOctal: large with grouping" { - var buf: [256]u8 = undefined; - const result = formatOctal(&buf, 0xFFFF_FFFF, .bits32); - // 32-bit: ceil(32/3) = 11 digits - try testing.expectEqualStrings("37 777 777 777", result.display); - try testing.expectEqualStrings("0o37777777777", result.raw); -} - -test "formatOctal: zero" { - var buf: [256]u8 = undefined; - const result = formatOctal(&buf, 0, .bits8); - // 8-bit: ceil(8/3) = 3 digits - try testing.expectEqualStrings("000", result.display); - try testing.expectEqualStrings("0o000", result.raw); -} - -test "formatDecimalUnsigned: simple" { - var buf: [256]u8 = undefined; - const result = formatDecimalUnsigned(&buf, 255); - try testing.expectEqualStrings("255", result.display); - try testing.expectEqualStrings("255", result.raw); -} - -test "formatDecimalUnsigned: large" { - var buf: [256]u8 = undefined; - const result = formatDecimalUnsigned(&buf, 4294967295); - try testing.expectEqualStrings("4,294,967,295", result.display); - try testing.expectEqualStrings("4294967295", result.raw); -} - -test "formatDecimalSigned: negative" { - var buf: [256]u8 = undefined; - const result = formatDecimalSigned(&buf, -1); - try testing.expectEqualStrings("-1", result.display); - try testing.expectEqualStrings("-1", result.raw); -} - -test "formatDecimalSigned: negative large" { - var buf: [256]u8 = undefined; - const result = formatDecimalSigned(&buf, -1234567); - try testing.expectEqualStrings("-1,234,567", result.display); - try testing.expectEqualStrings("-1234567", result.raw); -} - test "formatFloat: very large number uses scientific notation" { var buf: [256]u8 = undefined; const result = formatFloat(&buf, 1.5e16); @@ -824,55 +392,6 @@ test "formatFloat: regular float" { try testing.expect(std.mem.indexOf(u8, result.display, "3.14") != null); } -test "formatAscii: printable word 64-bit" { - var buf: [128]u8 = undefined; - // 0x0000006173636969 -> bytes 00 00 00 61 73 63 69 69 -> "...ascii" - const result = formatAscii(&buf, 0x0000006173636969, .bits64, .big); - try testing.expectEqualStrings("...ascii", result.raw); - // Display aligns one glyph per byte, space-separated - try testing.expectEqualStrings(" . . . a s c i i", result.display); -} - -test "formatAscii: little-endian reverses byte order" { - var buf: [128]u8 = undefined; - // Same value, little-endian: bytes reversed -> "iicsa..." with trailing dots - const result = formatAscii(&buf, 0x0000006173636969, .bits64, .little); - try testing.expectEqualStrings("iicsa...", result.raw); -} - -test "formatAscii: 8-bit printable" { - var buf: [128]u8 = undefined; - const result = formatAscii(&buf, 0x41, .bits8, .big); - try testing.expectEqualStrings("A", result.raw); - try testing.expectEqualStrings(" A", result.display); -} - -test "formatAscii: 8-bit non-printable becomes dot" { - var buf: [128]u8 = undefined; - const result = formatAscii(&buf, 0x00, .bits8, .big); - try testing.expectEqualStrings(".", result.raw); -} - -test "formatAscii: control and high bytes become dots" { - var buf: [128]u8 = undefined; - // 0x1B (ESC) and 0xFF are both non-printable - const result = formatAscii(&buf, 0x1BFF, .bits16, .big); - try testing.expectEqualStrings("..", result.raw); -} - -test "formatAscii: boundary bytes 0x20 and 0x7E are printable" { - var buf: [128]u8 = undefined; - // 0x20 = space, 0x7E = '~' - const result = formatAscii(&buf, 0x207E, .bits16, .big); - try testing.expectEqualStrings(" ~", result.raw); -} - -test "formatAscii: 0x7F is non-printable" { - var buf: [128]u8 = undefined; - const result = formatAscii(&buf, 0x7F, .bits8, .big); - try testing.expectEqualStrings(".", result.raw); -} - fn hasChar(s: []const u8, c: u8) bool { return std.mem.indexOfScalar(u8, s, c) != null; } @@ -1158,10 +677,10 @@ test "abbreviated huge values go through the same renderer as tiny ones" { } test "integerDigitCount ignores sign and fraction" { - try testing.expectEqual(@as(usize, 3), integerDigitCount("123")); - try testing.expectEqual(@as(usize, 3), integerDigitCount("-123")); - try testing.expectEqual(@as(usize, 3), integerDigitCount("123.456")); - try testing.expectEqual(@as(usize, 1), integerDigitCount("0.5")); + try testing.expectEqual(@as(usize, 3), grouping.integerDigitCount("123")); + try testing.expectEqual(@as(usize, 3), grouping.integerDigitCount("-123")); + try testing.expectEqual(@as(usize, 3), grouping.integerDigitCount("123.456")); + try testing.expectEqual(@as(usize, 1), grouping.integerDigitCount("0.5")); } // -- Grouping of values with a fractional part -- @@ -1353,42 +872,6 @@ test "isZeroText: recognises every spelling of zero" { try testing.expect(!isZeroText("10.00")); try testing.expect(!isZeroText("-0.5")); } - -// -- The most negative 128-bit value -- -// -// Formatting used to negate the value to get its magnitude, which overflows for -// minInt(i128) and panics. It is reachable: a 128-bit programmer word with only -// the sign bit set formats through here, so cycling the TUI to 128 bits and -// entering 2^127 killed the process. - -test "absoluteValue: the most negative value has a magnitude" { - try testing.expectEqual(@as(u128, 1 << 127), absoluteValue(std.math.minInt(i128))); - try testing.expectEqual(@as(u128, 5), absoluteValue(-5)); - try testing.expectEqual(@as(u128, 5), absoluteValue(5)); - try testing.expectEqual(@as(u128, 0), absoluteValue(0)); - try testing.expectEqual(@as(u128, std.math.maxInt(i128)), absoluteValue(std.math.maxInt(i128))); -} - -test "formatDecimalSigned: minInt(i128) formats instead of panicking" { - var buf: [256]u8 = undefined; - const result = formatDecimalSigned(&buf, std.math.minInt(i128)); - // -2^127 - try testing.expectEqualStrings("-170141183460469231731687303715884105728", result.raw); - try testing.expectEqualStrings("-170,141,183,460,469,231,731,687,303,715,884,105,728", result.display); -} - -test "formatDecimalSigned: the rest of the signed range still formats" { - var buf: [256]u8 = undefined; - try testing.expectEqualStrings("-1", formatDecimalSigned(&buf, -1).raw); - try testing.expectEqualStrings("-1,234", formatDecimalSigned(&buf, -1234).display); - try testing.expectEqualStrings("0", formatDecimalSigned(&buf, 0).raw); - try testing.expectEqualStrings( - "170,141,183,460,469,231,731,687,303,715,884,105,727", - formatDecimalSigned(&buf, std.math.maxInt(i128)).display, - ); -} - -// -- One amount formatter -- // // This logic existed three times: character for character in src/main.zig and // src/tui/financial.zig, each reimplementing the comma grouping that already lived @@ -1456,85 +939,3 @@ test "formatMoney: agrees with the grouping used for ordinary results" { // Same separators, differing only in the fixed decimal places. try testing.expect(std.mem.startsWith(u8, as_money, as_value)); } - -// -- One grouping implementation -- -// -// Grouping existed twice: once over text (groupedDecimalLen/writeGroupedDecimal) and -// once over integers (writeUnsignedWithCommas built digits in reverse and inserted -// its own separators). The integer path now writes plain digits and groups them, so -// there is a single definition of what a thousands group is. - -test "integer and text grouping agree on every width" { - var integer_buf: [512]u8 = undefined; - var text_buf: [512]u8 = undefined; - var plain_buf: [64]u8 = undefined; - - const values = [_]u128{ - 0, 1, - 9, 10, - 99, 100, - 999, 1000, - 1001, 12345, - 999999, 1000000, - 123456789, std.math.maxInt(u64), - std.math.maxInt(u128), 4294967295, - 3735928559, 1180591620717411303424, - }; - for (values) |value| { - const grouped = integer_buf[0..writeUnsignedWithCommas(&integer_buf, value)]; - - // Independently: render the digits, then group the text. - const plain = plain_buf[0..writeUnsignedInt(&plain_buf, value)]; - const via_text = text_buf[0..writeGroupedDecimal(&text_buf, plain)]; - - try testing.expectEqualStrings(via_text, grouped); - } -} - -test "integer grouping: known shapes" { - var buf: [64]u8 = undefined; - try testing.expectEqualStrings("0", buf[0..writeUnsignedWithCommas(&buf, 0)]); - try testing.expectEqualStrings("100", buf[0..writeUnsignedWithCommas(&buf, 100)]); - try testing.expectEqualStrings("1,000", buf[0..writeUnsignedWithCommas(&buf, 1000)]); - try testing.expectEqualStrings("4,294,967,295", buf[0..writeUnsignedWithCommas(&buf, 4294967295)]); - try testing.expectEqualStrings( - "340,282,366,920,938,463,463,374,607,431,768,211,455", - buf[0..writeUnsignedWithCommas(&buf, std.math.maxInt(u128))], - ); -} - -test "signed integer grouping keeps the sign outside the groups" { - var buf: [128]u8 = undefined; - try testing.expectEqualStrings("-1,234", buf[0..writeDecimalWithCommas(&buf, -1234)]); - try testing.expectEqualStrings("-1", buf[0..writeDecimalWithCommas(&buf, -1)]); - try testing.expectEqualStrings("0", buf[0..writeDecimalWithCommas(&buf, 0)]); - try testing.expectEqualStrings( - "-170,141,183,460,469,231,731,687,303,715,884,105,728", - buf[0..writeDecimalWithCommas(&buf, std.math.minInt(i128))], - ); -} - -test "splitDecimalText: one place that takes decimal text apart" { - const unsigned = splitDecimalText("1234.56"); - try testing.expectEqual(@as(usize, 0), unsigned.sign_len); - try testing.expectEqual(@as(usize, 4), unsigned.int_digits); - try testing.expectEqualStrings(".56", unsigned.tail); - - const negative = splitDecimalText("-1234.56"); - try testing.expectEqual(@as(usize, 1), negative.sign_len); - try testing.expectEqual(@as(usize, 4), negative.int_digits); - try testing.expectEqualStrings(".56", negative.tail); - - const whole = splitDecimalText("-70"); - try testing.expectEqual(@as(usize, 1), whole.sign_len); - try testing.expectEqual(@as(usize, 2), whole.int_digits); - try testing.expectEqualStrings("", whole.tail); - - const explicit_plus = splitDecimalText("+5"); - try testing.expectEqual(@as(usize, 1), explicit_plus.sign_len); - try testing.expectEqual(@as(usize, 1), explicit_plus.int_digits); - - const empty = splitDecimalText(""); - try testing.expectEqual(@as(usize, 0), empty.sign_len); - try testing.expectEqual(@as(usize, 0), empty.int_digits); -} diff --git a/engine/src/grouping.zig b/engine/src/grouping.zig new file mode 100644 index 0000000..2982288 --- /dev/null +++ b/engine/src/grouping.zig @@ -0,0 +1,186 @@ +//! The thousands rule, in one place. +//! +//! Both display paths group digits: `Integer` groups the decimal readings of a +//! fixed-width value, and `formatter` groups floats and exact `Number`s. Those two +//! each worked out where a sign ended and where the fraction began, and disagreed, so +//! `-1234.56` grouped differently from `1234.56`. This is the shared answer. +//! +//! Grouping operates on decimal text rather than on a number, because the callers do +//! not agree on what a number is: one has a `u128` and a width, another an `f64`, and +//! a third the exact decimal expansion of a rational. All three can produce digits. + +const std = @import("std"); +const Writer = std.Io.Writer; + +/// True when `text` is rendered numeric text: an optional sign, decimal digits with +/// at most one point, and an optional exponent. At least one digit before the +/// exponent. No separators, no spaces, no hex, no `inf` or `nan`. +/// +/// This is the precondition of everything else here, asserted rather than assumed, +/// and it is exported so a caller holding text of uncertain shape can ask first. +/// Every caller inside the engine passes digits it has just rendered; the TUI groups +/// what the user has typed, and used to gate that on `std.fmt.parseFloat` accepting +/// it, which is a wider language: `1_000` and `0x1p4` parse as numbers and came out +/// of here as `1_,000` and `0x,1p4`. +pub fn isNumericText(text: []const u8) bool { + var i: usize = 0; + if (i < text.len and (text[i] == '-' or text[i] == '+')) i += 1; + + var mantissa_digits: usize = 0; + while (i < text.len and std.ascii.isDigit(text[i])) : (i += 1) mantissa_digits += 1; + if (i < text.len and text[i] == '.') { + i += 1; + while (i < text.len and std.ascii.isDigit(text[i])) : (i += 1) mantissa_digits += 1; + } + if (mantissa_digits == 0) return false; + if (i == text.len) return true; + + // An exponent, if there is anything left. + if (text[i] != 'e' and text[i] != 'E') return false; + i += 1; + if (i < text.len and (text[i] == '-' or text[i] == '+')) i += 1; + var exponent_digits: usize = 0; + while (i < text.len and std.ascii.isDigit(text[i])) : (i += 1) exponent_digits += 1; + return exponent_digits > 0 and i == text.len; +} + +/// Write `text` with commas grouping its integer part, leaving any sign and any +/// fractional part alone. +/// +/// `text` must be numeric text without an exponent. Grouping scientific notation +/// would put commas inside the mantissa, so callers that might hold either ask +/// `lengthOf` first: it reports nothing to group for an exponent, which is the +/// signal to print the text as it stands. +pub fn print(w: *Writer, text: []const u8) Writer.Error!void { + std.debug.assert(isNumericText(text)); + std.debug.assert(std.mem.indexOfAny(u8, text, "eE") == null); + + const parts = split(text); + + try w.writeAll(text[0..parts.sign_len]); + for (0..parts.int_digits) |i| { + if (i > 0 and (parts.int_digits - i) % 3 == 0) try w.writeByte(','); + try w.writeByte(text[parts.sign_len + i]); + } + try w.writeAll(parts.tail); +} + +/// Bytes `print` will write for `text`. Equal to `text.len` when there is nothing to +/// group, which callers use to skip the copy entirely. +pub fn lengthOf(text: []const u8) usize { + std.debug.assert(isNumericText(text)); + // Scientific notation has no long integer part to group, and commas around an + // exponent would corrupt it. + if (std.mem.indexOfAny(u8, text, "eE") != null) return text.len; + const parts = split(text); + if (parts.int_digits <= 3) return text.len; + return text.len + (parts.int_digits - 1) / 3; +} + +/// Digits before the decimal point, ignoring any sign. The formatter uses this to +/// decide when an exact value is too wide to show in full. +pub fn integerDigitCount(text: []const u8) usize { + std.debug.assert(isNumericText(text)); + return split(text).int_digits; +} + +/// Decimal text taken apart: the sign, the integer digits, and everything from the +/// decimal point onward. +const Parts = struct { + sign_len: usize, + int_digits: usize, + tail: []const u8, +}; + +fn split(text: []const u8) Parts { + const sign_len: usize = if (text.len > 0 and (text[0] == '-' or text[0] == '+')) 1 else 0; + const dot = std.mem.indexOfScalar(u8, text, '.') orelse text.len; + return .{ + .sign_len = sign_len, + .int_digits = dot - sign_len, + .tail = text[dot..], + }; +} + +// -- Tests -- + +const testing = std.testing; + +fn grouped(buf: []u8, text: []const u8) ![]const u8 { + var w = Writer.fixed(buf); + try print(&w, text); + return w.buffered(); +} + +test "commas every three digits, sign and fraction untouched" { + var buf: [64]u8 = undefined; + try testing.expectEqualStrings("0", try grouped(&buf, "0")); + try testing.expectEqualStrings("100", try grouped(&buf, "100")); + try testing.expectEqualStrings("1,000", try grouped(&buf, "1000")); + try testing.expectEqualStrings("1,234,567", try grouped(&buf, "1234567")); + try testing.expectEqualStrings("-1,234.56", try grouped(&buf, "-1234.56")); + try testing.expectEqualStrings("231,677.04", try grouped(&buf, "231677.04")); + // A fractional part is never grouped, however long it is. + try testing.expectEqualStrings("0.123456789", try grouped(&buf, "0.123456789")); +} + +test "lengthOf: agrees with what print writes" { + var buf: [128]u8 = undefined; + for ([_][]const u8{ + "0", "12", "123", "1234", + "12345", "123456", "1234567", "-1234567", + "1234.5678", "-99.9", "+1000", "1000000000", + "0.5", "-1000000", "1234567890", "0.0000001", + }) |text| { + try testing.expectEqual(lengthOf(text), (try grouped(&buf, text)).len); + } +} + +test "lengthOf: scientific notation is left alone" { + // Commas around an exponent would corrupt it, so there is nothing to group. + try testing.expectEqual(@as(usize, 7), lengthOf("1.5e300")); + try testing.expectEqual(@as(usize, 7), lengthOf("-2.5E-7")); +} + +test "print: a full buffer is an error, not a panic" { + // The old buffer-based version indexed past the end. + var tiny: [3]u8 = undefined; + var w = Writer.fixed(&tiny); + try testing.expectError(error.WriteFailed, print(&w, "1234567")); +} + +test "integerDigitCount ignores sign and fraction" { + try testing.expectEqual(@as(usize, 3), integerDigitCount("123")); + try testing.expectEqual(@as(usize, 3), integerDigitCount("-123")); + try testing.expectEqual(@as(usize, 3), integerDigitCount("123.456")); + try testing.expectEqual(@as(usize, 1), integerDigitCount("0.5")); + try testing.expectEqual(@as(usize, 1), integerDigitCount("+5")); + try testing.expectEqual(@as(usize, 0), integerDigitCount(".5")); +} + +test "isNumericText: what this module will group" { + for ([_][]const u8{ + "0", "123", "-123", "+5", + "0.5", ".5", "5.", "-1234.56", + "1e5", "1E5", "1.5e300", "-2.5E-7", + "1e+20", "0.0", "-0", "340282366920938463463374607431768211455", + }) |text| { + try testing.expect(isNumericText(text)); + } +} + +test "isNumericText: what it will not" { + // The first two are the bug this predicate exists for: `std.fmt.parseFloat` + // accepts both, so the TUI treated them as plain numbers and grouped them into + // "1_,000" and "0x,1p4". + for ([_][]const u8{ + "1_000", "0x1p4", "0x10", "0b1010", + "", "-", "+", ".", + "-.", ".-5", "abcdefg", "1.2.3", + " 12", "12 ", "1,000", "inf", + "nan", "1e", "1e+", "5e5e5", + "--5", "1.5e3.2", + }) |text| { + try testing.expect(!isNumericText(text)); + } +} diff --git a/engine/src/programmer.zig b/engine/src/programmer.zig index 6096ad4..61d29a7 100644 --- a/engine/src/programmer.zig +++ b/engine/src/programmer.zig @@ -11,8 +11,11 @@ const ast = @import("ast.zig"); const Expr = ast.Expr; const BinaryOp = ast.BinaryOp; const Integer = @import("Integer.zig"); -const IntType = Integer.IntType; -const Endianness = std.builtin.Endian; +const BitWidth = Integer.BitWidth; +const Signedness = Integer.Signedness; +const parser_mod = @import("parser.zig"); +const Parser = parser_mod.Parser; +const bitwise = @import("bitwise.zig"); /// What programmer-mode evaluation can fail with: the parse, the fixed-width /// operators, and its own arithmetic and name errors. @@ -25,34 +28,36 @@ pub const Error = error{ UnknownFunction, UnknownVariable, } || parser_mod.Error || bitwise.Error; -const parser_mod = @import("parser.zig"); -const Parser = parser_mod.Parser; -const bitwise = @import("bitwise.zig"); /// What programmer mode needs to know: the integer type to compute in, and one /// display preference that does not affect arithmetic. pub const Config = struct { - int_type: IntType = .{}, + width: BitWidth = .bits64, + signedness: Signedness = .signed, /// Byte order for the HEX and ASCII rows only. Defaults to big-endian so the /// HEX row reads as the number itself (matching DEC/OCT/BIN); the little-endian /// view (x86 memory layout) is available via the toggle. - display_endian: Endianness = .big, + display_endian: std.builtin.Endian = .big, + + /// A value of the configured integer type. Every literal and every result goes + /// through here, so the width and signedness are attached once rather than + /// tracked alongside a bare pattern. + pub fn value(self: Config, raw: u128) Integer { + return .{ .raw = raw & self.width.mask(), .width = self.width, .signedness = self.signedness }; + } }; /// Evaluate an AST in programmer mode, producing an exact integer result. pub fn evalProgrammer(config: Config, expr: *const Expr) Error!Integer { - return .{ - .raw = try evalExpr(config, expr), - .int_type = config.int_type, - }; + return evalExpr(config, expr); } -/// Recursively evaluate an expression to a raw u128. -fn evalExpr(config: Config, expr: *const Expr) Error!u128 { +/// Recursively evaluate an expression. +fn evalExpr(config: Config, expr: *const Expr) Error!Integer { switch (expr.*) { .number => |n| { if (n.int_value) |int_val| { - return int_val & config.int_type.mask(); + return config.value(int_val); } // Float literal in programmer mode: truncate to integer. // (Number literals are always non-negative; unary minus is a @@ -61,23 +66,22 @@ fn evalExpr(config: Config, expr: *const Expr) Error!u128 { // The range check is not optional: `@intFromFloat` on an out-of-range // value is illegal behaviour, and `tally -p '1e40'` aborted the process // before this guard existed. - const value = n.float_value; - if (!std.math.isFinite(value) or value < 0) return Error.DomainError; - if (value >= 340282366920938463463374607431768211456.0) return Error.Overflow; - const val: u128 = @intFromFloat(value); - return val & config.int_type.mask(); + const float_value = n.float_value; + if (!std.math.isFinite(float_value) or float_value < 0) return Error.DomainError; + if (float_value >= 340282366920938463463374607431768211456.0) return Error.Overflow; + return config.value(@intFromFloat(float_value)); }, .string_literal => |text| { // Pack ASCII bytes into integer. // Big-endian packing: first char -> most significant used byte. - const max_bytes = @as(usize, config.int_type.bits()) / 8; + const max_bytes = @as(usize, config.width.bits()) / 8; if (text.len > max_bytes) return Error.Overflow; var result: u128 = 0; for (text) |byte| { if (byte > 0x7F) return Error.InvalidNumber; result = (result << 8) | byte; } - return result & config.int_type.mask(); + return config.value(result); }, .variable => { return Error.UnknownVariable; @@ -87,16 +91,15 @@ fn evalExpr(config: Config, expr: *const Expr) Error!u128 { }, .unary => |u| { const operand = try evalExpr(config, u.operand); - const domain = config.int_type; return switch (u.op) { - .negate => bitwise.negate(domain, operand), - .bitwise_not => bitwise.not(domain, operand), + .negate => bitwise.negate(operand), + .bitwise_not => bitwise.not(operand), }; }, .binary => |b| { const left = try evalExpr(config, b.left); const right = try evalExpr(config, b.right); - return evalBinaryOp(config, b.op, left, right); + return evalBinaryOp(b.op, left, right); }, .call => { // No function calls in programmer mode @@ -105,26 +108,28 @@ fn evalExpr(config: Config, expr: *const Expr) Error!u128 { } } -/// Evaluate a binary operation on two u128 values, masked to bit width. +/// Evaluate a binary operation on two values of the configured type. /// /// The bitwise operators, shifts and rotations are not here: they live in /// `bitwise.zig`, which standard mode uses too, so the two modes cannot drift /// apart again. What remains is the arithmetic, which genuinely differs between the /// modes: it wraps at the width here and is exact rational arithmetic there. -fn evalBinaryOp(config: Config, op: BinaryOp, left: u128, right: u128) Error!u128 { - const mask = config.int_type.mask(); +fn evalBinaryOp(op: BinaryOp, left_in: Integer, right_in: Integer) Error!Integer { + const mask = left_in.mask(); + const left = left_in.masked(); + const right = right_in.masked(); - const result: u128 = switch (op) { - .add => (left +% right) & mask, - .sub => (left -% right) & mask, - .mul => (left *% right) & mask, + const raw: u128 = switch (op) { + .add => left +% right, + .sub => left -% right, + .mul => left *% right, .div => blk: { if (right == 0) return Error.DivisionByZero; - break :blk (left / right) & mask; + break :blk left / right; }, .mod => blk: { if (right == 0) return Error.DivisionByZero; - break :blk (left % right) & mask; + break :blk left % right; }, .pow => blk: { // Integer exponentiation @@ -140,15 +145,14 @@ fn evalBinaryOp(config: Config, op: BinaryOp, left: u128, right: u128) Error!u12 // The fixed-width operators, in the shared implementation. `inline else` // resolves the operator at comptime, so an operator added to `BinaryOp` // that `bitwise.fromBinaryOp` does not know is a compile error here. - inline else => |fixed_op| try bitwise.apply( - config.int_type, + inline else => |fixed_op| return try bitwise.apply( comptime bitwise.fromBinaryOp(fixed_op).?, - left, - right, + left_in, + right_in, ), }; - return result; + return left_in.withRaw(raw); } /// High-level: parse and evaluate a string in programmer mode. @@ -169,12 +173,12 @@ fn testProg(source: []const u8) !Integer { return testProgWith(source, .{}); } -/// Tests care about the integer type, never about the display byte order, so they -/// pass the type directly. -fn testProgWith(source: []const u8, int_type: IntType) !Integer { +/// Tests care about the width and signedness, never about the display byte order, +/// so they pass a partial config. +fn testProgWith(source: []const u8, config: Config) !Integer { var arena = std.heap.ArenaAllocator.init(std.heap.page_allocator); defer _ = arena.deinit(); - return evalProgrammerString(arena.allocator(), source, .{ .int_type = int_type }); + return evalProgrammerString(arena.allocator(), source, config); } test "prog: simple number" { @@ -421,7 +425,7 @@ test "prog: ASCII literal in expression" { test "prog: ASCII literal overflow 8-bit" { var arena = std.heap.ArenaAllocator.init(std.heap.page_allocator); defer _ = arena.deinit(); - const result = evalProgrammerString(arena.allocator(), "'AB'", .{ .int_type = .{ .width = .bits8 } }); + const result = evalProgrammerString(arena.allocator(), "'AB'", .{ .width = .bits8 }); try testing.expectError(Error.Overflow, result); } diff --git a/src/main.zig b/src/main.zig index 3d7a66b..da74151 100644 --- a/src/main.zig +++ b/src/main.zig @@ -74,10 +74,10 @@ pub fn parseArgs(allocator: std.mem.Allocator, args: []const []const u8) ParsedA } else if (std.mem.eql(u8, arg, "--version")) { return .{ .output = .{ .text = "tally 0.1.0\n", .is_error = false } }; } else if (std.mem.eql(u8, arg, "--signed")) { - config.int_type.signedness = .signed; + config.signedness = .signed; mode = .programmer; } else if (std.mem.eql(u8, arg, "--unsigned")) { - config.int_type.signedness = .unsigned; + config.signedness = .unsigned; mode = .programmer; } else { // Flags that take a value, accepted as either `--bits 8` or `--bits=8`. @@ -85,7 +85,7 @@ pub fn parseArgs(allocator: std.mem.Allocator, args: []const []const u8) ParsedA .absent => {}, .missing => return .{ .output = .{ .text = bits_usage, .is_error = true } }, .value => |text| { - config.int_type.width = parseBitWidth(text) orelse + config.width = parseBitWidth(text) orelse return .{ .output = .{ .text = bits_usage, .is_error = true } }; // The width only means something in programmer mode: standard // mode is fixed at 64-bit signed. Asking for a width is asking @@ -482,60 +482,51 @@ fn isDisplayableInt(value: f64) bool { /// The decimal display is already computed; append hex/oct/bin. fn formatStandardMultiBase(buf: []u8, dec_display: []const u8, value: f64) CliResult { const int_val: u128 = @intFromFloat(value); - const bw = engine.formatter.displayWidthFor(int_val); - - var hex_buf: [256]u8 = undefined; - var oct_buf: [256]u8 = undefined; - var bin_buf: [512]u8 = undefined; - const hex = engine.formatter.formatHex(&hex_buf, int_val, bw, .big); - const oct = engine.formatter.formatOctal(&oct_buf, int_val, bw); - const bin = engine.formatter.formatBinary(&bin_buf, int_val, bw); + const int: engine.Integer = .{ + .raw = int_val, + .width = engine.formatter.displayWidthFor(int_val), + .signedness = .unsigned, + }; // dec_display points into buf, so copy it out before we overwrite buf. var dec_copy: [128]u8 = undefined; const dec_len = @min(dec_display.len, dec_copy.len); @memcpy(dec_copy[0..dec_len], dec_display[0..dec_len]); - const output = std.fmt.bufPrint(buf, - \\{s} - \\ hex: {s} - \\ oct: {s} - \\ bin: {s} - , .{ dec_copy[0..dec_len], hex.display, oct.display, bin.display }) catch { + // The rows print themselves, so nothing here needs a buffer per base. + var w = std.Io.Writer.fixed(buf); + w.print("{s}\n hex: {f}\n oct: {f}\n bin: {f}", .{ + dec_copy[0..dec_len], + int.as(.hex, .{}), + int.as(.octal, .{}), + int.as(.binary, .{}), + }) catch { return .{ .output = "error: buffer overflow\n", .is_error = true }; }; - return .{ .output = output, .is_error = false }; + return .{ .output = w.buffered(), .is_error = false }; } fn formatProgrammerResult(buf: []u8, result: engine.Integer, config: engine.programmer.Config) CliResult { - const value = result.unsignedValue(); - const signed = result.signedValue(); - - var hex_buf: [256]u8 = undefined; - var dec_buf: [256]u8 = undefined; - var sdec_buf: [256]u8 = undefined; - var oct_buf: [256]u8 = undefined; - var bin_buf: [512]u8 = undefined; - - const hex = engine.formatter.formatHex(&hex_buf, value, config.int_type.width, config.display_endian); - const dec = engine.formatter.formatDecimalUnsigned(&dec_buf, value); - const sdec = engine.formatter.formatDecimalSigned(&sdec_buf, signed); - const oct = engine.formatter.formatOctal(&oct_buf, value, config.int_type.width); - const bin = engine.formatter.formatBinary(&bin_buf, value, config.int_type.width); - - const output = std.fmt.bufPrint(buf, - \\ dec(signed): {s} - \\ dec(unsigned): {s} - \\ hex: {s} - \\ oct: {s} - \\ bin: {s} + var w = std.Io.Writer.fixed(buf); + w.print( + \\ dec(signed): {f} + \\ dec(unsigned): {f} + \\ hex: {f} + \\ oct: {f} + \\ bin: {f} \\ - , .{ sdec.display, dec.display, hex.display, oct.display, bin.display }) catch { + , .{ + result.as(.decimal_signed, .{}), + result.as(.decimal_unsigned, .{}), + result.as(.hex, .{ .endian = config.display_endian }), + result.as(.octal, .{}), + result.as(.binary, .{}), + }) catch { return .{ .output = "error: buffer overflow\n", .is_error = true }; }; - return .{ .output = output, .is_error = false }; + return .{ .output = w.buffered(), .is_error = false }; } /// Turn an engine error into a CLI line. @@ -944,7 +935,7 @@ test "parseArgs: --bits sets the width, in either spelling, and implies -p" { }) |args| { const e = parseArgs(testing.allocator, args).expression; defer testing.allocator.free(e.text); - try testing.expectEqual(engine.Integer.BitWidth.bits8, e.config.int_type.width); + try testing.expectEqual(engine.Integer.BitWidth.bits8, e.config.width); // Asking for a width is asking for programmer mode: standard mode is fixed // at 64-bit signed, so the flag would mean nothing there. try testing.expectEqual(Mode.programmer, e.mode); @@ -971,12 +962,12 @@ test "parseArgs: every documented bit width is accepted" { test "parseArgs: --signed and --unsigned set the signedness and imply -p" { const signed = parseArgs(testing.allocator, &.{ "--signed", "0xFF" }).expression; defer testing.allocator.free(signed.text); - try testing.expectEqual(engine.Integer.Signedness.signed, signed.config.int_type.signedness); + try testing.expectEqual(engine.Integer.Signedness.signed, signed.config.signedness); try testing.expectEqual(Mode.programmer, signed.mode); const unsigned = parseArgs(testing.allocator, &.{ "--unsigned", "0xFF" }).expression; defer testing.allocator.free(unsigned.text); - try testing.expectEqual(engine.Integer.Signedness.unsigned, unsigned.config.int_type.signedness); + try testing.expectEqual(engine.Integer.Signedness.unsigned, unsigned.config.signedness); try testing.expectEqual(Mode.programmer, unsigned.mode); } @@ -999,8 +990,8 @@ test "parseArgs: --endian sets the byte order and accepts both spellings" { test "parseArgs: the flags combine, and order does not matter" { const e = parseArgs(testing.allocator, &.{ "--unsigned", "--bits", "16", "--endian", "little", "0xFF", "+", "1" }).expression; defer testing.allocator.free(e.text); - try testing.expectEqual(engine.Integer.BitWidth.bits16, e.config.int_type.width); - try testing.expectEqual(engine.Integer.Signedness.unsigned, e.config.int_type.signedness); + try testing.expectEqual(engine.Integer.BitWidth.bits16, e.config.width); + try testing.expectEqual(engine.Integer.Signedness.unsigned, e.config.signedness); try testing.expectEqual(std.builtin.Endian.little, e.config.display_endian); try testing.expectEqualStrings("0xFF + 1", e.text); } @@ -1050,29 +1041,31 @@ test "the programmer config reaches the result" { const alloc = arena.allocator(); // 8-bit: 0xFF + 1 wraps to 0. - const narrow = evaluateWith(alloc, "0xFF + 1", .programmer, .{ .int_type = .{ .width = .bits8 } }, &buf); + const narrow = evaluateWith(alloc, "0xFF + 1", .programmer, .{ .width = .bits8 }, &buf); try testing.expect(!narrow.is_error); try testing.expect(std.mem.indexOf(u8, narrow.output, "dec(unsigned): 0\n") != null); // Signedness decides whether >> extends the sign. const signed = evaluateWith(alloc, "0xFF >> 1", .programmer, .{ - .int_type = .{ .width = .bits8, .signedness = .signed }, + .width = .bits8, + .signedness = .signed, }, &buf); try testing.expect(std.mem.indexOf(u8, signed.output, "dec(signed): -1") != null); const unsigned = evaluateWith(alloc, "0xFF >> 1", .programmer, .{ - .int_type = .{ .width = .bits8, .signedness = .unsigned }, + .width = .bits8, + .signedness = .unsigned, }, &buf); try testing.expect(std.mem.indexOf(u8, unsigned.output, "dec(unsigned): 127") != null); // Byte order reverses the hex row and nothing else. const little = evaluateWith(alloc, "0xDEAD", .programmer, .{ - .int_type = .{ .width = .bits16 }, + .width = .bits16, .display_endian = .little, }, &buf); try testing.expect(std.mem.indexOf(u8, little.output, "hex: AD DE") != null); const big = evaluateWith(alloc, "0xDEAD", .programmer, .{ - .int_type = .{ .width = .bits16 }, + .width = .bits16, .display_endian = .big, }, &buf); try testing.expect(std.mem.indexOf(u8, big.output, "hex: DE AD") != null); diff --git a/src/tui.zig b/src/tui.zig index 3049129..7ffb32c 100644 --- a/src/tui.zig +++ b/src/tui.zig @@ -373,7 +373,7 @@ pub const App = struct { self.value_zone_active = true; self.prog_field = target.field; if (target.bit) |bit| { - if (bit < self.prog_config.int_type.width.bits()) self.bit_cursor = bit; + if (bit < self.prog_config.width.bits()) self.bit_cursor = bit; } else { self.alignCursorToField(); } @@ -381,10 +381,10 @@ pub const App = struct { .toggle_bit => |bit| { self.value_zone_active = true; self.prog_field = if (self.prog_field == .bin) .bin else .bits; - if (bit < self.prog_config.int_type.width.bits()) { + if (bit < self.prog_config.width.bits()) { self.bit_cursor = bit; self.prog_value ^= @as(u128, 1) << bit; - self.prog_value &= self.prog_config.int_type.width.mask(); + self.prog_value &= self.prog_config.width.mask(); } }, .focus_input => self.value_zone_active = false, @@ -449,15 +449,15 @@ pub const App = struct { const signed: i128 = @intFromFloat(ans); self.prog_value = @bitCast(signed); } - self.prog_value &= self.prog_config.int_type.width.mask(); + self.prog_value &= self.prog_config.width.mask(); } else { self.float_format = .f64; self.float_view_active = true; self.syncFloatWidth(); self.prog_value = @as(u64, @bitCast(ans)); self.prog_field = .bits; - if (self.bit_cursor >= self.prog_config.int_type.width.bits()) { - self.bit_cursor = @intCast(self.prog_config.int_type.width.bits() - 1); + if (self.bit_cursor >= self.prog_config.width.bits()) { + self.bit_cursor = @intCast(self.prog_config.width.bits() - 1); } } } @@ -651,8 +651,8 @@ pub const App = struct { self.syncFloatWidth(); // Keep the bit grid focused so arrows/space edit bits directly. self.prog_field = .bits; - if (self.bit_cursor >= self.prog_config.int_type.width.bits()) { - self.bit_cursor = @intCast(self.prog_config.int_type.width.bits() - 1); + if (self.bit_cursor >= self.prog_config.width.bits()) { + self.bit_cursor = @intCast(self.prog_config.width.bits() - 1); } } ctx.redraw = true; @@ -789,7 +789,7 @@ pub const App = struct { // Up/Down: move between fields if (key.matches(vaxis.Key.up, .{})) { if (self.prog_field == .bits) { - const width = self.prog_config.int_type.width.bits(); + const width = self.prog_config.width.bits(); const bits_per_row: u8 = if (width > 32) 32 else width; if (@as(u8, self.bit_cursor) + bits_per_row < width) { self.bit_cursor += @intCast(bits_per_row); @@ -802,7 +802,7 @@ pub const App = struct { } if (key.matches(vaxis.Key.down, .{})) { if (self.prog_field == .bits) { - const width = self.prog_config.int_type.width.bits(); + const width = self.prog_config.width.bits(); const bits_per_row: u8 = if (width > 32) 32 else width; if (self.bit_cursor >= bits_per_row) { self.bit_cursor -= @intCast(bits_per_row); @@ -818,7 +818,7 @@ pub const App = struct { if (key.matches(vaxis.Key.left, .{})) { const step = self.fieldBitStep(); if (step > 0) { - const width = self.prog_config.int_type.width.bits(); + const width = self.prog_config.width.bits(); if (@as(u16, self.bit_cursor) + step < width) { self.bit_cursor += @intCast(step); } @@ -839,7 +839,7 @@ pub const App = struct { if (key.matches(' ', .{})) { if (self.prog_field == .bits) { self.prog_value ^= @as(u128, 1) << self.bit_cursor; - self.prog_value &= self.prog_config.int_type.width.mask(); + self.prog_value &= self.prog_config.width.mask(); } return; } @@ -867,7 +867,7 @@ pub const App = struct { const shift: u7 = self.bit_cursor & 0x7C; // round down to nibble boundary const mask = ~(@as(u128, 0xF) << shift); self.prog_value = (self.prog_value & mask) | (@as(u128, n) << shift); - self.prog_value &= self.prog_config.int_type.width.mask(); + self.prog_value &= self.prog_config.width.mask(); // Move cursor right (toward LSB) if (shift >= 4) self.bit_cursor -= 4; } @@ -879,7 +879,7 @@ pub const App = struct { const shift: u7 = (self.bit_cursor / 3) * 3; // round down to octal boundary const mask = ~(@as(u128, 0x7) << shift); self.prog_value = (self.prog_value & mask) | (@as(u128, digit) << shift); - self.prog_value &= self.prog_config.int_type.width.mask(); + self.prog_value &= self.prog_config.width.mask(); if (shift >= 3) self.bit_cursor -= 3; } }, @@ -889,7 +889,7 @@ pub const App = struct { if (self.bit_cursor > 0) self.bit_cursor -= 1; } else if (char == '1') { self.prog_value |= @as(u128, 1) << self.bit_cursor; - self.prog_value &= self.prog_config.int_type.width.mask(); + self.prog_value &= self.prog_config.width.mask(); if (self.bit_cursor > 0) self.bit_cursor -= 1; } }, @@ -897,7 +897,7 @@ pub const App = struct { if (char >= '0' and char <= '9') { // Operate on the in-width portion so hidden upper bits do // not corrupt the arithmetic; the edit commits to width. - const m = self.prog_config.int_type.width.mask(); + const m = self.prog_config.width.mask(); self.prog_value = (((self.prog_value & m) *% 10) +% (char - '0')) & m; } }, @@ -912,15 +912,15 @@ pub const App = struct { /// upper bits. The display masks to width and shows a warning while the /// value does not fit. Explicit value edits still commit to width. fn cycleBitWidth(self: *App) void { - self.prog_config.int_type.width = switch (self.prog_config.int_type.width) { + self.prog_config.width = switch (self.prog_config.width) { .bits8 => .bits16, .bits16 => .bits32, .bits32 => .bits64, .bits64 => .bits128, .bits128 => .bits8, }; - if (self.bit_cursor >= self.prog_config.int_type.width.bits()) { - self.bit_cursor = @intCast(self.prog_config.int_type.width.bits() - 1); + if (self.bit_cursor >= self.prog_config.width.bits()) { + self.bit_cursor = @intCast(self.prog_config.width.bits() - 1); } } @@ -932,7 +932,7 @@ pub const App = struct { } fn toggleSignedness(self: *App) void { - self.prog_config.int_type.signedness = switch (self.prog_config.int_type.signedness) { + self.prog_config.signedness = switch (self.prog_config.signedness) { .signed => .unsigned, .unsigned => .signed, }; @@ -948,12 +948,12 @@ pub const App = struct { /// Snap the bit width to match the active float format (f32 -> 32, f64 -> 64). fn syncFloatWidth(self: *App) void { - self.prog_config.int_type.width = switch (self.float_format) { + self.prog_config.width = switch (self.float_format) { .f32 => .bits32, .f64 => .bits64, }; - if (self.bit_cursor >= self.prog_config.int_type.width.bits()) { - self.bit_cursor = @intCast(self.prog_config.int_type.width.bits() - 1); + if (self.bit_cursor >= self.prog_config.width.bits()) { + self.bit_cursor = @intCast(self.prog_config.width.bits() - 1); } } @@ -1045,17 +1045,15 @@ pub const App = struct { as_float < 340282366920938463463374607431768211456.0) { const int_val: u128 = @intFromFloat(as_float); - const bw = engine.formatter.displayWidthFor(int_val); - var hex_buf: [256]u8 = undefined; - var oct_buf: [256]u8 = undefined; - var bin_buf: [512]u8 = undefined; - const hex = engine.formatter.formatHex(&hex_buf, int_val, bw, .big); - const oct = engine.formatter.formatOctal(&oct_buf, int_val, bw); - const bin = engine.formatter.formatBinary(&bin_buf, int_val, bw); + const int: engine.Integer = .{ + .raw = int_val, + .width = engine.formatter.displayWidthFor(int_val), + .signedness = .unsigned, + }; details = .{ - try std.fmt.allocPrint(self.allocator, "hex: {s}", .{hex.display}), - try std.fmt.allocPrint(self.allocator, "oct: {s}", .{oct.display}), - try std.fmt.allocPrint(self.allocator, "bin: {s}", .{bin.display}), + try std.fmt.allocPrint(self.allocator, "hex: {f}", .{int.as(.hex, .{})}), + try std.fmt.allocPrint(self.allocator, "oct: {f}", .{int.as(.octal, .{})}), + try std.fmt.allocPrint(self.allocator, "bin: {f}", .{int.as(.binary, .{})}), }; } @@ -1163,9 +1161,8 @@ pub const App = struct { fn submitProgrammer(self: *App, expr_text: []const u8) !void { const is_error, const display_text = if (engine.evalProgrammerString(self.allocator, expr_text, self.prog_config)) |int| blk: { self.prog_value = int.unsignedValue(); - var buf: [256]u8 = undefined; - const dec = engine.formatter.formatDecimalUnsigned(&buf, int.unsignedValue()); - break :blk .{ false, try self.allocator.dupe(u8, dec.display) }; + const text = try std.fmt.allocPrint(self.allocator, "{f}", .{int.as(.decimal_unsigned, .{})}); + break :blk .{ false, text }; } else |err| blk: { break :blk .{ true, try self.allocator.dupe(u8, errorStr(err)) }; }; @@ -1769,9 +1766,9 @@ test "other modes still get their own keys after financial mode was added" { // Programmer mode's bit-width cycle still works. app.setMode(.programmer); - const width_before = app.prog_config.int_type.width; + const width_before = app.prog_config.width; try press(&app, &ctx, .{ .codepoint = 'w', .mods = .{ .ctrl = true } }); - try testing.expect(app.prog_config.int_type.width != width_before); + try testing.expect(app.prog_config.width != width_before); } test "shift-tab walks back through the modes" { @@ -2023,7 +2020,7 @@ test "render: programmer mode warns when the value exceeds the display width" { defer app.deinit(); app.setMode(.programmer); app.prog_value = 0xDEADBEEF; - app.prog_config.int_type.width = .bits8; + app.prog_config.width = .bits8; const rows = try renderApp(arena, &app, 100, 30); try testing.expect(test_render.contains(rows, "value exceeds 8 bits")); @@ -2040,11 +2037,11 @@ test "render: programmer mode is well formed at every width and setting" { app.prog_value = 0xFEDCBA9876543210; for ([_]engine.BitWidth{ .bits8, .bits16, .bits32, .bits64, .bits128 }) |width| { - app.prog_config.int_type.width = width; + app.prog_config.width = width; for ([_]std.builtin.Endian{ .little, .big }) |endian| { app.prog_config.display_endian = endian; for ([_]engine.Integer.Signedness{ .signed, .unsigned }) |signedness| { - app.prog_config.int_type.signedness = signedness; + app.prog_config.signedness = signedness; const rows = try renderApp(arena, &app, 100, 40); try testing.expect(test_render.furniture(rows).intact()); var buf: [24]u8 = undefined; @@ -2065,7 +2062,7 @@ test "render: the float view decodes a known bit pattern" { app.setMode(.programmer); app.float_view_active = true; app.float_format = .f32; - app.prog_config.int_type.width = .bits32; + app.prog_config.width = .bits32; app.prog_value = @as(u32, @bitCast(@as(f32, 1.0))); const rows = try renderApp(arena, &app, 100, 30); @@ -2106,7 +2103,7 @@ test "render: the float view decodes every classification by name" { }; for (cases) |case| { app.float_format = case.format; - app.prog_config.int_type.width = if (case.format == .f32) .bits32 else .bits64; + app.prog_config.width = if (case.format == .f32) .bits32 else .bits64; app.prog_value = case.bits; const rows = try renderApp(arena, &app, 100, 34); try testing.expect(test_render.furniture(rows).intact()); @@ -2347,7 +2344,7 @@ test "render: 128-bit programmer mode keeps its input line on a short terminal" var app = testApp(); defer app.deinit(); app.setMode(.programmer); - app.prog_config.int_type.width = .bits128; + app.prog_config.width = .bits128; // Four grid rows plus six base rows do not fit in 16 rows. The view used to // draw them anyway, putting the BIN row on the prompt. @@ -2555,12 +2552,12 @@ test "programmer mode: every clickable control acts" { try app.applyAction(&ctx, .{ .prog_field = .{ .field = .oct, .bit = null } }); try testing.expectEqual(App.ProgField.oct, app.prog_field); // An out-of-width bit is ignored rather than moving the cursor off the value. - app.prog_config.int_type.width = .bits8; + app.prog_config.width = .bits8; try app.applyAction(&ctx, .{ .prog_field = .{ .field = .bits, .bit = 100 } }); try testing.expect(app.bit_cursor < 8); // Toggling bits. - app.prog_config.int_type.width = .bits32; + app.prog_config.width = .bits32; app.prog_value = 0; try app.applyAction(&ctx, .{ .toggle_bit = 3 }); try testing.expectEqual(@as(u128, 8), app.prog_value); @@ -2570,15 +2567,15 @@ test "programmer mode: every clickable control acts" { try testing.expectEqual(@as(u128, 0), app.prog_value); // Settings. - const width_before = app.prog_config.int_type.width; + const width_before = app.prog_config.width; try app.applyAction(&ctx, .cycle_width); - try testing.expect(app.prog_config.int_type.width != width_before); + try testing.expect(app.prog_config.width != width_before); const endian_before = app.prog_config.display_endian; try app.applyAction(&ctx, .toggle_endian); try testing.expect(app.prog_config.display_endian != endian_before); - const signed_before = app.prog_config.int_type.signedness; + const signed_before = app.prog_config.signedness; try app.applyAction(&ctx, .toggle_signedness); - try testing.expect(app.prog_config.int_type.signedness != signed_before); + try testing.expect(app.prog_config.signedness != signed_before); // Float overlay and its format toggle. try app.applyAction(&ctx, .toggle_float); @@ -2605,15 +2602,15 @@ test "programmer mode: bit width cycles through every size and keeps the cursor app.setMode(.programmer); app.value_zone_active = true; app.bit_cursor = 100; - app.prog_config.int_type.width = .bits8; + app.prog_config.width = .bits8; var seen: usize = 0; while (seen < 6) : (seen += 1) { try press(&app, &ctx, .{ .codepoint = 'w', .mods = .{ .ctrl = true } }); - try testing.expect(app.bit_cursor < app.prog_config.int_type.width.bits()); + try testing.expect(app.bit_cursor < app.prog_config.width.bits()); } // Six steps through five widths lands one past the start: 8, 16, 32, 64, 128, 8, 16. - try testing.expectEqual(engine.BitWidth.bits16, app.prog_config.int_type.width); + try testing.expectEqual(engine.BitWidth.bits16, app.prog_config.width); } test "programmer mode: typing edits the focused field in its own base" { @@ -2623,7 +2620,7 @@ test "programmer mode: typing edits the focused field in its own base" { defer ctx.cmds.deinit(testing.allocator); app.setMode(.programmer); app.value_zone_active = true; - app.prog_config.int_type.width = .bits32; + app.prog_config.width = .bits32; // Hex nibble entry. app.prog_field = .hex; @@ -2668,7 +2665,7 @@ test "programmer mode: arrows and space navigate the bit grid" { defer ctx.cmds.deinit(testing.allocator); app.setMode(.programmer); app.value_zone_active = true; - app.prog_config.int_type.width = .bits64; + app.prog_config.width = .bits64; app.prog_field = .bits; app.bit_cursor = 0; @@ -2736,7 +2733,7 @@ test "float view: typing a decimal stores the nearest bit pattern" { app.setMode(.programmer); app.float_view_active = true; app.float_format = .f64; - app.prog_config.int_type.width = .bits64; + app.prog_config.width = .bits64; try app.input.insertSliceAtCursor("3.14"); try press(&app, &ctx, .{ .codepoint = vaxis.Key.enter }); @@ -2751,7 +2748,7 @@ test "float view: typing a decimal stores the nearest bit pattern" { // In the float overlay Ctrl-W swaps format instead of cycling width. try press(&app, &ctx, .{ .codepoint = 'w', .mods = .{ .ctrl = true } }); try testing.expectEqual(engine.FloatFormat.f32, app.float_format); - try testing.expectEqual(engine.BitWidth.bits32, app.prog_config.int_type.width); + try testing.expectEqual(engine.BitWidth.bits32, app.prog_config.width); // Ctrl-F leaves the overlay. try press(&app, &ctx, .{ .codepoint = 'f', .mods = .{ .ctrl = true } }); @@ -3024,7 +3021,7 @@ test "render: programmer mode draws the cursor in whichever field is focused" { for (fields) |field| { app.prog_field = field; for ([_]engine.BitWidth{ .bits8, .bits64, .bits128 }) |width| { - app.prog_config.int_type.width = width; + app.prog_config.width = width; app.bit_cursor = @intCast(@min(5, width.bits() - 1)); const rows = try renderApp(arena, &app, 110, 40); try testing.expect(test_render.furniture(rows).intact()); diff --git a/src/tui/financial.zig b/src/tui/financial.zig index 8a103b8..2e89ddd 100644 --- a/src/tui/financial.zig +++ b/src/tui/financial.zig @@ -30,6 +30,7 @@ const C = draw.C; const financial = engine.financial; const formatter = engine.formatter; +const grouping = engine.grouping; /// Which calculation the form is showing. pub const Form = enum { @@ -192,14 +193,25 @@ pub const Field = struct { } /// Write the display form into `dest` and return it: a plain number gets - /// thousands separators, an expression is shown exactly as typed (there is + /// thousands separators, anything else is shown exactly as typed (there is /// nothing meaningful to group in "12 * 30"). + /// + /// The gate is what the grouper accepts, not what `parseFloat` accepts. Those are + /// different languages: `parseFloat` takes `1_000` and `0x1p4`, which are not + /// expressions by the `isExpression` test and used to be grouped into `1_,000` + /// and `0x,1p4`. pub fn writeDisplay(self: *const Field, dest: []u8) []const u8 { const raw = self.text(); - if (self.isExpression()) return raw[0..@min(raw.len, dest.len)]; - const needed = formatter.groupedDecimalLen(raw); - if (needed == raw.len or needed > dest.len) return raw[0..@min(raw.len, dest.len)]; - return dest[0..formatter.writeGroupedDecimal(dest, raw)]; + const as_typed = raw[0..@min(raw.len, dest.len)]; + if (!grouping.isNumericText(raw)) return as_typed; + // Nothing to group: three digits or fewer, or scientific notation. + if (grouping.lengthOf(raw) == raw.len) return as_typed; + + var w = std.Io.Writer.fixed(dest); + // No room for the grouped form: show the digits ungrouped rather than a + // truncated, wrong-looking number. + grouping.print(&w, raw) catch return as_typed; + return w.buffered(); } }; @@ -989,6 +1001,29 @@ test "Field: display groups a plain number and leaves an expression alone" { try testing.expectEqualStrings("1e6", field.writeDisplay(&buf)); } +test "Field: display leaves anything the grouper does not accept alone" { + // `std.fmt.parseFloat` accepts both of these, so they are not expressions by the + // `isExpression` test, and gating on that put commas three characters apart + // wherever they happened to fall: "1_,000" and "0x,1p4". The gate is now what the + // grouper accepts. + var buf: [64]u8 = undefined; + var field: Field = .{}; + + field.set("1_000"); + try testing.expectEqualStrings("1_000", field.writeDisplay(&buf)); + field.set("1_0_0_0"); + try testing.expectEqualStrings("1_0_0_0", field.writeDisplay(&buf)); + field.set("0x1p4"); + try testing.expectEqualStrings("0x1p4", field.writeDisplay(&buf)); + field.set("0x10"); + try testing.expectEqualStrings("0x10", field.writeDisplay(&buf)); + + // And the values still parse, so the field is not treated as an expression. + field.set("1_000"); + try testing.expectEqual(@as(?f64, 1000), field.value()); + try testing.expect(!field.isExpression()); +} + test "Field: display falls back to the raw text when the buffer is too small" { var tiny: [4]u8 = undefined; var field: Field = .{}; diff --git a/src/tui/float_view.zig b/src/tui/float_view.zig index 0af6d4f..e243586 100644 --- a/src/tui/float_view.zig +++ b/src/tui/float_view.zig @@ -37,7 +37,7 @@ pub fn drawFloatView(app: *tui.App, surface: *vxfw.Surface, width: u16, height: _ = width; const format = app.float_format; const total = format.totalBits(); - const bw = app.prog_config.int_type.width; + const bw = app.prog_config.width; const bits: u64 = @truncate(app.prog_value & bw.mask()); const info = fi.decompose(format, bits); diff --git a/src/tui/programmer.zig b/src/tui/programmer.zig index bc9ae80..74db5a2 100644 --- a/src/tui/programmer.zig +++ b/src/tui/programmer.zig @@ -8,8 +8,13 @@ const draw = @import("draw.zig"); const tui = @import("../tui.zig"); const C = draw.C; +/// Stand-in when a row does not fit its buffer. Drawing cannot fail, so a row that +/// cannot be rendered says so rather than propagating. The buffers below are sized +/// from the widest supported value, so this is unreachable in practice. +const too_wide: []const u8 = "(too wide)"; + pub fn drawProgrammerMode(app: *tui.App, surface: *vxfw.Surface, width: u16, height: u16) void { - const bw = app.prog_config.int_type.width; + const bw = app.prog_config.width; const val = app.prog_value & bw.mask(); const focused = app.prog_field; @@ -17,7 +22,7 @@ pub fn drawProgrammerMode(app: *tui.App, surface: *vxfw.Surface, width: u16, hei var config_buf: [80]u8 = undefined; const config_str = std.fmt.bufPrint(&config_buf, "Bits: {d} Signed: {s} Endian: {s}", .{ bw.bits(), - if (app.prog_config.int_type.signedness == .signed) "yes" else "no", + if (app.prog_config.signedness == .signed) "yes" else "no", if (app.prog_config.display_endian == .little) "LE" else "BE", }) catch "Bits: ??"; draw.writeStr(surface, 2, 2, config_str, .{ .fg = C.muted }); @@ -65,86 +70,86 @@ pub fn drawProgrammerMode(app: *tui.App, surface: *vxfw.Surface, width: u16, hei drawBitGrid(app, surface, grid_start, val, bw); - const int = engine.Integer.init(val, .{ .width = bw, .signedness = .signed }); + const int: engine.Integer = .{ .raw = val, .width = bw, .signedness = .signed }; // DEC(s) var sdec_buf: [256]u8 = undefined; - const sdec = engine.formatter.formatDecimalSigned(&sdec_buf, int.signedValue()); + const sdec = int.as(.decimal_signed, .{}).render(&sdec_buf) catch too_wide; const sdec_style: vaxis.Style = if (focused == .dec_signed) .{ .fg = C.bg, .bg = C.cyan, .bold = true } else .{ .fg = C.cyan }; draw.writeStr(surface, base_start, 2, "DEC(s):", sdec_style); - draw.writeStr(surface, base_start, 11, sdec.display, if (focused == .dec_signed) .{ .fg = C.fg, .bold = true } else .{ .fg = C.fg }); + draw.writeStr(surface, base_start, 11, sdec, if (focused == .dec_signed) .{ .fg = C.fg, .bold = true } else .{ .fg = C.fg }); app.addRegion(base_start, 0, width, .{ .prog_field = .{ .field = .dec_signed, .bit = null } }); // DEC(u) var udec_buf: [256]u8 = undefined; - const udec = engine.formatter.formatDecimalUnsigned(&udec_buf, val); + const udec = int.as(.decimal_unsigned, .{}).render(&udec_buf) catch too_wide; const udec_style: vaxis.Style = if (focused == .dec_unsigned) .{ .fg = C.bg, .bg = C.cyan, .bold = true } else .{ .fg = C.cyan }; draw.writeStr(surface, base_start + 1, 2, "DEC(u):", udec_style); - draw.writeStr(surface, base_start + 1, 11, udec.display, if (focused == .dec_unsigned) .{ .fg = C.fg, .bold = true } else .{ .fg = C.fg }); + draw.writeStr(surface, base_start + 1, 11, udec, if (focused == .dec_unsigned) .{ .fg = C.fg, .bold = true } else .{ .fg = C.fg }); app.addRegion(base_start + 1, 0, width, .{ .prog_field = .{ .field = .dec_unsigned, .bit = null } }); // HEX var hex_buf: [256]u8 = undefined; - const hex = engine.formatter.formatHex(&hex_buf, val, bw, app.prog_config.display_endian); + const hex = int.as(.hex, .{ .endian = app.prog_config.display_endian }).render(&hex_buf) catch too_wide; const hex_style: vaxis.Style = if (focused == .hex) .{ .fg = C.bg, .bg = C.green, .bold = true } else .{ .fg = C.cyan }; draw.writeStr(surface, base_start + 2, 2, "HEX:", hex_style); if (focused == .hex) { - drawFieldWithCursor(surface, base_start + 2, 11, hex.display, app.bit_cursor, 4, bw.bits(), C.green); + drawFieldWithCursor(surface, base_start + 2, 11, hex, app.bit_cursor, 4, bw.bits(), C.green); } else { - draw.writeStr(surface, base_start + 2, 11, hex.display, .{ .fg = C.green }); + draw.writeStr(surface, base_start + 2, 11, hex, .{ .fg = C.green }); } // Row-wide fallback focuses the field; per-digit regions (added next) place // the cursor on the exact nibble that was clicked. app.addRegion(base_start + 2, 0, width, .{ .prog_field = .{ .field = .hex, .bit = null } }); - registerDigitRegions(app, base_start + 2, 11, hex.display, 4, .hex, false); + registerDigitRegions(app, base_start + 2, 11, hex, 4, .hex, false); // ASCII (derived, read-only): one glyph per byte, aligned under HEX. var ascii_buf: [128]u8 = undefined; - const ascii = engine.formatter.formatAscii(&ascii_buf, val, bw, app.prog_config.display_endian); + const ascii = int.as(.ascii, .{ .endian = app.prog_config.display_endian }).render(&ascii_buf) catch too_wide; draw.writeStr(surface, base_start + 3, 2, "ASCII:", .{ .fg = C.cyan }); - draw.writeStr(surface, base_start + 3, 11, ascii.display, .{ .fg = C.orange }); + draw.writeStr(surface, base_start + 3, 11, ascii, .{ .fg = C.orange }); // OCT var oct_buf: [256]u8 = undefined; - const oct = engine.formatter.formatOctal(&oct_buf, val, bw); + const oct = int.as(.octal, .{}).render(&oct_buf) catch too_wide; const oct_style: vaxis.Style = if (focused == .oct) .{ .fg = C.bg, .bg = C.purple, .bold = true } else .{ .fg = C.cyan }; draw.writeStr(surface, base_start + 4, 2, "OCT:", oct_style); if (focused == .oct) { - drawFieldWithCursor(surface, base_start + 4, 11, oct.display, app.bit_cursor, 3, bw.bits(), C.purple); + drawFieldWithCursor(surface, base_start + 4, 11, oct, app.bit_cursor, 3, bw.bits(), C.purple); } else { - draw.writeStr(surface, base_start + 4, 11, oct.display, .{ .fg = C.purple }); + draw.writeStr(surface, base_start + 4, 11, oct, .{ .fg = C.purple }); } app.addRegion(base_start + 4, 0, width, .{ .prog_field = .{ .field = .oct, .bit = null } }); - registerDigitRegions(app, base_start + 4, 11, oct.display, 3, .oct, false); + registerDigitRegions(app, base_start + 4, 11, oct, 3, .oct, false); // BIN var bin_buf: [512]u8 = undefined; - const bin = engine.formatter.formatBinary(&bin_buf, val, bw); + const bin = int.as(.binary, .{}).render(&bin_buf) catch too_wide; const bin_style: vaxis.Style = if (focused == .bin) .{ .fg = C.bg, .bg = C.yellow, .bold = true } else .{ .fg = C.cyan }; draw.writeStr(surface, base_start + 5, 2, "BIN:", bin_style); if (focused == .bin) { - drawFieldWithCursor(surface, base_start + 5, 11, bin.display, app.bit_cursor, 1, bw.bits(), C.yellow); + drawFieldWithCursor(surface, base_start + 5, 11, bin, app.bit_cursor, 1, bw.bits(), C.yellow); } else { - draw.writeStr(surface, base_start + 5, 11, bin.display, .{ .fg = C.yellow }); + draw.writeStr(surface, base_start + 5, 11, bin, .{ .fg = C.yellow }); } app.addRegion(base_start + 5, 0, width, .{ .prog_field = .{ .field = .bin, .bit = null } }); // A binary digit IS a single bit, so clicking one flips it directly. - registerDigitRegions(app, base_start + 5, 11, bin.display, 1, .bin, true); + registerDigitRegions(app, base_start + 5, 11, bin, 1, .bin, true); // History const hist_start = base_start + 7;