diff --git a/.kiro/specs/calculator/design.md b/.kiro/specs/calculator/design.md index 76e792d..d38946a 100644 --- a/.kiro/specs/calculator/design.md +++ b/.kiro/specs/calculator/design.md @@ -144,17 +144,24 @@ pub const Expr = union(enum) { }; pub const Number = struct { - float_value: f64, - /// For programmer mode: the exact integer when the literal fits u64. - int_value: ?u64, base: Base, - /// The literal exactly as written, pointing into the source. Decimal literals - /// are re-parsed from this rather than from `float_value`, because - /// `float_value` has already rounded and `0.1` cannot be recovered from it. + /// The literal exactly as written, pointing into the source. The only + /// representation kept, because it is the only lossless one: each reader parses + /// it at the precision it works in (`Number.parse` for the exact tier, + /// `Base.parseDigits` for a fixed-width pattern). This node used to carry an + /// `f64` and a `?u64` computed up front, which capped a literal at 128 + /// characters and 64 bits; see Task 5.18. text: []const u8, }; -pub const Base = enum { decimal, hex, octal, binary }; +pub const Base = enum { + decimal, + hex, + octal, + binary, + /// The literal's digits as a value, separators skipped, `Overflow` past u128. + pub fn parseDigits(self: Base, text: []const u8) error{Overflow}!u128 { ... } +}; pub const Binary = struct { op: BinaryOp, diff --git a/.kiro/specs/calculator/tasks.md b/.kiro/specs/calculator/tasks.md index 6b93c74..75f6723 100644 --- a/.kiro/specs/calculator/tasks.md +++ b/.kiro/specs/calculator/tasks.md @@ -914,6 +914,62 @@ plans a `--raw` flag, but today it is exercised only by tests. 100% line coverage, engine 99.44%. CLI output byte-identical across all five rows, both byte orders, ASCII packing and the multi-base standard-mode view. +### Task 5.18: A literal is its text; each reader parses it at its own precision + +The tokenizer computed two lossy representations of every numeric literal up front, +in a fixed 128-byte buffer: an `f64` and a `?u64`. Both went into the AST node +alongside the source text, and every limit in the pipeline came from that decision. + +Five reachable defects, one cause: + +``` +600-digit literal error: invalid number now: too many digits +130-character literal error: invalid number now: parses exactly +1e100001 inf now: exponent too large +0x1FFFFFFFFFFFFFFFF, programmer mode error: invalid number now: the value +u128 max in any base, programmer mode error / overflow now: the value +``` + +`Rational.parseDecimal` accepts 512 digits and reports `TooManyDigits`, and no +frontend could reach either, because a literal over 128 characters died in the +tokenizer first with the wrong message. `int_value` was a `u64` in a mode whose widest +type is 128 bits, so a 17-nibble hex literal was "invalid number". And +`evaluator.literalToNumber` swallowed every `Number.parse` error and fell back to the +f64, so `1e100001` printed `inf` instead of saying the exponent was out of range. + +- `ast.Expr.Number` is now `{ base, text }`. The text is the only lossless form of a + literal, so it is the only one kept. +- `Base` gained `parseDigits`, which walks the literal's own text skipping separators + and accumulating into a `u128` with checked arithmetic. No buffer, so length is not + a limit; `Overflow` past 128 bits, which is the widest pattern `Integer` holds. +- The exact tier parses decimals with `Number.parse` and propagates the error. +- Programmer mode parses decimals through the exact tier too, then truncates toward + zero. It has no float path at all now, which removes the `@intFromFloat` range + guards (illegal behaviour past u128, which is what aborted `tally -p '1e40'`) and + makes the truncation exact: `1e30` used to arrive as + 1000000000000000019884624838656. +- Validity moved into `readNumber`, which knows it consumed no digits after a `0x` or + an `e`. `TokenKind.invalid_number` carries that, so the parser still says "invalid + number" for `0x` and "unexpected token" for `$` without re-deriving either from the + text. +- `Token` carries the base the scan recognised, so nothing downstream re-reads the + prefix. + +`parseNumber`, `NumberValue`, the 128-byte buffer, the u64 cap and the f64 fallback +are all gone. + +A latent crash surfaced once literals could be that wide. `Integer.signedValue` +consulted `signedness` and `@intCast` the pattern when it said unsigned, so a 128-bit +unsigned value with the top bit set panicked ("integer does not fit in destination +type") as soon as anything printed it, which every display path does. It was +unreachable only because such a literal could not be entered. Reading a pattern as +two's complement is a `@bitCast` and cannot fail, and it is what +`Notation.decimal_signed` documented all along: the design mockups show +`DEC 4,294,967,295` beside `SIGN -1`, two readings of one pattern, where the old +implementation made those rows identical for every unsigned value. `isNegative` still +consults `signedness`, because that is the arithmetic question the shift operators +ask. + ### Task 5.15: A value renders itself `formatter.zig` had six functions taking `(buf, value: u128, bit_width, endian)`, @@ -1196,9 +1252,9 @@ STILL OPEN, in the order I would take them: Fixed by the de-duplication pass above. 10. Assignment parses in prefix position (`1 + x = 2` mutates `x`), and assignment to a constant name is silently discarded. -11. Literals longer than 128 characters are rejected by a fixed tokenizer buffer, +11. ~~Literals longer than 128 characters are rejected by a fixed tokenizer buffer, and non-decimal literals are capped at 64 bits, both below what the exact tier - supports. + supports.~~ Fixed by Task 5.18. 12. `engine/src/c_api.zig` has no tests and appears in no coverage report, because no test target builds the shared library. Deferred until Phase 6 gives it a caller; recorded here so it is a known gap rather than an oversight. diff --git a/engine/src/Integer.zig b/engine/src/Integer.zig index 83ef215..d8391a0 100644 --- a/engine/src/Integer.zig +++ b/engine/src/Integer.zig @@ -29,17 +29,33 @@ signedness: Signedness = .signed, /// negative range, so `>>` is a zero fill there and a large shift distance is large /// rather than negative. pub fn isNegative(self: Integer) bool { - const sign_bit = @as(u128, 1) << @intCast(self.width.bits() - 1); - return self.signedness == .signed and self.unsignedValue() & sign_bit != 0; + return self.signedness == .signed and self.unsignedValue() & self.signBit() != 0; } -/// Interpret as a signed value, sign-extended from the width. +/// The two's complement reading of the pattern, sign-extended from the width. +/// +/// Independent of `signedness`, unlike `isNegative`: the programmer view shows both +/// readings of one pattern side by side (design.md's DEC and SIGN rows), and the +/// point of showing both is that they differ. An unsigned 8-bit `0xFF` is 255 read +/// as unsigned and -1 read as signed. What `signedness` decides is what the pattern +/// *means* to arithmetic, which is what `isNegative` answers. +/// +/// This used to consult `signedness` and `@intCast` the pattern when it said +/// unsigned, which made the two decimal rows identical for every unsigned value and +/// panicked outright on a 128-bit unsigned pattern with the top bit set, since no +/// such value fits an i128. Reading the pattern is always a `@bitCast`, so it cannot +/// fail. pub fn signedValue(self: Integer) i128 { const pattern = self.unsignedValue(); - if (self.isNegative()) return @bitCast(pattern | ~self.width.mask()); + if (pattern & self.signBit() != 0) return @bitCast(pattern | ~self.width.mask()); return @intCast(pattern); } +/// The mask of the top bit at this width, which is the sign bit of a signed value. +fn signBit(self: Integer) u128 { + return @as(u128, 1) << @intCast(self.width.bits() - 1); +} + /// Interpret as an unsigned value. /// /// Also the bit pattern itself, truncated to the width, since an unsigned reading of @@ -324,12 +340,18 @@ test "the default is 64-bit signed, which is what standard mode uses" { try testing.expectEqual(@as(i128, -1), value.signedValue()); } -test "signedValue reads the top bit only when the value is signed" { +test "signedValue reads the pattern, not the configured signedness" { const signed: Integer = .{ .raw = 0xFF, .width = .bits8 }; const unsigned: Integer = .{ .raw = 0xFF, .width = .bits8, .signedness = .unsigned }; + // One pattern, two readings. Both rows of the programmer view come from here, + // so an unsigned value's signed reading has to be the signed one. try testing.expectEqual(@as(i128, -1), signed.signedValue()); - try testing.expectEqual(@as(i128, 255), unsigned.signedValue()); + try testing.expectEqual(@as(i128, -1), unsigned.signedValue()); + try testing.expectEqual(@as(u128, 255), unsigned.unsignedValue()); + + // What signedness decides is meaning, not rendering: only a signed value has a + // negative range, which is what the shift operators consult. try testing.expect(signed.isNegative()); try testing.expect(!unsigned.isNegative()); @@ -338,9 +360,20 @@ test "signedValue reads the top bit only when the value is signed" { const max: Integer = .{ .raw = 0x7F, .width = .bits8 }; try testing.expectEqual(@as(i128, 127), max.signedValue()); - // A 128-bit value has no bits above the width to fill. - const wide: Integer = .{ .raw = BitWidth.bits128.mask(), .width = .bits128 }; + // A 128-bit unsigned pattern with the top bit set has no i128 to be cast to, + // and reading it used to panic. It is -1 read as two's complement, and every + // display path asks for that reading. + const wide: Integer = .{ + .raw = std.math.maxInt(u128), + .width = .bits128, + .signedness = .unsigned, + }; try testing.expectEqual(@as(i128, -1), wide.signedValue()); + try testing.expectEqual(std.math.maxInt(u128), wide.unsignedValue()); + + // The signed-by-configuration case of the same pattern. + const wide_signed: Integer = .{ .raw = BitWidth.bits128.mask(), .width = .bits128 }; + try testing.expectEqual(@as(i128, -1), wide_signed.signedValue()); } test "bits above the width never reach an interpretation" { @@ -363,8 +396,12 @@ test "unsignedValue masks correctly" { test "the same bits read two ways, which is why the width travels with the value" { const signed: Integer = .{ .raw = 0xFF, .width = .bits8 }; const unsigned: Integer = .{ .raw = 0xFF, .width = .bits8, .signedness = .unsigned }; + // The two readings of one pattern are the pattern's, so both values give both. + // What differs is `isNegative`, which is the arithmetic meaning. try testing.expectEqual(@as(i128, -1), signed.signedValue()); - try testing.expectEqual(@as(i128, 255), unsigned.signedValue()); + try testing.expectEqual(@as(i128, -1), unsigned.signedValue()); + try testing.expectEqual(@as(u128, 255), signed.unsignedValue()); + try testing.expectEqual(@as(u128, 255), unsigned.unsignedValue()); } test "withRaw keeps the type and masks the new pattern" { diff --git a/engine/src/ast.zig b/engine/src/ast.zig index 357e282..cee5e87 100644 --- a/engine/src/ast.zig +++ b/engine/src/ast.zig @@ -24,18 +24,19 @@ pub const Expr = union(enum) { assignment: Assignment, pub const Number = struct { - float_value: f64, - /// If the number is a pure integer, stores the exact value. - int_value: ?u64, + /// The base the literal was written in, for the multi-base display + /// (FR-1.9) and to tell the readers below which parse applies. base: Base, - /// The literal's source text, with separators intact. Points into the - /// expression source, which outlives the AST. + /// The literal's source text, prefix and separators intact. Points into + /// the expression source, which outlives the AST. /// - /// Kept so the evaluator can reconstruct the literal *exactly* as a - /// rational: `float_value` has already lost information by the time it - /// exists (0.1 is not representable in binary), so it cannot be the - /// basis for exact arithmetic. See design.md 2.7. - text: []const u8 = "", + /// The only representation kept, because it is the only lossless one. This + /// node used to carry an `f64` and a `?u64` that the tokenizer computed up + /// front, which capped a literal at 128 characters and 64 bits and made + /// `TooManyDigits` unreachable. Each consumer now parses the text with the + /// precision it actually works in: `Number.parse` for the exact tier, + /// `Base.parseDigits` for a fixed-width pattern. See design.md 2.7. + text: []const u8, }; pub const Unary = struct { diff --git a/engine/src/evaluator.zig b/engine/src/evaluator.zig index 8d6f75f..eac1c12 100644 --- a/engine/src/evaluator.zig +++ b/engine/src/evaluator.zig @@ -197,25 +197,20 @@ fn evalExact(env: *Environment, scratch: Allocator, expr: *const Expr) Error!Num } } -/// Turn a literal into a Number, exactly where possible. +/// Turn a literal into a Number, exactly. /// -/// Decimal literals are re-parsed from their source text rather than taken from -/// `float_value`, because `float_value` has already rounded: `0.1` cannot be -/// recovered from its binary approximation. +/// Both bases parse the literal's own source text, because that is the only +/// lossless form of it: `0.1` cannot be recovered from the nearest f64, and +/// `0x1FFFFFFFFFFFFFFFF` does not fit the u64 the tokenizer used to precompute. +/// +/// Failures propagate. This used to swallow every `Number.parse` error and fall +/// back to a float, so `1e100001` printed `inf` instead of reporting that the +/// exponent was out of range, and the exact tier silently became the inexact one. fn literalToNumber(scratch: Allocator, n: ast.Expr.Number) Error!Number { - if (n.base == .decimal and n.text.len > 0) { - if (Number.parse(scratch, n.text)) |value| return value else |_| { - // Fall through to the approximations below rather than failing: the - // tokenizer already accepted this text, so a parse mismatch here - // should degrade, not error. - } - } - // Non-decimal literals are integers; use the exact integer the tokenizer - // recovered when it fits, otherwise accept the float approximation. - if (n.int_value) |int_val| { - return try Number.fromInt(scratch, int_val); - } - return Number.fromFloat(n.float_value); + if (n.base == .decimal) return Number.parse(scratch, n.text); + // A non-decimal literal is a bit pattern, so it is an integer of at most the + // 128 bits `Integer` holds. Wider than that is `Overflow`, not a float. + return Number.fromInt(scratch, try n.base.parseDigits(n.text)); } /// Evaluate a binary operation. diff --git a/engine/src/parser.zig b/engine/src/parser.zig index 37a76a9..dc34343 100644 --- a/engine/src/parser.zig +++ b/engine/src/parser.zig @@ -19,7 +19,6 @@ const tokenizer_mod = @import("tokenizer.zig"); const Tokenizer = tokenizer_mod.Tokenizer; const TokenKind = tokenizer_mod.TokenKind; const Token = tokenizer_mod.Token; -const parseNumber = tokenizer_mod.parseNumber; /// What parsing can fail with. /// @@ -143,15 +142,14 @@ pub const Parser = struct { switch (tok.kind) { .number => { self.advance(); - const text = tok.text(self.source); - const num = parseNumber(text) catch return Error.InvalidNumber; return self.makeNode(.{ .number = .{ - .float_value = num.float, - .int_value = num.int_value, - .base = num.base, - .text = text, + .base = tok.base, + .text = tok.text(self.source), } }); }, + .invalid_number => { + return Error.InvalidNumber; + }, .string_literal => { self.advance(); const text = tok.text(self.source); @@ -382,6 +380,16 @@ pub const Parser = struct { const testing = std.testing; +/// A literal node's captured source text. +/// +/// The node no longer carries a precomputed `f64` and `?u64`, so what a parser test +/// can assert about a literal is the span it captured and the base it recognised. +/// What the text means belongs to `number.zig` and `Base.parseDigits`, and is tested +/// there against the full width and precision they support. +fn expectLiteral(expected: []const u8, expr: *const Expr) !void { + try testing.expectEqualStrings(expected, expr.number.text); +} + // Error-path tests used to need an arena because a failed parse leaked its // partial tree. They no longer do (the parser cleans up after itself), but the // arena helper is kept for the tests already written against it. @@ -425,22 +433,22 @@ pub fn freeExpr(allocator: Allocator, expr: *Expr) void { test "parse simple number" { const expr = try testParse("42"); defer freeExpr(testing.allocator, expr); - try testing.expectEqual(@as(f64, 42.0), expr.number.float_value); - try testing.expectEqual(@as(?u64, 42), expr.number.int_value); + try expectLiteral("42", expr); } test "parse hex number" { const expr = try testParse("0xFF"); defer freeExpr(testing.allocator, expr); - try testing.expectEqual(@as(?u64, 255), expr.number.int_value); + try expectLiteral("0xFF", expr); + try testing.expectEqual(tokenizer_mod.Base.hex, expr.number.base); } test "parse addition" { const expr = try testParse("2 + 3"); defer freeExpr(testing.allocator, expr); try testing.expectEqual(BinaryOp.add, expr.binary.op); - try testing.expectEqual(@as(f64, 2.0), expr.binary.left.number.float_value); - try testing.expectEqual(@as(f64, 3.0), expr.binary.right.number.float_value); + try expectLiteral("2", expr.binary.left); + try expectLiteral("3", expr.binary.right); } test "parse precedence: mul before add" { @@ -448,7 +456,7 @@ test "parse precedence: mul before add" { const expr = try testParse("2 + 3 * 4"); defer freeExpr(testing.allocator, expr); try testing.expectEqual(BinaryOp.add, expr.binary.op); - try testing.expectEqual(@as(f64, 2.0), expr.binary.left.number.float_value); + try expectLiteral("2", expr.binary.left); try testing.expectEqual(BinaryOp.mul, expr.binary.right.binary.op); } @@ -457,7 +465,7 @@ test "parse precedence: power right-associative" { const expr = try testParse("2^3^4"); defer freeExpr(testing.allocator, expr); try testing.expectEqual(BinaryOp.pow, expr.binary.op); - try testing.expectEqual(@as(f64, 2.0), expr.binary.left.number.float_value); + try expectLiteral("2", expr.binary.left); try testing.expectEqual(BinaryOp.pow, expr.binary.right.binary.op); } @@ -465,7 +473,7 @@ test "parse unary negation" { const expr = try testParse("-5"); defer freeExpr(testing.allocator, expr); try testing.expectEqual(UnaryOp.negate, expr.unary.op); - try testing.expectEqual(@as(f64, 5.0), expr.unary.operand.number.float_value); + try expectLiteral("5", expr.unary.operand); } test "parse negation in expression" { @@ -489,7 +497,7 @@ test "parse function call" { defer freeExpr(testing.allocator, expr); try testing.expectEqualStrings("sin", expr.call.name); try testing.expectEqual(@as(usize, 1), expr.call.args.len); - try testing.expectApproxEqAbs(@as(f64, 3.14), expr.call.args[0].number.float_value, 1e-10); + try expectLiteral("3.14", expr.call.args[0]); } test "parse multi-arg function call" { @@ -509,7 +517,7 @@ test "parse assignment" { const expr = try testParse("X = 42"); defer freeExpr(testing.allocator, expr); try testing.expectEqualStrings("X", expr.assignment.name); - try testing.expectEqual(@as(f64, 42.0), expr.assignment.value.number.float_value); + try expectLiteral("42", expr.assignment.value); } test "parse adjacent number and identifier is an error (no implicit mul)" { @@ -668,6 +676,8 @@ test "a failed parse leaves nothing allocated" { "", // empty input "1 + 2 3", // trailing token after a binary expression "-(1 + ", // nested failure under a unary + "0x", // a base prefix with no digits + "1e", // an exponent with no digits }; for (bad) |source| { var parser = Parser.init(testing.allocator, source); @@ -804,7 +814,7 @@ test "nesting within the depth limit parses" { var parser = Parser.init(testing.allocator, deep.items); const expr = try parser.parse(); defer freeExpr(testing.allocator, expr); - try testing.expectEqual(@as(f64, 7.0), expr.number.float_value); + try expectLiteral("7", expr); } test "unbalanced deep nesting is rejected without leaking the partial tree" { @@ -835,3 +845,14 @@ test "the depth limit also covers nested calls and unary operators" { var unary_parser = Parser.init(testing.allocator, unary.items); try testing.expectError(Error.InvalidExpression, unary_parser.parse()); } + +test "a numeric span that is not a number is an invalid number, not an unexpected token" { + // The tokenizer marks these while it scans, and the distinction is the error + // the user sees: "invalid number" for `0x`, "unexpected token" for `$`. + for ([_][]const u8{ "0x", "0o", "1e", "1e+", "2 + 0x" }) |source| { + var parser = Parser.init(testing.allocator, source); + try testing.expectError(Error.InvalidNumber, parser.parse()); + } + var stray = Parser.init(testing.allocator, "1 $ 2"); + try testing.expectError(Error.UnexpectedToken, stray.parse()); +} diff --git a/engine/src/programmer.zig b/engine/src/programmer.zig index 1d56755..66e937f 100644 --- a/engine/src/programmer.zig +++ b/engine/src/programmer.zig @@ -16,9 +16,12 @@ const Signedness = Integer.Signedness; const parser_mod = @import("parser.zig"); const Parser = parser_mod.Parser; const bitwise = @import("bitwise.zig"); +const number_mod = @import("number.zig"); +const Number = number_mod.Number; /// What programmer-mode evaluation can fail with: the parse, the fixed-width -/// operators, and its own arithmetic and name errors. +/// operators, its own arithmetic and name errors, and the exact parse of a decimal +/// literal (`number_mod.Error`), since that is how a literal becomes a value here. pub const Error = error{ DivisionByZero, DomainError, @@ -27,7 +30,7 @@ pub const Error = error{ Overflow, UnknownFunction, UnknownVariable, -} || parser_mod.Error || bitwise.Error; +} || parser_mod.Error || bitwise.Error || number_mod.Error; /// What programmer mode needs to know: the integer type to compute in, and one /// display preference that does not affect arithmetic. @@ -48,29 +51,18 @@ pub const Config = struct { }; /// Evaluate an AST in programmer mode, producing an exact integer result. -pub fn evalProgrammer(config: Config, expr: *const Expr) Error!Integer { - return evalExpr(config, expr); +/// +/// The allocator is for parsing decimal literals, which go through the exact tier +/// so that `3.99` truncates the value the user wrote rather than an f64's +/// approximation of it. Nothing else here allocates. +pub fn evalProgrammer(allocator: Allocator, config: Config, expr: *const Expr) Error!Integer { + return evalExpr(allocator, config, expr); } /// Recursively evaluate an expression. -fn evalExpr(config: Config, expr: *const Expr) Error!Integer { +fn evalExpr(allocator: Allocator, config: Config, expr: *const Expr) Error!Integer { switch (expr.*) { - .number => |n| { - if (n.int_value) |int_val| { - return config.value(int_val); - } - // Float literal in programmer mode: truncate to integer. - // (Number literals are always non-negative; unary minus is a - // separate operator handled below.) - // - // The range check is not optional: `@intFromFloat` on an out-of-range - // value is illegal behaviour, and `tally -p '1e40'` aborted the process - // before this guard existed. - const float_value = n.float_value; - if (!std.math.isFinite(float_value) or float_value < 0) return Error.DomainError; - if (float_value >= 340282366920938463463374607431768211456.0) return Error.Overflow; - return config.value(@intFromFloat(float_value)); - }, + .number => |n| return literalToInteger(allocator, config, n), .string_literal => |text| { // Pack ASCII bytes into integer. // Big-endian packing: first char -> most significant used byte. @@ -90,15 +82,15 @@ fn evalExpr(config: Config, expr: *const Expr) Error!Integer { return Error.InvalidOperandType; }, .unary => |u| { - const operand = try evalExpr(config, u.operand); + const operand = try evalExpr(allocator, config, u.operand); return switch (u.op) { .negate => bitwise.negate(operand), .bitwise_not => bitwise.not(operand), }; }, .binary => |b| { - const left = try evalExpr(config, b.left); - const right = try evalExpr(config, b.right); + const left = try evalExpr(allocator, config, b.left); + const right = try evalExpr(allocator, config, b.right); return evalBinaryOp(b.op, left, right); }, .call => { @@ -108,6 +100,28 @@ fn evalExpr(config: Config, expr: *const Expr) Error!Integer { } } +/// A literal as a value of the configured integer type. +/// +/// A non-decimal literal is a bit pattern and converts directly. A decimal literal +/// is a number, so it is parsed exactly and then truncated toward zero: programmer +/// mode computes on integers, and `3.99` means 3. +/// +/// The truncation used to run through an f64, which cost precision on the way in +/// (`1e30` arrived as 1000000000000000019884624838656) and needed range guards +/// because `@intFromFloat` on an out-of-range value is illegal behaviour: `tally -p +/// '1e40'` aborted the process before those guards existed. Parsing exactly removes +/// both problems, since a value too wide for the width is simply `Overflow`. +fn literalToInteger(allocator: Allocator, config: Config, n: ast.Expr.Number) Error!Integer { + if (n.base != .decimal) return config.value(try n.base.parseDigits(n.text)); + + var parsed = try Number.parse(allocator, n.text); + defer parsed.deinit(); + // Literals are never negative here: unary minus is its own operator. + var whole = try Number.floor(allocator, parsed); + defer whole.deinit(); + return config.value(whole.asExactInt(u128) orelse return Error.Overflow); +} + /// Evaluate a binary operation on two values of the configured type. /// /// The bitwise operators, shifts and rotations are not here: they live in @@ -162,7 +176,7 @@ pub fn evalProgrammerString(allocator: Allocator, source: []const u8, config: Co // Same ownership rule as evalStringInfo: the tree is ours to release, and the // returned Integer does not borrow from it. defer parser_mod.freeExpr(allocator, expr); - return evalProgrammer(config, expr); + return evalProgrammer(allocator, config, expr); } // -- Tests -- @@ -430,7 +444,7 @@ test "prog: ASCII literal overflow 8-bit" { } test "prog: float literal truncates to integer" { - // 3.14 has no int_value, so the float branch truncates to 3 + // Programmer mode computes on integers, so a fractional literal truncates. const result = try testProg("3.14"); try testing.expectEqual(@as(u128, 3), result.unsignedValue()); } @@ -475,10 +489,11 @@ test "no leak: evalProgrammerString releases the parsed tree" { } } -test "programmer mode: a float literal out of range errors instead of aborting" { +test "programmer mode: a decimal literal truncates exactly, and past the width overflows" { const config: Config = .{}; - // `tally -p '1e40'` used to abort the process here: @intFromFloat on a value - // past u128 is illegal behaviour, and only 3.14 was ever tested. + // `tally -p '1e40'` used to abort the process: `@intFromFloat` on a value past + // u128 is illegal behaviour. There is no float on this path any more, so a + // literal too wide for the pattern is simply an overflow. try std.testing.expectError( Error.Overflow, evalProgrammerString(std.testing.allocator, "1e40", config), @@ -487,18 +502,39 @@ test "programmer mode: a float literal out of range errors instead of aborting" Error.Overflow, evalProgrammerString(std.testing.allocator, "1e100", config), ); - // Still truncates the values that do fit. - const small = try evalProgrammerString(std.testing.allocator, "3.99", config); - try std.testing.expectEqual(@as(u128, 3), small.unsignedValue()); - const large = try evalProgrammerString(std.testing.allocator, "1e30", config); - try std.testing.expect(large.unsignedValue() != 0); -} - -test "programmer mode: an infinite or NaN literal is a domain error" { - const config: Config = .{}; - // 10^400 overflows the exact tier's float projection to infinity. + // 10^400 is a 401-digit integer, which the exact tier parses without complaint + // and no width can hold. It used to report DomainError, because the f64 + // projection of it was infinity. try std.testing.expectError( - Error.DomainError, + Error.Overflow, evalProgrammerString(std.testing.allocator, "1e400", config), ); + + // Fractional literals still truncate toward zero: programmer mode computes on + // integers. + const small = try evalProgrammerString(std.testing.allocator, "3.99", config); + try std.testing.expectEqual(@as(u128, 3), small.unsignedValue()); + + // And the truncation is of the value written, not of an f64 of it. Through the + // old float path this arrived as 1000000000000000019884624838656. + const exact = try evalProgrammerString(std.testing.allocator, "1e30", .{ .width = .bits128 }); + try std.testing.expectEqual(@as(u128, 1_000_000_000_000_000_000_000_000_000_000), exact.unsignedValue()); +} + +test "programmer mode: the full width is reachable from every base" { + // Open item 11: the tokenizer's `int_value` was a u64, so a literal wider than + // 64 bits was "invalid number" in a mode whose widest type is 128 bits. + const config: Config = .{ .width = .bits128, .signedness = .unsigned }; + const cases = [_][]const u8{ + "0xFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFF", + "340282366920938463463374607431768211455", + "0o3777777777777777777777777777777777777777777", + }; + for (cases) |source| { + const result = try evalProgrammerString(std.testing.allocator, source, config); + try std.testing.expectEqual(std.math.maxInt(u128), result.unsignedValue()); + } + + const seventeen_nibbles = try evalProgrammerString(std.testing.allocator, "0x1FFFFFFFFFFFFFFFF", config); + try std.testing.expectEqual(@as(u128, 0x1FFFFFFFFFFFFFFFF), seventeen_nibbles.unsignedValue()); } diff --git a/engine/src/tokenizer.zig b/engine/src/tokenizer.zig index 7264e8b..b247784 100644 --- a/engine/src/tokenizer.zig +++ b/engine/src/tokenizer.zig @@ -15,6 +15,56 @@ pub const Base = enum { hex, octal, binary, + + /// Characters that separate digits without being digits (FR-1.8). + fn isSeparator(ch: u8) bool { + return ch == '_' or ch == ',' or ch == ' '; + } + + pub fn radix(self: Base) u8 { + return switch (self) { + .decimal => 10, + .hex => 16, + .octal => 8, + .binary => 2, + }; + } + + /// The prefix that selects this base in source: "0x", "0o", "0b", or nothing. + pub fn prefix(self: Base) []const u8 { + return switch (self) { + .decimal => "", + .hex => "0x", + .octal => "0o", + .binary => "0b", + }; + } + + /// The value of a literal written in this base. + /// + /// `text` is a literal's source text exactly as the tokenizer produced it, + /// prefix and separators included: "0xFF_FF", "0b1010", "1 000". Separators are + /// skipped rather than stripped into a buffer, so length is not a limit. + /// + /// The only failure is a value too wide for `u128`, which is the widest pattern + /// `Integer` holds and the widest FR-1.9 shows. A decimal value that needs more + /// room than that belongs to the exact tier, which parses the same text into a + /// rational instead; see `Number.parse`. + /// + /// Asserts the text is digits of this base and separators. Every caller passes + /// a span the tokenizer just lexed, which cannot contain anything else. + pub fn parseDigits(self: Base, text: []const u8) error{Overflow}!u128 { + const digits = text[self.prefix().len..]; + var value: u128 = 0; + for (digits) |ch| { + if (isSeparator(ch)) continue; + const digit = std.fmt.charToDigit(ch, self.radix()) catch + @panic("literal text carries a digit outside its own base"); + value = try std.math.mul(u128, value, self.radix()); + value = try std.math.add(u128, value, digit); + } + return value; + } }; pub const TokenKind = enum { @@ -48,6 +98,10 @@ pub const TokenKind = enum { // Special eof, invalid, + /// A numeric span that is not a number: a base prefix with no digits after it, + /// or an exponent with no digits after it. Distinct from `invalid` so the parser + /// can say "invalid number" for `0x` and "unexpected token" for `$`. + invalid_number, // String literal (single-quoted, for ASCII byte packing in programmer mode) string_literal, }; @@ -58,6 +112,9 @@ pub const Token = struct { start: usize, /// Byte length of this token in the source. len: usize, + /// For a `number`, the base its prefix selected. The tokenizer knows this while + /// it scans, so nothing downstream has to re-read the prefix to find out. + base: Base = .decimal, /// Extract the token's text from the source. pub fn text(self: Token, source: []const u8) []const u8 { @@ -65,81 +122,6 @@ pub const Token = struct { } }; -/// Parsed number value from a token. -pub const NumberValue = struct { - float: f64, - /// If the number is a pure integer (no decimal point, no exponent), this - /// holds the exact integer value. - int_value: ?u64, - base: Base, -}; - -/// Parse a number token's text into a value. -/// Handles 0x (hex), 0o (octal), 0b (binary), decimal integers, and floats. -/// Underscores are ignored as digit separators. -pub fn parseNumber(token_text: []const u8) !NumberValue { - // Strip underscores for parsing - var buf: [128]u8 = undefined; - var buf_len: usize = 0; - for (token_text) |c| { - if (c != '_' and c != ',' and c != ' ') { - if (buf_len >= buf.len) return error.InvalidNumber; - buf[buf_len] = c; - buf_len += 1; - } - } - const clean = buf[0..buf_len]; - - if (clean.len == 0) return error.InvalidNumber; - - // Check base prefix - if (clean.len >= 2 and clean[0] == '0') { - switch (clean[1]) { - 'x', 'X' => { - const digits = clean[2..]; - if (digits.len == 0) return error.InvalidNumber; - const val = std.fmt.parseInt(u64, digits, 16) catch return error.InvalidNumber; - return .{ .float = @floatFromInt(val), .int_value = val, .base = .hex }; - }, - 'o', 'O' => { - const digits = clean[2..]; - if (digits.len == 0) return error.InvalidNumber; - const val = std.fmt.parseInt(u64, digits, 8) catch return error.InvalidNumber; - return .{ .float = @floatFromInt(val), .int_value = val, .base = .octal }; - }, - 'b', 'B' => { - const digits = clean[2..]; - if (digits.len == 0) return error.InvalidNumber; - const val = std.fmt.parseInt(u64, digits, 2) catch return error.InvalidNumber; - return .{ .float = @floatFromInt(val), .int_value = val, .base = .binary }; - }, - else => {}, - } - } - - // Check if it's a pure integer (no '.', no 'e'/'E') - var is_integer = true; - for (clean) |c| { - if (c == '.' or c == 'e' or c == 'E') { - is_integer = false; - break; - } - } - - if (is_integer) { - const val = std.fmt.parseInt(u64, clean, 10) catch { - // Could be too large for u64, try as float - const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber; - return .{ .float = f, .int_value = null, .base = .decimal }; - }; - return .{ .float = @floatFromInt(val), .int_value = val, .base = .decimal }; - } - - // Float - const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber; - return .{ .float = f, .int_value = null, .base = .decimal }; -} - // -- Raw Tokenizer -- /// Tokenizing is mode-independent: `0xFF`, `<<` and `'A'` are recognized in both @@ -248,24 +230,14 @@ pub const Tokenizer = struct { if (self.source[self.pos] == '0' and self.pos + 1 < self.source.len) { const next_ch = self.source[self.pos + 1]; switch (next_ch) { - 'x', 'X' => { - self.pos += 2; - self.consumeBaseDigits(isHexDigit); - return .{ .kind = .number, .start = start, .len = self.pos - start }; - }, - 'o', 'O' => { - self.pos += 2; - self.consumeBaseDigits(isOctalDigit); - return .{ .kind = .number, .start = start, .len = self.pos - start }; - }, + 'x', 'X' => return self.readBaseDigits(start, .hex, isHexDigit), + 'o', 'O' => return self.readBaseDigits(start, .octal, isOctalDigit), 'b', 'B' => { // Disambiguate: 0b... is binary only if followed by 0 or 1 if (self.pos + 2 < self.source.len and (self.source[self.pos + 2] == '0' or self.source[self.pos + 2] == '1')) { - self.pos += 2; - self.consumeBaseDigits(isBinaryDigit); - return .{ .kind = .number, .start = start, .len = self.pos - start }; + return self.readBaseDigits(start, .binary, isBinaryDigit); } // Otherwise fall through to decimal }, @@ -296,12 +268,34 @@ pub const Tokenizer = struct { { self.pos += 1; } + const exponent_start = self.pos; self.consumeDigits(isDecDigit); + // "1e" and "1e+" are not numbers. Rejected while the span is being + // scanned, because this is where it is known: nothing downstream can + // tell "1e" from a number without re-lexing it. + if (self.pos == exponent_start) { + return .{ .kind = .invalid_number, .start = start, .len = self.pos - start }; + } } return .{ .kind = .number, .start = start, .len = self.pos - start }; } + /// A prefixed literal: `0x`, `0o` or `0b` and the digits of that base. A prefix + /// with no digits after it ("0x", "0o ") is not a number. + fn readBaseDigits( + self: *Tokenizer, + start: usize, + base: Base, + predicate: *const fn (u8) bool, + ) Token { + self.pos += 2; + const digits_start = self.pos; + self.consumeBaseDigits(predicate); + const kind: TokenKind = if (self.pos == digits_start) .invalid_number else .number; + return .{ .kind = kind, .start = start, .len = self.pos - start, .base = base }; + } + fn consumeDigits(self: *Tokenizer, predicate: *const fn (u8) bool) void { while (self.pos < self.source.len) { const ch = self.source[self.pos]; @@ -664,73 +658,88 @@ test "tokenize base literal still groups with spaces and underscores" { try testing.expectEqualStrings("0xFF_FF", underscored.next().text("0xFF_FF")); } -test "parseNumber huge decimal exceeds u64 and falls back to float here" { - // The tokenizer's own integer channel is a u64, so a value this large has no - // `int_value` at this layer. That is NOT a precision limit of the engine: - // the evaluator re-parses the literal text into an exact rational (see - // evaluator.literalToNumber), so `99999999999999999999999999` still - // evaluates exactly. This test pins the tokenizer's contract only. - const result = try parseNumber("99999999999999999999999999"); - try testing.expectEqual(@as(?u64, null), result.int_value); - try testing.expectEqual(Base.decimal, result.base); - try testing.expect(result.float > 1e25); +// -- Base.parseDigits tests -- +// +// This replaced `parseNumber`, which built a `NumberValue{ float: f64, int_value: +// ?u64, base }` for every literal in a fixed 128-byte buffer. That capped a literal +// at 128 characters and a non-decimal one at 64 bits, and made `Rational`'s +// `TooManyDigits` (512 digits) unreachable from any frontend. The AST now carries the +// literal's text, and each consumer parses it with the precision it works in. + +fn expectDigits(expected: u128, base: Base, text: []const u8) !void { + try testing.expectEqual(expected, try base.parseDigits(text)); } -// -- parseNumber tests -- - -test "parseNumber decimal integer" { - const result = try parseNumber("42"); - try testing.expectEqual(@as(f64, 42.0), result.float); - try testing.expectEqual(@as(?u64, 42), result.int_value); - try testing.expectEqual(Base.decimal, result.base); +test "parseDigits: each base, prefix skipped" { + try expectDigits(42, .decimal, "42"); + try expectDigits(255, .hex, "0xFF"); + try expectDigits(10, .binary, "0b1010"); + try expectDigits(511, .octal, "0o777"); + try expectDigits(0, .decimal, "0"); } -test "parseNumber decimal with underscores" { - const result = try parseNumber("1_000_000"); - try testing.expectEqual(@as(?u64, 1_000_000), result.int_value); +test "parseDigits: separators are skipped, not stripped into a buffer" { + try expectDigits(1_000_000, .decimal, "1_000_000"); + try expectDigits(1_000_000, .decimal, "1,000,000"); + try expectDigits(0xFFFF, .hex, "0xFF_FF"); + try expectDigits(0xFFFF, .hex, "0xFF FF"); + // 128 characters was the old buffer's limit. Length is no longer a limit at + // all, so a literal made almost entirely of separators still parses. + try expectDigits(1, .decimal, "1" ++ ("_" ** 400)); } -test "parseNumber hex" { - const result = try parseNumber("0xFF"); - try testing.expectEqual(@as(?u64, 255), result.int_value); - try testing.expectEqual(Base.hex, result.base); +test "parseDigits: the whole width is reachable, which u64 could not do" { + // The literals open item 11 named: 17 hex nibbles and the top of the range. + try expectDigits(0x1FFFFFFFFFFFFFFFF, .hex, "0x1FFFFFFFFFFFFFFFF"); + try expectDigits(std.math.maxInt(u128), .hex, "0xFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFF"); + try expectDigits(std.math.maxInt(u128), .decimal, "340282366920938463463374607431768211455"); + try expectDigits(1 << 127, .binary, "0b1" ++ ("0" ** 127)); } -test "parseNumber binary" { - const result = try parseNumber("0b1010"); - try testing.expectEqual(@as(?u64, 10), result.int_value); - try testing.expectEqual(Base.binary, result.base); +test "parseDigits: past u128 is Overflow, not a float" { + try testing.expectError(error.Overflow, Base.hex.parseDigits("0x1" ++ ("0" ** 32))); + try testing.expectError( + error.Overflow, + Base.decimal.parseDigits("340282366920938463463374607431768211456"), + ); } -test "parseNumber octal" { - const result = try parseNumber("0o777"); - try testing.expectEqual(@as(?u64, 511), result.int_value); - try testing.expectEqual(Base.octal, result.base); +// -- Spans that are not numbers -- + +test "tokenize a base prefix with no digits is not a number" { + // `parseNumber` used to catch these after the fact. The scan knows it consumed + // no digits, so it says so, and the parser reports "invalid number" rather than + // "unexpected token". + for ([_][]const u8{ "0x", "0o", "0x ", "0x+1" }) |source| { + var tok = Tokenizer.init(source); + try testing.expectEqual(TokenKind.invalid_number, tok.next().kind); + } + // `0b` without a 0 or 1 after it is a zero and an identifier, which is the + // existing disambiguation rule and not an invalid literal. + var binary = Tokenizer.init("0b"); + try testing.expectEqual(TokenKind.number, binary.next().kind); + try testing.expectEqual(TokenKind.identifier, binary.next().kind); } -test "parseNumber float" { - const result = try parseNumber("3.14"); - try testing.expectApproxEqAbs(@as(f64, 3.14), result.float, 1e-10); - try testing.expectEqual(@as(?u64, null), result.int_value); - try testing.expectEqual(Base.decimal, result.base); +test "tokenize an exponent with no digits is not a number" { + for ([_][]const u8{ "1e", "1E", "1e+", "1e-", "1.5e" }) |source| { + var tok = Tokenizer.init(source); + try testing.expectEqual(TokenKind.invalid_number, tok.next().kind); + } } -test "parseNumber float with exponent" { - const result = try parseNumber("1.5e10"); - try testing.expectEqual(@as(f64, 1.5e10), result.float); - try testing.expectEqual(@as(?u64, null), result.int_value); -} - -test "parseNumber hex with underscores" { - const result = try parseNumber("0xFF_FF"); - try testing.expectEqual(@as(?u64, 0xFFFF), result.int_value); - try testing.expectEqual(Base.hex, result.base); -} - -test "parseNumber with commas" { - const result = try parseNumber("1,000,000"); - try testing.expectEqual(@as(?u64, 1_000_000), result.int_value); - try testing.expectEqual(Base.decimal, result.base); +test "tokenize records the base while it scans" { + const cases = [_]struct { source: []const u8, base: Base }{ + .{ .source = "42", .base = .decimal }, + .{ .source = "3.14", .base = .decimal }, + .{ .source = "0xFF", .base = .hex }, + .{ .source = "0o777", .base = .octal }, + .{ .source = "0b1010", .base = .binary }, + }; + for (cases) |c| { + var tok = Tokenizer.init(c.source); + try testing.expectEqual(c.base, tok.next().base); + } } // -- ImplicitMulStream tests --