human review: Integer.zig/ast.zig/tokenizer.zig

This commit is contained in:
Emil Lerch 2026-08-06 10:10:17 -07:00
parent 8da691d455
commit 951788d9d4
Signed by: lobo
GPG key ID: A7B62D657EF764F8
8 changed files with 406 additions and 244 deletions

View file

@ -144,17 +144,24 @@ pub const Expr = union(enum) {
};
pub const Number = struct {
float_value: f64,
/// For programmer mode: the exact integer when the literal fits u64.
int_value: ?u64,
base: Base,
/// The literal exactly as written, pointing into the source. Decimal literals
/// are re-parsed from this rather than from `float_value`, because
/// `float_value` has already rounded and `0.1` cannot be recovered from it.
/// The literal exactly as written, pointing into the source. The only
/// representation kept, because it is the only lossless one: each reader parses
/// it at the precision it works in (`Number.parse` for the exact tier,
/// `Base.parseDigits` for a fixed-width pattern). This node used to carry an
/// `f64` and a `?u64` computed up front, which capped a literal at 128
/// characters and 64 bits; see Task 5.18.
text: []const u8,
};
pub const Base = enum { decimal, hex, octal, binary };
pub const Base = enum {
decimal,
hex,
octal,
binary,
/// The literal's digits as a value, separators skipped, `Overflow` past u128.
pub fn parseDigits(self: Base, text: []const u8) error{Overflow}!u128 { ... }
};
pub const Binary = struct {
op: BinaryOp,

View file

@ -914,6 +914,62 @@ plans a `--raw` flag, but today it is exercised only by tests.
100% line coverage, engine 99.44%. CLI output byte-identical across all five rows,
both byte orders, ASCII packing and the multi-base standard-mode view.
### Task 5.18: A literal is its text; each reader parses it at its own precision
The tokenizer computed two lossy representations of every numeric literal up front,
in a fixed 128-byte buffer: an `f64` and a `?u64`. Both went into the AST node
alongside the source text, and every limit in the pipeline came from that decision.
Five reachable defects, one cause:
```
600-digit literal error: invalid number now: too many digits
130-character literal error: invalid number now: parses exactly
1e100001 inf now: exponent too large
0x1FFFFFFFFFFFFFFFF, programmer mode error: invalid number now: the value
u128 max in any base, programmer mode error / overflow now: the value
```
`Rational.parseDecimal` accepts 512 digits and reports `TooManyDigits`, and no
frontend could reach either, because a literal over 128 characters died in the
tokenizer first with the wrong message. `int_value` was a `u64` in a mode whose widest
type is 128 bits, so a 17-nibble hex literal was "invalid number". And
`evaluator.literalToNumber` swallowed every `Number.parse` error and fell back to the
f64, so `1e100001` printed `inf` instead of saying the exponent was out of range.
- `ast.Expr.Number` is now `{ base, text }`. The text is the only lossless form of a
literal, so it is the only one kept.
- `Base` gained `parseDigits`, which walks the literal's own text skipping separators
and accumulating into a `u128` with checked arithmetic. No buffer, so length is not
a limit; `Overflow` past 128 bits, which is the widest pattern `Integer` holds.
- The exact tier parses decimals with `Number.parse` and propagates the error.
- Programmer mode parses decimals through the exact tier too, then truncates toward
zero. It has no float path at all now, which removes the `@intFromFloat` range
guards (illegal behaviour past u128, which is what aborted `tally -p '1e40'`) and
makes the truncation exact: `1e30` used to arrive as
1000000000000000019884624838656.
- Validity moved into `readNumber`, which knows it consumed no digits after a `0x` or
an `e`. `TokenKind.invalid_number` carries that, so the parser still says "invalid
number" for `0x` and "unexpected token" for `$` without re-deriving either from the
text.
- `Token` carries the base the scan recognised, so nothing downstream re-reads the
prefix.
`parseNumber`, `NumberValue`, the 128-byte buffer, the u64 cap and the f64 fallback
are all gone.
A latent crash surfaced once literals could be that wide. `Integer.signedValue`
consulted `signedness` and `@intCast` the pattern when it said unsigned, so a 128-bit
unsigned value with the top bit set panicked ("integer does not fit in destination
type") as soon as anything printed it, which every display path does. It was
unreachable only because such a literal could not be entered. Reading a pattern as
two's complement is a `@bitCast` and cannot fail, and it is what
`Notation.decimal_signed` documented all along: the design mockups show
`DEC 4,294,967,295` beside `SIGN -1`, two readings of one pattern, where the old
implementation made those rows identical for every unsigned value. `isNegative` still
consults `signedness`, because that is the arithmetic question the shift operators
ask.
### Task 5.15: A value renders itself
`formatter.zig` had six functions taking `(buf, value: u128, bit_width, endian)`,
@ -1196,9 +1252,9 @@ STILL OPEN, in the order I would take them:
Fixed by the de-duplication pass above.
10. Assignment parses in prefix position (`1 + x = 2` mutates `x`), and assignment
to a constant name is silently discarded.
11. Literals longer than 128 characters are rejected by a fixed tokenizer buffer,
11. ~~Literals longer than 128 characters are rejected by a fixed tokenizer buffer,
and non-decimal literals are capped at 64 bits, both below what the exact tier
supports.
supports.~~ Fixed by Task 5.18.
12. `engine/src/c_api.zig` has no tests and appears in no coverage report, because
no test target builds the shared library. Deferred until Phase 6 gives it a
caller; recorded here so it is a known gap rather than an oversight.

View file

@ -29,17 +29,33 @@ signedness: Signedness = .signed,
/// negative range, so `>>` is a zero fill there and a large shift distance is large
/// rather than negative.
pub fn isNegative(self: Integer) bool {
const sign_bit = @as(u128, 1) << @intCast(self.width.bits() - 1);
return self.signedness == .signed and self.unsignedValue() & sign_bit != 0;
return self.signedness == .signed and self.unsignedValue() & self.signBit() != 0;
}
/// Interpret as a signed value, sign-extended from the width.
/// The two's complement reading of the pattern, sign-extended from the width.
///
/// Independent of `signedness`, unlike `isNegative`: the programmer view shows both
/// readings of one pattern side by side (design.md's DEC and SIGN rows), and the
/// point of showing both is that they differ. An unsigned 8-bit `0xFF` is 255 read
/// as unsigned and -1 read as signed. What `signedness` decides is what the pattern
/// *means* to arithmetic, which is what `isNegative` answers.
///
/// This used to consult `signedness` and `@intCast` the pattern when it said
/// unsigned, which made the two decimal rows identical for every unsigned value and
/// panicked outright on a 128-bit unsigned pattern with the top bit set, since no
/// such value fits an i128. Reading the pattern is always a `@bitCast`, so it cannot
/// fail.
pub fn signedValue(self: Integer) i128 {
const pattern = self.unsignedValue();
if (self.isNegative()) return @bitCast(pattern | ~self.width.mask());
if (pattern & self.signBit() != 0) return @bitCast(pattern | ~self.width.mask());
return @intCast(pattern);
}
/// The mask of the top bit at this width, which is the sign bit of a signed value.
fn signBit(self: Integer) u128 {
return @as(u128, 1) << @intCast(self.width.bits() - 1);
}
/// Interpret as an unsigned value.
///
/// Also the bit pattern itself, truncated to the width, since an unsigned reading of
@ -324,12 +340,18 @@ test "the default is 64-bit signed, which is what standard mode uses" {
try testing.expectEqual(@as(i128, -1), value.signedValue());
}
test "signedValue reads the top bit only when the value is signed" {
test "signedValue reads the pattern, not the configured signedness" {
const signed: Integer = .{ .raw = 0xFF, .width = .bits8 };
const unsigned: Integer = .{ .raw = 0xFF, .width = .bits8, .signedness = .unsigned };
// One pattern, two readings. Both rows of the programmer view come from here,
// so an unsigned value's signed reading has to be the signed one.
try testing.expectEqual(@as(i128, -1), signed.signedValue());
try testing.expectEqual(@as(i128, 255), unsigned.signedValue());
try testing.expectEqual(@as(i128, -1), unsigned.signedValue());
try testing.expectEqual(@as(u128, 255), unsigned.unsignedValue());
// What signedness decides is meaning, not rendering: only a signed value has a
// negative range, which is what the shift operators consult.
try testing.expect(signed.isNegative());
try testing.expect(!unsigned.isNegative());
@ -338,9 +360,20 @@ test "signedValue reads the top bit only when the value is signed" {
const max: Integer = .{ .raw = 0x7F, .width = .bits8 };
try testing.expectEqual(@as(i128, 127), max.signedValue());
// A 128-bit value has no bits above the width to fill.
const wide: Integer = .{ .raw = BitWidth.bits128.mask(), .width = .bits128 };
// A 128-bit unsigned pattern with the top bit set has no i128 to be cast to,
// and reading it used to panic. It is -1 read as two's complement, and every
// display path asks for that reading.
const wide: Integer = .{
.raw = std.math.maxInt(u128),
.width = .bits128,
.signedness = .unsigned,
};
try testing.expectEqual(@as(i128, -1), wide.signedValue());
try testing.expectEqual(std.math.maxInt(u128), wide.unsignedValue());
// The signed-by-configuration case of the same pattern.
const wide_signed: Integer = .{ .raw = BitWidth.bits128.mask(), .width = .bits128 };
try testing.expectEqual(@as(i128, -1), wide_signed.signedValue());
}
test "bits above the width never reach an interpretation" {
@ -363,8 +396,12 @@ test "unsignedValue masks correctly" {
test "the same bits read two ways, which is why the width travels with the value" {
const signed: Integer = .{ .raw = 0xFF, .width = .bits8 };
const unsigned: Integer = .{ .raw = 0xFF, .width = .bits8, .signedness = .unsigned };
// The two readings of one pattern are the pattern's, so both values give both.
// What differs is `isNegative`, which is the arithmetic meaning.
try testing.expectEqual(@as(i128, -1), signed.signedValue());
try testing.expectEqual(@as(i128, 255), unsigned.signedValue());
try testing.expectEqual(@as(i128, -1), unsigned.signedValue());
try testing.expectEqual(@as(u128, 255), signed.unsignedValue());
try testing.expectEqual(@as(u128, 255), unsigned.unsignedValue());
}
test "withRaw keeps the type and masks the new pattern" {

View file

@ -24,18 +24,19 @@ pub const Expr = union(enum) {
assignment: Assignment,
pub const Number = struct {
float_value: f64,
/// If the number is a pure integer, stores the exact value.
int_value: ?u64,
/// The base the literal was written in, for the multi-base display
/// (FR-1.9) and to tell the readers below which parse applies.
base: Base,
/// The literal's source text, with separators intact. Points into the
/// expression source, which outlives the AST.
/// The literal's source text, prefix and separators intact. Points into
/// the expression source, which outlives the AST.
///
/// Kept so the evaluator can reconstruct the literal *exactly* as a
/// rational: `float_value` has already lost information by the time it
/// exists (0.1 is not representable in binary), so it cannot be the
/// basis for exact arithmetic. See design.md 2.7.
text: []const u8 = "",
/// The only representation kept, because it is the only lossless one. This
/// node used to carry an `f64` and a `?u64` that the tokenizer computed up
/// front, which capped a literal at 128 characters and 64 bits and made
/// `TooManyDigits` unreachable. Each consumer now parses the text with the
/// precision it actually works in: `Number.parse` for the exact tier,
/// `Base.parseDigits` for a fixed-width pattern. See design.md 2.7.
text: []const u8,
};
pub const Unary = struct {

View file

@ -197,25 +197,20 @@ fn evalExact(env: *Environment, scratch: Allocator, expr: *const Expr) Error!Num
}
}
/// Turn a literal into a Number, exactly where possible.
/// Turn a literal into a Number, exactly.
///
/// Decimal literals are re-parsed from their source text rather than taken from
/// `float_value`, because `float_value` has already rounded: `0.1` cannot be
/// recovered from its binary approximation.
/// Both bases parse the literal's own source text, because that is the only
/// lossless form of it: `0.1` cannot be recovered from the nearest f64, and
/// `0x1FFFFFFFFFFFFFFFF` does not fit the u64 the tokenizer used to precompute.
///
/// Failures propagate. This used to swallow every `Number.parse` error and fall
/// back to a float, so `1e100001` printed `inf` instead of reporting that the
/// exponent was out of range, and the exact tier silently became the inexact one.
fn literalToNumber(scratch: Allocator, n: ast.Expr.Number) Error!Number {
if (n.base == .decimal and n.text.len > 0) {
if (Number.parse(scratch, n.text)) |value| return value else |_| {
// Fall through to the approximations below rather than failing: the
// tokenizer already accepted this text, so a parse mismatch here
// should degrade, not error.
}
}
// Non-decimal literals are integers; use the exact integer the tokenizer
// recovered when it fits, otherwise accept the float approximation.
if (n.int_value) |int_val| {
return try Number.fromInt(scratch, int_val);
}
return Number.fromFloat(n.float_value);
if (n.base == .decimal) return Number.parse(scratch, n.text);
// A non-decimal literal is a bit pattern, so it is an integer of at most the
// 128 bits `Integer` holds. Wider than that is `Overflow`, not a float.
return Number.fromInt(scratch, try n.base.parseDigits(n.text));
}
/// Evaluate a binary operation.

View file

@ -19,7 +19,6 @@ const tokenizer_mod = @import("tokenizer.zig");
const Tokenizer = tokenizer_mod.Tokenizer;
const TokenKind = tokenizer_mod.TokenKind;
const Token = tokenizer_mod.Token;
const parseNumber = tokenizer_mod.parseNumber;
/// What parsing can fail with.
///
@ -143,15 +142,14 @@ pub const Parser = struct {
switch (tok.kind) {
.number => {
self.advance();
const text = tok.text(self.source);
const num = parseNumber(text) catch return Error.InvalidNumber;
return self.makeNode(.{ .number = .{
.float_value = num.float,
.int_value = num.int_value,
.base = num.base,
.text = text,
.base = tok.base,
.text = tok.text(self.source),
} });
},
.invalid_number => {
return Error.InvalidNumber;
},
.string_literal => {
self.advance();
const text = tok.text(self.source);
@ -382,6 +380,16 @@ pub const Parser = struct {
const testing = std.testing;
/// A literal node's captured source text.
///
/// The node no longer carries a precomputed `f64` and `?u64`, so what a parser test
/// can assert about a literal is the span it captured and the base it recognised.
/// What the text means belongs to `number.zig` and `Base.parseDigits`, and is tested
/// there against the full width and precision they support.
fn expectLiteral(expected: []const u8, expr: *const Expr) !void {
try testing.expectEqualStrings(expected, expr.number.text);
}
// Error-path tests used to need an arena because a failed parse leaked its
// partial tree. They no longer do (the parser cleans up after itself), but the
// arena helper is kept for the tests already written against it.
@ -425,22 +433,22 @@ pub fn freeExpr(allocator: Allocator, expr: *Expr) void {
test "parse simple number" {
const expr = try testParse("42");
defer freeExpr(testing.allocator, expr);
try testing.expectEqual(@as(f64, 42.0), expr.number.float_value);
try testing.expectEqual(@as(?u64, 42), expr.number.int_value);
try expectLiteral("42", expr);
}
test "parse hex number" {
const expr = try testParse("0xFF");
defer freeExpr(testing.allocator, expr);
try testing.expectEqual(@as(?u64, 255), expr.number.int_value);
try expectLiteral("0xFF", expr);
try testing.expectEqual(tokenizer_mod.Base.hex, expr.number.base);
}
test "parse addition" {
const expr = try testParse("2 + 3");
defer freeExpr(testing.allocator, expr);
try testing.expectEqual(BinaryOp.add, expr.binary.op);
try testing.expectEqual(@as(f64, 2.0), expr.binary.left.number.float_value);
try testing.expectEqual(@as(f64, 3.0), expr.binary.right.number.float_value);
try expectLiteral("2", expr.binary.left);
try expectLiteral("3", expr.binary.right);
}
test "parse precedence: mul before add" {
@ -448,7 +456,7 @@ test "parse precedence: mul before add" {
const expr = try testParse("2 + 3 * 4");
defer freeExpr(testing.allocator, expr);
try testing.expectEqual(BinaryOp.add, expr.binary.op);
try testing.expectEqual(@as(f64, 2.0), expr.binary.left.number.float_value);
try expectLiteral("2", expr.binary.left);
try testing.expectEqual(BinaryOp.mul, expr.binary.right.binary.op);
}
@ -457,7 +465,7 @@ test "parse precedence: power right-associative" {
const expr = try testParse("2^3^4");
defer freeExpr(testing.allocator, expr);
try testing.expectEqual(BinaryOp.pow, expr.binary.op);
try testing.expectEqual(@as(f64, 2.0), expr.binary.left.number.float_value);
try expectLiteral("2", expr.binary.left);
try testing.expectEqual(BinaryOp.pow, expr.binary.right.binary.op);
}
@ -465,7 +473,7 @@ test "parse unary negation" {
const expr = try testParse("-5");
defer freeExpr(testing.allocator, expr);
try testing.expectEqual(UnaryOp.negate, expr.unary.op);
try testing.expectEqual(@as(f64, 5.0), expr.unary.operand.number.float_value);
try expectLiteral("5", expr.unary.operand);
}
test "parse negation in expression" {
@ -489,7 +497,7 @@ test "parse function call" {
defer freeExpr(testing.allocator, expr);
try testing.expectEqualStrings("sin", expr.call.name);
try testing.expectEqual(@as(usize, 1), expr.call.args.len);
try testing.expectApproxEqAbs(@as(f64, 3.14), expr.call.args[0].number.float_value, 1e-10);
try expectLiteral("3.14", expr.call.args[0]);
}
test "parse multi-arg function call" {
@ -509,7 +517,7 @@ test "parse assignment" {
const expr = try testParse("X = 42");
defer freeExpr(testing.allocator, expr);
try testing.expectEqualStrings("X", expr.assignment.name);
try testing.expectEqual(@as(f64, 42.0), expr.assignment.value.number.float_value);
try expectLiteral("42", expr.assignment.value);
}
test "parse adjacent number and identifier is an error (no implicit mul)" {
@ -668,6 +676,8 @@ test "a failed parse leaves nothing allocated" {
"", // empty input
"1 + 2 3", // trailing token after a binary expression
"-(1 + ", // nested failure under a unary
"0x", // a base prefix with no digits
"1e", // an exponent with no digits
};
for (bad) |source| {
var parser = Parser.init(testing.allocator, source);
@ -804,7 +814,7 @@ test "nesting within the depth limit parses" {
var parser = Parser.init(testing.allocator, deep.items);
const expr = try parser.parse();
defer freeExpr(testing.allocator, expr);
try testing.expectEqual(@as(f64, 7.0), expr.number.float_value);
try expectLiteral("7", expr);
}
test "unbalanced deep nesting is rejected without leaking the partial tree" {
@ -835,3 +845,14 @@ test "the depth limit also covers nested calls and unary operators" {
var unary_parser = Parser.init(testing.allocator, unary.items);
try testing.expectError(Error.InvalidExpression, unary_parser.parse());
}
test "a numeric span that is not a number is an invalid number, not an unexpected token" {
// The tokenizer marks these while it scans, and the distinction is the error
// the user sees: "invalid number" for `0x`, "unexpected token" for `$`.
for ([_][]const u8{ "0x", "0o", "1e", "1e+", "2 + 0x" }) |source| {
var parser = Parser.init(testing.allocator, source);
try testing.expectError(Error.InvalidNumber, parser.parse());
}
var stray = Parser.init(testing.allocator, "1 $ 2");
try testing.expectError(Error.UnexpectedToken, stray.parse());
}

View file

@ -16,9 +16,12 @@ const Signedness = Integer.Signedness;
const parser_mod = @import("parser.zig");
const Parser = parser_mod.Parser;
const bitwise = @import("bitwise.zig");
const number_mod = @import("number.zig");
const Number = number_mod.Number;
/// What programmer-mode evaluation can fail with: the parse, the fixed-width
/// operators, and its own arithmetic and name errors.
/// operators, its own arithmetic and name errors, and the exact parse of a decimal
/// literal (`number_mod.Error`), since that is how a literal becomes a value here.
pub const Error = error{
DivisionByZero,
DomainError,
@ -27,7 +30,7 @@ pub const Error = error{
Overflow,
UnknownFunction,
UnknownVariable,
} || parser_mod.Error || bitwise.Error;
} || parser_mod.Error || bitwise.Error || number_mod.Error;
/// What programmer mode needs to know: the integer type to compute in, and one
/// display preference that does not affect arithmetic.
@ -48,29 +51,18 @@ pub const Config = struct {
};
/// Evaluate an AST in programmer mode, producing an exact integer result.
pub fn evalProgrammer(config: Config, expr: *const Expr) Error!Integer {
return evalExpr(config, expr);
///
/// The allocator is for parsing decimal literals, which go through the exact tier
/// so that `3.99` truncates the value the user wrote rather than an f64's
/// approximation of it. Nothing else here allocates.
pub fn evalProgrammer(allocator: Allocator, config: Config, expr: *const Expr) Error!Integer {
return evalExpr(allocator, config, expr);
}
/// Recursively evaluate an expression.
fn evalExpr(config: Config, expr: *const Expr) Error!Integer {
fn evalExpr(allocator: Allocator, config: Config, expr: *const Expr) Error!Integer {
switch (expr.*) {
.number => |n| {
if (n.int_value) |int_val| {
return config.value(int_val);
}
// Float literal in programmer mode: truncate to integer.
// (Number literals are always non-negative; unary minus is a
// separate operator handled below.)
//
// The range check is not optional: `@intFromFloat` on an out-of-range
// value is illegal behaviour, and `tally -p '1e40'` aborted the process
// before this guard existed.
const float_value = n.float_value;
if (!std.math.isFinite(float_value) or float_value < 0) return Error.DomainError;
if (float_value >= 340282366920938463463374607431768211456.0) return Error.Overflow;
return config.value(@intFromFloat(float_value));
},
.number => |n| return literalToInteger(allocator, config, n),
.string_literal => |text| {
// Pack ASCII bytes into integer.
// Big-endian packing: first char -> most significant used byte.
@ -90,15 +82,15 @@ fn evalExpr(config: Config, expr: *const Expr) Error!Integer {
return Error.InvalidOperandType;
},
.unary => |u| {
const operand = try evalExpr(config, u.operand);
const operand = try evalExpr(allocator, config, u.operand);
return switch (u.op) {
.negate => bitwise.negate(operand),
.bitwise_not => bitwise.not(operand),
};
},
.binary => |b| {
const left = try evalExpr(config, b.left);
const right = try evalExpr(config, b.right);
const left = try evalExpr(allocator, config, b.left);
const right = try evalExpr(allocator, config, b.right);
return evalBinaryOp(b.op, left, right);
},
.call => {
@ -108,6 +100,28 @@ fn evalExpr(config: Config, expr: *const Expr) Error!Integer {
}
}
/// A literal as a value of the configured integer type.
///
/// A non-decimal literal is a bit pattern and converts directly. A decimal literal
/// is a number, so it is parsed exactly and then truncated toward zero: programmer
/// mode computes on integers, and `3.99` means 3.
///
/// The truncation used to run through an f64, which cost precision on the way in
/// (`1e30` arrived as 1000000000000000019884624838656) and needed range guards
/// because `@intFromFloat` on an out-of-range value is illegal behaviour: `tally -p
/// '1e40'` aborted the process before those guards existed. Parsing exactly removes
/// both problems, since a value too wide for the width is simply `Overflow`.
fn literalToInteger(allocator: Allocator, config: Config, n: ast.Expr.Number) Error!Integer {
if (n.base != .decimal) return config.value(try n.base.parseDigits(n.text));
var parsed = try Number.parse(allocator, n.text);
defer parsed.deinit();
// Literals are never negative here: unary minus is its own operator.
var whole = try Number.floor(allocator, parsed);
defer whole.deinit();
return config.value(whole.asExactInt(u128) orelse return Error.Overflow);
}
/// Evaluate a binary operation on two values of the configured type.
///
/// The bitwise operators, shifts and rotations are not here: they live in
@ -162,7 +176,7 @@ pub fn evalProgrammerString(allocator: Allocator, source: []const u8, config: Co
// Same ownership rule as evalStringInfo: the tree is ours to release, and the
// returned Integer does not borrow from it.
defer parser_mod.freeExpr(allocator, expr);
return evalProgrammer(config, expr);
return evalProgrammer(allocator, config, expr);
}
// -- Tests --
@ -430,7 +444,7 @@ test "prog: ASCII literal overflow 8-bit" {
}
test "prog: float literal truncates to integer" {
// 3.14 has no int_value, so the float branch truncates to 3
// Programmer mode computes on integers, so a fractional literal truncates.
const result = try testProg("3.14");
try testing.expectEqual(@as(u128, 3), result.unsignedValue());
}
@ -475,10 +489,11 @@ test "no leak: evalProgrammerString releases the parsed tree" {
}
}
test "programmer mode: a float literal out of range errors instead of aborting" {
test "programmer mode: a decimal literal truncates exactly, and past the width overflows" {
const config: Config = .{};
// `tally -p '1e40'` used to abort the process here: @intFromFloat on a value
// past u128 is illegal behaviour, and only 3.14 was ever tested.
// `tally -p '1e40'` used to abort the process: `@intFromFloat` on a value past
// u128 is illegal behaviour. There is no float on this path any more, so a
// literal too wide for the pattern is simply an overflow.
try std.testing.expectError(
Error.Overflow,
evalProgrammerString(std.testing.allocator, "1e40", config),
@ -487,18 +502,39 @@ test "programmer mode: a float literal out of range errors instead of aborting"
Error.Overflow,
evalProgrammerString(std.testing.allocator, "1e100", config),
);
// Still truncates the values that do fit.
const small = try evalProgrammerString(std.testing.allocator, "3.99", config);
try std.testing.expectEqual(@as(u128, 3), small.unsignedValue());
const large = try evalProgrammerString(std.testing.allocator, "1e30", config);
try std.testing.expect(large.unsignedValue() != 0);
}
test "programmer mode: an infinite or NaN literal is a domain error" {
const config: Config = .{};
// 10^400 overflows the exact tier's float projection to infinity.
// 10^400 is a 401-digit integer, which the exact tier parses without complaint
// and no width can hold. It used to report DomainError, because the f64
// projection of it was infinity.
try std.testing.expectError(
Error.DomainError,
Error.Overflow,
evalProgrammerString(std.testing.allocator, "1e400", config),
);
// Fractional literals still truncate toward zero: programmer mode computes on
// integers.
const small = try evalProgrammerString(std.testing.allocator, "3.99", config);
try std.testing.expectEqual(@as(u128, 3), small.unsignedValue());
// And the truncation is of the value written, not of an f64 of it. Through the
// old float path this arrived as 1000000000000000019884624838656.
const exact = try evalProgrammerString(std.testing.allocator, "1e30", .{ .width = .bits128 });
try std.testing.expectEqual(@as(u128, 1_000_000_000_000_000_000_000_000_000_000), exact.unsignedValue());
}
test "programmer mode: the full width is reachable from every base" {
// Open item 11: the tokenizer's `int_value` was a u64, so a literal wider than
// 64 bits was "invalid number" in a mode whose widest type is 128 bits.
const config: Config = .{ .width = .bits128, .signedness = .unsigned };
const cases = [_][]const u8{
"0xFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFF",
"340282366920938463463374607431768211455",
"0o3777777777777777777777777777777777777777777",
};
for (cases) |source| {
const result = try evalProgrammerString(std.testing.allocator, source, config);
try std.testing.expectEqual(std.math.maxInt(u128), result.unsignedValue());
}
const seventeen_nibbles = try evalProgrammerString(std.testing.allocator, "0x1FFFFFFFFFFFFFFFF", config);
try std.testing.expectEqual(@as(u128, 0x1FFFFFFFFFFFFFFFF), seventeen_nibbles.unsignedValue());
}

View file

@ -15,6 +15,56 @@ pub const Base = enum {
hex,
octal,
binary,
/// Characters that separate digits without being digits (FR-1.8).
fn isSeparator(ch: u8) bool {
return ch == '_' or ch == ',' or ch == ' ';
}
pub fn radix(self: Base) u8 {
return switch (self) {
.decimal => 10,
.hex => 16,
.octal => 8,
.binary => 2,
};
}
/// The prefix that selects this base in source: "0x", "0o", "0b", or nothing.
pub fn prefix(self: Base) []const u8 {
return switch (self) {
.decimal => "",
.hex => "0x",
.octal => "0o",
.binary => "0b",
};
}
/// The value of a literal written in this base.
///
/// `text` is a literal's source text exactly as the tokenizer produced it,
/// prefix and separators included: "0xFF_FF", "0b1010", "1 000". Separators are
/// skipped rather than stripped into a buffer, so length is not a limit.
///
/// The only failure is a value too wide for `u128`, which is the widest pattern
/// `Integer` holds and the widest FR-1.9 shows. A decimal value that needs more
/// room than that belongs to the exact tier, which parses the same text into a
/// rational instead; see `Number.parse`.
///
/// Asserts the text is digits of this base and separators. Every caller passes
/// a span the tokenizer just lexed, which cannot contain anything else.
pub fn parseDigits(self: Base, text: []const u8) error{Overflow}!u128 {
const digits = text[self.prefix().len..];
var value: u128 = 0;
for (digits) |ch| {
if (isSeparator(ch)) continue;
const digit = std.fmt.charToDigit(ch, self.radix()) catch
@panic("literal text carries a digit outside its own base");
value = try std.math.mul(u128, value, self.radix());
value = try std.math.add(u128, value, digit);
}
return value;
}
};
pub const TokenKind = enum {
@ -48,6 +98,10 @@ pub const TokenKind = enum {
// Special
eof,
invalid,
/// A numeric span that is not a number: a base prefix with no digits after it,
/// or an exponent with no digits after it. Distinct from `invalid` so the parser
/// can say "invalid number" for `0x` and "unexpected token" for `$`.
invalid_number,
// String literal (single-quoted, for ASCII byte packing in programmer mode)
string_literal,
};
@ -58,6 +112,9 @@ pub const Token = struct {
start: usize,
/// Byte length of this token in the source.
len: usize,
/// For a `number`, the base its prefix selected. The tokenizer knows this while
/// it scans, so nothing downstream has to re-read the prefix to find out.
base: Base = .decimal,
/// Extract the token's text from the source.
pub fn text(self: Token, source: []const u8) []const u8 {
@ -65,81 +122,6 @@ pub const Token = struct {
}
};
/// Parsed number value from a token.
pub const NumberValue = struct {
float: f64,
/// If the number is a pure integer (no decimal point, no exponent), this
/// holds the exact integer value.
int_value: ?u64,
base: Base,
};
/// Parse a number token's text into a value.
/// Handles 0x (hex), 0o (octal), 0b (binary), decimal integers, and floats.
/// Underscores are ignored as digit separators.
pub fn parseNumber(token_text: []const u8) !NumberValue {
// Strip underscores for parsing
var buf: [128]u8 = undefined;
var buf_len: usize = 0;
for (token_text) |c| {
if (c != '_' and c != ',' and c != ' ') {
if (buf_len >= buf.len) return error.InvalidNumber;
buf[buf_len] = c;
buf_len += 1;
}
}
const clean = buf[0..buf_len];
if (clean.len == 0) return error.InvalidNumber;
// Check base prefix
if (clean.len >= 2 and clean[0] == '0') {
switch (clean[1]) {
'x', 'X' => {
const digits = clean[2..];
if (digits.len == 0) return error.InvalidNumber;
const val = std.fmt.parseInt(u64, digits, 16) catch return error.InvalidNumber;
return .{ .float = @floatFromInt(val), .int_value = val, .base = .hex };
},
'o', 'O' => {
const digits = clean[2..];
if (digits.len == 0) return error.InvalidNumber;
const val = std.fmt.parseInt(u64, digits, 8) catch return error.InvalidNumber;
return .{ .float = @floatFromInt(val), .int_value = val, .base = .octal };
},
'b', 'B' => {
const digits = clean[2..];
if (digits.len == 0) return error.InvalidNumber;
const val = std.fmt.parseInt(u64, digits, 2) catch return error.InvalidNumber;
return .{ .float = @floatFromInt(val), .int_value = val, .base = .binary };
},
else => {},
}
}
// Check if it's a pure integer (no '.', no 'e'/'E')
var is_integer = true;
for (clean) |c| {
if (c == '.' or c == 'e' or c == 'E') {
is_integer = false;
break;
}
}
if (is_integer) {
const val = std.fmt.parseInt(u64, clean, 10) catch {
// Could be too large for u64, try as float
const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber;
return .{ .float = f, .int_value = null, .base = .decimal };
};
return .{ .float = @floatFromInt(val), .int_value = val, .base = .decimal };
}
// Float
const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber;
return .{ .float = f, .int_value = null, .base = .decimal };
}
// -- Raw Tokenizer --
/// Tokenizing is mode-independent: `0xFF`, `<<` and `'A'` are recognized in both
@ -248,24 +230,14 @@ pub const Tokenizer = struct {
if (self.source[self.pos] == '0' and self.pos + 1 < self.source.len) {
const next_ch = self.source[self.pos + 1];
switch (next_ch) {
'x', 'X' => {
self.pos += 2;
self.consumeBaseDigits(isHexDigit);
return .{ .kind = .number, .start = start, .len = self.pos - start };
},
'o', 'O' => {
self.pos += 2;
self.consumeBaseDigits(isOctalDigit);
return .{ .kind = .number, .start = start, .len = self.pos - start };
},
'x', 'X' => return self.readBaseDigits(start, .hex, isHexDigit),
'o', 'O' => return self.readBaseDigits(start, .octal, isOctalDigit),
'b', 'B' => {
// Disambiguate: 0b... is binary only if followed by 0 or 1
if (self.pos + 2 < self.source.len and
(self.source[self.pos + 2] == '0' or self.source[self.pos + 2] == '1'))
{
self.pos += 2;
self.consumeBaseDigits(isBinaryDigit);
return .{ .kind = .number, .start = start, .len = self.pos - start };
return self.readBaseDigits(start, .binary, isBinaryDigit);
}
// Otherwise fall through to decimal
},
@ -296,12 +268,34 @@ pub const Tokenizer = struct {
{
self.pos += 1;
}
const exponent_start = self.pos;
self.consumeDigits(isDecDigit);
// "1e" and "1e+" are not numbers. Rejected while the span is being
// scanned, because this is where it is known: nothing downstream can
// tell "1e" from a number without re-lexing it.
if (self.pos == exponent_start) {
return .{ .kind = .invalid_number, .start = start, .len = self.pos - start };
}
}
return .{ .kind = .number, .start = start, .len = self.pos - start };
}
/// A prefixed literal: `0x`, `0o` or `0b` and the digits of that base. A prefix
/// with no digits after it ("0x", "0o ") is not a number.
fn readBaseDigits(
self: *Tokenizer,
start: usize,
base: Base,
predicate: *const fn (u8) bool,
) Token {
self.pos += 2;
const digits_start = self.pos;
self.consumeBaseDigits(predicate);
const kind: TokenKind = if (self.pos == digits_start) .invalid_number else .number;
return .{ .kind = kind, .start = start, .len = self.pos - start, .base = base };
}
fn consumeDigits(self: *Tokenizer, predicate: *const fn (u8) bool) void {
while (self.pos < self.source.len) {
const ch = self.source[self.pos];
@ -664,73 +658,88 @@ test "tokenize base literal still groups with spaces and underscores" {
try testing.expectEqualStrings("0xFF_FF", underscored.next().text("0xFF_FF"));
}
test "parseNumber huge decimal exceeds u64 and falls back to float here" {
// The tokenizer's own integer channel is a u64, so a value this large has no
// `int_value` at this layer. That is NOT a precision limit of the engine:
// the evaluator re-parses the literal text into an exact rational (see
// evaluator.literalToNumber), so `99999999999999999999999999` still
// evaluates exactly. This test pins the tokenizer's contract only.
const result = try parseNumber("99999999999999999999999999");
try testing.expectEqual(@as(?u64, null), result.int_value);
try testing.expectEqual(Base.decimal, result.base);
try testing.expect(result.float > 1e25);
// -- Base.parseDigits tests --
//
// This replaced `parseNumber`, which built a `NumberValue{ float: f64, int_value:
// ?u64, base }` for every literal in a fixed 128-byte buffer. That capped a literal
// at 128 characters and a non-decimal one at 64 bits, and made `Rational`'s
// `TooManyDigits` (512 digits) unreachable from any frontend. The AST now carries the
// literal's text, and each consumer parses it with the precision it works in.
fn expectDigits(expected: u128, base: Base, text: []const u8) !void {
try testing.expectEqual(expected, try base.parseDigits(text));
}
// -- parseNumber tests --
test "parseNumber decimal integer" {
const result = try parseNumber("42");
try testing.expectEqual(@as(f64, 42.0), result.float);
try testing.expectEqual(@as(?u64, 42), result.int_value);
try testing.expectEqual(Base.decimal, result.base);
test "parseDigits: each base, prefix skipped" {
try expectDigits(42, .decimal, "42");
try expectDigits(255, .hex, "0xFF");
try expectDigits(10, .binary, "0b1010");
try expectDigits(511, .octal, "0o777");
try expectDigits(0, .decimal, "0");
}
test "parseNumber decimal with underscores" {
const result = try parseNumber("1_000_000");
try testing.expectEqual(@as(?u64, 1_000_000), result.int_value);
test "parseDigits: separators are skipped, not stripped into a buffer" {
try expectDigits(1_000_000, .decimal, "1_000_000");
try expectDigits(1_000_000, .decimal, "1,000,000");
try expectDigits(0xFFFF, .hex, "0xFF_FF");
try expectDigits(0xFFFF, .hex, "0xFF FF");
// 128 characters was the old buffer's limit. Length is no longer a limit at
// all, so a literal made almost entirely of separators still parses.
try expectDigits(1, .decimal, "1" ++ ("_" ** 400));
}
test "parseNumber hex" {
const result = try parseNumber("0xFF");
try testing.expectEqual(@as(?u64, 255), result.int_value);
try testing.expectEqual(Base.hex, result.base);
test "parseDigits: the whole width is reachable, which u64 could not do" {
// The literals open item 11 named: 17 hex nibbles and the top of the range.
try expectDigits(0x1FFFFFFFFFFFFFFFF, .hex, "0x1FFFFFFFFFFFFFFFF");
try expectDigits(std.math.maxInt(u128), .hex, "0xFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFF");
try expectDigits(std.math.maxInt(u128), .decimal, "340282366920938463463374607431768211455");
try expectDigits(1 << 127, .binary, "0b1" ++ ("0" ** 127));
}
test "parseNumber binary" {
const result = try parseNumber("0b1010");
try testing.expectEqual(@as(?u64, 10), result.int_value);
try testing.expectEqual(Base.binary, result.base);
test "parseDigits: past u128 is Overflow, not a float" {
try testing.expectError(error.Overflow, Base.hex.parseDigits("0x1" ++ ("0" ** 32)));
try testing.expectError(
error.Overflow,
Base.decimal.parseDigits("340282366920938463463374607431768211456"),
);
}
test "parseNumber octal" {
const result = try parseNumber("0o777");
try testing.expectEqual(@as(?u64, 511), result.int_value);
try testing.expectEqual(Base.octal, result.base);
// -- Spans that are not numbers --
test "tokenize a base prefix with no digits is not a number" {
// `parseNumber` used to catch these after the fact. The scan knows it consumed
// no digits, so it says so, and the parser reports "invalid number" rather than
// "unexpected token".
for ([_][]const u8{ "0x", "0o", "0x ", "0x+1" }) |source| {
var tok = Tokenizer.init(source);
try testing.expectEqual(TokenKind.invalid_number, tok.next().kind);
}
// `0b` without a 0 or 1 after it is a zero and an identifier, which is the
// existing disambiguation rule and not an invalid literal.
var binary = Tokenizer.init("0b");
try testing.expectEqual(TokenKind.number, binary.next().kind);
try testing.expectEqual(TokenKind.identifier, binary.next().kind);
}
test "parseNumber float" {
const result = try parseNumber("3.14");
try testing.expectApproxEqAbs(@as(f64, 3.14), result.float, 1e-10);
try testing.expectEqual(@as(?u64, null), result.int_value);
try testing.expectEqual(Base.decimal, result.base);
test "tokenize an exponent with no digits is not a number" {
for ([_][]const u8{ "1e", "1E", "1e+", "1e-", "1.5e" }) |source| {
var tok = Tokenizer.init(source);
try testing.expectEqual(TokenKind.invalid_number, tok.next().kind);
}
}
test "parseNumber float with exponent" {
const result = try parseNumber("1.5e10");
try testing.expectEqual(@as(f64, 1.5e10), result.float);
try testing.expectEqual(@as(?u64, null), result.int_value);
}
test "parseNumber hex with underscores" {
const result = try parseNumber("0xFF_FF");
try testing.expectEqual(@as(?u64, 0xFFFF), result.int_value);
try testing.expectEqual(Base.hex, result.base);
}
test "parseNumber with commas" {
const result = try parseNumber("1,000,000");
try testing.expectEqual(@as(?u64, 1_000_000), result.int_value);
try testing.expectEqual(Base.decimal, result.base);
test "tokenize records the base while it scans" {
const cases = [_]struct { source: []const u8, base: Base }{
.{ .source = "42", .base = .decimal },
.{ .source = "3.14", .base = .decimal },
.{ .source = "0xFF", .base = .hex },
.{ .source = "0o777", .base = .octal },
.{ .source = "0b1010", .base = .binary },
};
for (cases) |c| {
var tok = Tokenizer.init(c.source);
try testing.expectEqual(c.base, tok.next().base);
}
}
// -- ImplicitMulStream tests --