tally/engine/src/tokenizer.zig

793 lines
30 KiB
Zig

//! Expression tokenizer for Tally.
//!
//! Converts an input string into a sequence of tokens for the parser.
//! Supports multiple number bases (decimal, hex 0x, octal 0o, binary 0b),
//! operators, identifiers (functions/variables), single-quoted ASCII string
//! literals, and space/comma/underscore digit separators.
const std = @import("std");
const types = @import("types.zig");
const Mode = types.Mode;
const Base = types.Base;
pub const TokenKind = enum {
// Literals
number,
// Identifiers (function names, variable names, keywords like "to", "rol", "ror")
identifier,
// Operators
plus,
minus,
star,
slash,
percent,
caret, // ^ (always exponentiation, in both modes: see FR-2.12)
star_star, // ** (power in programmer)
ampersand, // &
pipe, // |
tilde, // ~
shift_left, // <<
shift_right, // >> (arithmetic)
shift_right_logical, // >>>
// Delimiters
left_paren,
right_paren,
comma,
semicolon,
equals, // = (assignment)
// Special
eof,
invalid,
// String literal (single-quoted, for ASCII byte packing in programmer mode)
string_literal,
};
pub const Token = struct {
kind: TokenKind,
/// Byte offset into the source where this token starts.
start: usize,
/// Byte length of this token in the source.
len: usize,
/// Extract the token's text from the source.
pub fn text(self: Token, source: []const u8) []const u8 {
return source[self.start..][0..self.len];
}
};
/// Parsed number value from a token.
pub const NumberValue = struct {
float: f64,
/// If the number is a pure integer (no decimal point, no exponent), this
/// holds the exact integer value.
int_value: ?u64,
base: Base,
};
/// Parse a number token's text into a value.
/// Handles 0x (hex), 0o (octal), 0b (binary), decimal integers, and floats.
/// Underscores are ignored as digit separators.
pub fn parseNumber(token_text: []const u8) !NumberValue {
// Strip underscores for parsing
var buf: [128]u8 = undefined;
var buf_len: usize = 0;
for (token_text) |c| {
if (c != '_' and c != ',' and c != ' ') {
if (buf_len >= buf.len) return error.InvalidNumber;
buf[buf_len] = c;
buf_len += 1;
}
}
const clean = buf[0..buf_len];
if (clean.len == 0) return error.InvalidNumber;
// Check base prefix
if (clean.len >= 2 and clean[0] == '0') {
switch (clean[1]) {
'x', 'X' => {
const digits = clean[2..];
if (digits.len == 0) return error.InvalidNumber;
const val = std.fmt.parseInt(u64, digits, 16) catch return error.InvalidNumber;
return .{ .float = @floatFromInt(val), .int_value = val, .base = .hex };
},
'o', 'O' => {
const digits = clean[2..];
if (digits.len == 0) return error.InvalidNumber;
const val = std.fmt.parseInt(u64, digits, 8) catch return error.InvalidNumber;
return .{ .float = @floatFromInt(val), .int_value = val, .base = .octal };
},
'b', 'B' => {
const digits = clean[2..];
if (digits.len == 0) return error.InvalidNumber;
const val = std.fmt.parseInt(u64, digits, 2) catch return error.InvalidNumber;
return .{ .float = @floatFromInt(val), .int_value = val, .base = .binary };
},
else => {},
}
}
// Check if it's a pure integer (no '.', no 'e'/'E')
var is_integer = true;
for (clean) |c| {
if (c == '.' or c == 'e' or c == 'E') {
is_integer = false;
break;
}
}
if (is_integer) {
const val = std.fmt.parseInt(u64, clean, 10) catch {
// Could be too large for u64, try as float
const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber;
return .{ .float = f, .int_value = null, .base = .decimal };
};
return .{ .float = @floatFromInt(val), .int_value = val, .base = .decimal };
}
// Float
const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber;
return .{ .float = f, .int_value = null, .base = .decimal };
}
// -- Raw Tokenizer --
pub const Tokenizer = struct {
source: []const u8,
pos: usize,
mode: Mode,
pub fn init(source: []const u8, mode: Mode) Tokenizer {
return .{
.source = source,
.pos = 0,
.mode = mode,
};
}
pub fn next(self: *Tokenizer) Token {
self.skipWhitespace();
if (self.pos >= self.source.len) {
return .{ .kind = .eof, .start = self.pos, .len = 0 };
}
const start = self.pos;
const c = self.source[self.pos];
switch (c) {
'+' => return self.singleChar(.plus, start),
'-' => return self.singleChar(.minus, start),
'/' => return self.singleChar(.slash, start),
'%' => return self.singleChar(.percent, start),
'&' => return self.singleChar(.ampersand, start),
'|' => return self.singleChar(.pipe, start),
'~' => return self.singleChar(.tilde, start),
'(' => return self.singleChar(.left_paren, start),
')' => return self.singleChar(.right_paren, start),
',' => return self.singleChar(.comma, start),
';' => return self.singleChar(.semicolon, start),
'=' => return self.singleChar(.equals, start),
'^' => return self.singleChar(.caret, start),
'*' => {
self.pos += 1;
if (self.pos < self.source.len and self.source[self.pos] == '*') {
self.pos += 1;
return .{ .kind = .star_star, .start = start, .len = 2 };
}
return .{ .kind = .star, .start = start, .len = 1 };
},
'<' => {
self.pos += 1;
if (self.pos < self.source.len and self.source[self.pos] == '<') {
self.pos += 1;
return .{ .kind = .shift_left, .start = start, .len = 2 };
}
return .{ .kind = .invalid, .start = start, .len = 1 };
},
'>' => {
self.pos += 1;
if (self.pos < self.source.len and self.source[self.pos] == '>') {
self.pos += 1;
if (self.pos < self.source.len and self.source[self.pos] == '>') {
self.pos += 1;
return .{ .kind = .shift_right_logical, .start = start, .len = 3 };
}
return .{ .kind = .shift_right, .start = start, .len = 2 };
}
return .{ .kind = .invalid, .start = start, .len = 1 };
},
'0'...'9' => return self.readNumber(start),
'a'...'z', 'A'...'Z', '_' => return self.readIdentifier(start),
'.' => {
// Could be start of a decimal number like .5
if (self.pos + 1 < self.source.len and
self.source[self.pos + 1] >= '0' and self.source[self.pos + 1] <= '9')
{
return self.readNumber(start);
}
self.pos += 1;
return .{ .kind = .invalid, .start = start, .len = 1 };
},
'\'' => return self.readStringLiteral(start),
else => {
self.pos += 1;
return .{ .kind = .invalid, .start = start, .len = 1 };
},
}
}
fn singleChar(self: *Tokenizer, kind: TokenKind, start: usize) Token {
self.pos += 1;
return .{ .kind = kind, .start = start, .len = 1 };
}
fn skipWhitespace(self: *Tokenizer) void {
while (self.pos < self.source.len) {
switch (self.source[self.pos]) {
' ', '\t', '\r', '\n' => self.pos += 1,
else => break,
}
}
}
fn readNumber(self: *Tokenizer, start: usize) Token {
// Check for base prefix: 0x, 0o, 0b
if (self.source[self.pos] == '0' and self.pos + 1 < self.source.len) {
const next_ch = self.source[self.pos + 1];
switch (next_ch) {
'x', 'X' => {
self.pos += 2;
self.consumeBaseDigits(isHexDigit);
return .{ .kind = .number, .start = start, .len = self.pos - start };
},
'o', 'O' => {
self.pos += 2;
self.consumeBaseDigits(isOctalDigit);
return .{ .kind = .number, .start = start, .len = self.pos - start };
},
'b', 'B' => {
// Disambiguate: 0b... is binary only if followed by 0 or 1
if (self.pos + 2 < self.source.len and
(self.source[self.pos + 2] == '0' or self.source[self.pos + 2] == '1'))
{
self.pos += 2;
self.consumeBaseDigits(isBinaryDigit);
return .{ .kind = .number, .start = start, .len = self.pos - start };
}
// Otherwise fall through to decimal
},
else => {},
}
}
// Decimal number (possibly floating point)
self.consumeDigits(isDecDigit);
// Fractional part
if (self.pos < self.source.len and self.source[self.pos] == '.') {
if (self.pos + 1 < self.source.len and
self.source[self.pos + 1] >= '0' and self.source[self.pos + 1] <= '9')
{
self.pos += 1; // consume '.'
self.consumeDigits(isDecDigit);
}
}
// Exponent part (e or E)
if (self.pos < self.source.len and
(self.source[self.pos] == 'e' or self.source[self.pos] == 'E'))
{
self.pos += 1;
if (self.pos < self.source.len and
(self.source[self.pos] == '+' or self.source[self.pos] == '-'))
{
self.pos += 1;
}
self.consumeDigits(isDecDigit);
}
return .{ .kind = .number, .start = start, .len = self.pos - start };
}
fn consumeDigits(self: *Tokenizer, predicate: *const fn (u8) bool) void {
while (self.pos < self.source.len) {
const ch = self.source[self.pos];
if (predicate(ch)) {
self.pos += 1;
} else if (ch == '_') {
// Digit separator
self.pos += 1;
} else if (ch == ',') {
if (self.commaIsGrouping(predicate)) {
self.pos += 1;
} else {
break;
}
} else {
break;
}
}
}
/// Decide whether the comma at `self.pos` groups digits or separates
/// arguments.
///
/// It groups only when followed by EXACTLY three digits, which is what a
/// thousands group is. Checking merely for "a digit follows" is not enough:
/// it made `log(100,10)` lex as `log(10010)` and return 4.0004 instead of 2,
/// and `max(1,2)` lex as `max(12)`, which then failed as an unknown function.
/// Every such call worked only if the author happened to put a space after
/// the comma, which is why the tests and the help examples all passed.
///
/// `max(1,234)` remains ambiguous by construction: three digits follow, so it
/// reads as `max(1234)`. FR-1.8 resolves that in favour of the grouping, and a
/// space is the way to ask for two arguments.
fn commaIsGrouping(self: *Tokenizer, predicate: *const fn (u8) bool) bool {
var digits: usize = 0;
var i = self.pos + 1;
while (i < self.source.len and predicate(self.source[i])) : (i += 1) {
digits += 1;
// More than a group: not grouping, whatever follows.
if (digits > 3) return false;
}
if (digits != 3) return false;
// A fourth digit cannot appear (the loop above would have counted it), so
// the group is well formed if what follows is not another digit. What may
// follow is another group, a decimal point, an exponent, an operator, or
// the end of input.
return true;
}
/// Like consumeDigits but also treats spaces as separators (only when the
/// space is followed by a run of valid digits, so "0xFF + 1" stops at the
/// space and "0xFF and 1" is not swallowed - "and" has a non-hex letter).
/// Used for hex/oct/bin literals which display with space grouping.
fn consumeBaseDigits(self: *Tokenizer, predicate: *const fn (u8) bool) void {
while (self.pos < self.source.len) {
const ch = self.source[self.pos];
if (predicate(ch)) {
self.pos += 1;
} else if (ch == '_') {
self.pos += 1;
} else if (ch == ' ') {
if (self.spaceContinuesNumber(predicate)) {
self.pos += 1;
} else {
break;
}
} else {
break;
}
}
}
/// After a space inside a base literal, decide whether the following text
/// is another digit group (continue the number) or a word like a keyword
/// operator (stop the number). Returns true only if the maximal
/// identifier-run right after the space consists entirely of valid digits.
fn spaceContinuesNumber(self: *Tokenizer, predicate: *const fn (u8) bool) bool {
var j = self.pos + 1;
var saw_any = false;
while (j < self.source.len and isIdentChar(self.source[j])) : (j += 1) {
if (!predicate(self.source[j])) return false;
saw_any = true;
}
return saw_any;
}
fn isIdentChar(c: u8) bool {
return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or
(c >= '0' and c <= '9') or c == '_';
}
fn readStringLiteral(self: *Tokenizer, start: usize) Token {
self.pos += 1; // consume opening quote
while (self.pos < self.source.len and self.source[self.pos] != '\'') {
self.pos += 1;
}
if (self.pos < self.source.len) {
self.pos += 1; // consume closing quote
}
return .{ .kind = .string_literal, .start = start, .len = self.pos - start };
}
fn readIdentifier(self: *Tokenizer, start: usize) Token {
while (self.pos < self.source.len) {
const ch = self.source[self.pos];
if ((ch >= 'a' and ch <= 'z') or
(ch >= 'A' and ch <= 'Z') or
(ch >= '0' and ch <= '9') or
ch == '_')
{
self.pos += 1;
} else {
break;
}
}
return .{ .kind = .identifier, .start = start, .len = self.pos - start };
}
fn isHexDigit(c: u8) bool {
return (c >= '0' and c <= '9') or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F');
}
fn isOctalDigit(c: u8) bool {
return c >= '0' and c <= '7';
}
fn isBinaryDigit(c: u8) bool {
return c == '0' or c == '1';
}
fn isDecDigit(c: u8) bool {
return c >= '0' and c <= '9';
}
};
// -- Tests --
const testing = std.testing;
test "tokenize simple arithmetic" {
var tok = Tokenizer.init("2 + 3 * 4", .standard);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.plus, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.star, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize hex number" {
var tok = Tokenizer.init("0xFF", .programmer);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("0xFF", t.text("0xFF"));
}
test "tokenize binary number" {
var tok = Tokenizer.init("0b1010", .programmer);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("0b1010", t.text("0b1010"));
}
test "tokenize octal number" {
var tok = Tokenizer.init("0o777", .programmer);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("0o777", t.text("0o777"));
}
test "tokenize shift operators" {
var tok = Tokenizer.init("x << 3 >> 1 >>> 2", .programmer);
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.shift_left, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.shift_right, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.shift_right_logical, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize star_star" {
var tok = Tokenizer.init("2**10", .programmer);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.star_star, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
}
test "tokenize number with underscores" {
var tok = Tokenizer.init("1_000_000", .standard);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("1_000_000", t.text("1_000_000"));
}
test "tokenize hex with underscores" {
var tok = Tokenizer.init("0xFF_FF", .programmer);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("0xFF_FF", t.text("0xFF_FF"));
}
test "tokenize number with commas" {
var tok = Tokenizer.init("1,000,000", .standard);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("1,000,000", t.text("1,000,000"));
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize hex with spaces" {
var tok = Tokenizer.init("0xFF FF FF FF", .programmer);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("0xFF FF FF FF", t.text("0xFF FF FF FF"));
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize binary with spaces" {
var tok = Tokenizer.init("0b1111 0000", .programmer);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("0b1111 0000", t.text("0b1111 0000"));
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize octal with spaces" {
var tok = Tokenizer.init("0o777 111", .programmer);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("0o777 111", t.text("0o777 111"));
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize base literal space before operator stops" {
var tok = Tokenizer.init("0b1010 + 1", .programmer);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.plus, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize comma not eaten in function args" {
var tok = Tokenizer.init("max(1, 2)", .standard);
try testing.expectEqual(TokenKind.identifier, tok.next().kind); // max
try testing.expectEqual(TokenKind.left_paren, tok.next().kind); // (
try testing.expectEqual(TokenKind.number, tok.next().kind); // 1
try testing.expectEqual(TokenKind.comma, tok.next().kind); // ,
try testing.expectEqual(TokenKind.number, tok.next().kind); // 2
try testing.expectEqual(TokenKind.right_paren, tok.next().kind); // )
}
test "tokenize function call" {
var tok = Tokenizer.init("sin(3.14)", .standard);
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.left_paren, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.right_paren, tok.next().kind);
}
test "tokenize floating point with exponent" {
var tok = Tokenizer.init("1.5e10", .standard);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("1.5e10", t.text("1.5e10"));
}
test "tokenize negative exponent" {
var tok = Tokenizer.init("2.5e-3", .standard);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("2.5e-3", t.text("2.5e-3"));
}
test "tokenize number starting with dot" {
var tok = Tokenizer.init(".5", .standard);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings(".5", t.text(".5"));
}
test "tokenize assignment" {
var tok = Tokenizer.init("X = 42", .standard);
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.equals, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
}
test "tokenize all bitwise ops" {
var tok = Tokenizer.init("a & b | c ^ ~d", .programmer);
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.ampersand, tok.next().kind);
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.pipe, tok.next().kind);
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.caret, tok.next().kind);
try testing.expectEqual(TokenKind.tilde, tok.next().kind);
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize empty string" {
var tok = Tokenizer.init("", .standard);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize whitespace only" {
var tok = Tokenizer.init(" \t\n ", .standard);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize semicolon" {
var tok = Tokenizer.init(";", .standard);
try testing.expectEqual(TokenKind.semicolon, tok.next().kind);
}
test "tokenize bare less-than is invalid" {
var tok = Tokenizer.init("<", .programmer);
try testing.expectEqual(TokenKind.invalid, tok.next().kind);
}
test "tokenize bare greater-than is invalid" {
var tok = Tokenizer.init(">", .programmer);
try testing.expectEqual(TokenKind.invalid, tok.next().kind);
}
test "tokenize lone dot is invalid" {
var tok = Tokenizer.init(".x", .standard);
try testing.expectEqual(TokenKind.invalid, tok.next().kind);
}
test "tokenize unrecognized character is invalid" {
// '@' is not handled by any dispatch case, so it hits the else branch
var tok = Tokenizer.init("@", .standard);
const t = tok.next();
try testing.expectEqual(TokenKind.invalid, t.kind);
try testing.expectEqual(@as(usize, 1), t.len);
}
test "tokenize base literal does not take a comma as a separator" {
// Base literals group with spaces and underscores (FR-1.8); commas are the
// decimal grouping character. Accepting them here only reintroduced the
// argument-separator ambiguity in another place.
var tok = Tokenizer.init("0xFF,FF", .programmer);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings("0xFF", t.text("0xFF,FF"));
try testing.expectEqual(TokenKind.comma, tok.next().kind);
// The trailing "FF" is a bare identifier now, not a continuation of the
// literal, which is exactly the point: it is not silently absorbed.
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "tokenize base literal still groups with spaces and underscores" {
var tok = Tokenizer.init("0xFF FF", .programmer);
try testing.expectEqualStrings("0xFF FF", tok.next().text("0xFF FF"));
var underscored = Tokenizer.init("0xFF_FF", .programmer);
try testing.expectEqualStrings("0xFF_FF", underscored.next().text("0xFF_FF"));
}
test "parseNumber huge decimal exceeds u64 and falls back to float here" {
// The tokenizer's own integer channel is a u64, so a value this large has no
// `int_value` at this layer. That is NOT a precision limit of the engine:
// the evaluator re-parses the literal text into an exact rational (see
// evaluator.literalToNumber), so `99999999999999999999999999` still
// evaluates exactly. This test pins the tokenizer's contract only.
const result = try parseNumber("99999999999999999999999999");
try testing.expectEqual(@as(?u64, null), result.int_value);
try testing.expectEqual(Base.decimal, result.base);
try testing.expect(result.float > 1e25);
}
// -- parseNumber tests --
test "parseNumber decimal integer" {
const result = try parseNumber("42");
try testing.expectEqual(@as(f64, 42.0), result.float);
try testing.expectEqual(@as(?u64, 42), result.int_value);
try testing.expectEqual(Base.decimal, result.base);
}
test "parseNumber decimal with underscores" {
const result = try parseNumber("1_000_000");
try testing.expectEqual(@as(?u64, 1_000_000), result.int_value);
}
test "parseNumber hex" {
const result = try parseNumber("0xFF");
try testing.expectEqual(@as(?u64, 255), result.int_value);
try testing.expectEqual(Base.hex, result.base);
}
test "parseNumber binary" {
const result = try parseNumber("0b1010");
try testing.expectEqual(@as(?u64, 10), result.int_value);
try testing.expectEqual(Base.binary, result.base);
}
test "parseNumber octal" {
const result = try parseNumber("0o777");
try testing.expectEqual(@as(?u64, 511), result.int_value);
try testing.expectEqual(Base.octal, result.base);
}
test "parseNumber float" {
const result = try parseNumber("3.14");
try testing.expectApproxEqAbs(@as(f64, 3.14), result.float, 1e-10);
try testing.expectEqual(@as(?u64, null), result.int_value);
try testing.expectEqual(Base.decimal, result.base);
}
test "parseNumber float with exponent" {
const result = try parseNumber("1.5e10");
try testing.expectEqual(@as(f64, 1.5e10), result.float);
try testing.expectEqual(@as(?u64, null), result.int_value);
}
test "parseNumber hex with underscores" {
const result = try parseNumber("0xFF_FF");
try testing.expectEqual(@as(?u64, 0xFFFF), result.int_value);
try testing.expectEqual(Base.hex, result.base);
}
test "parseNumber with commas" {
const result = try parseNumber("1,000,000");
try testing.expectEqual(@as(?u64, 1_000_000), result.int_value);
try testing.expectEqual(Base.decimal, result.base);
}
// -- ImplicitMulStream tests --
test "no implicit mul: spaces are just whitespace" {
// Spaces between tokens don't create implicit multiplication
var tok = Tokenizer.init("2 3", .standard);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
// -- The comma grouping rule --
//
// A comma groups digits only when exactly three digits follow. The old rule was
// "a digit follows", which silently merged function arguments: `log(100,10)`
// became `log(10010)` and returned 4.0004 instead of 2. Every affected call
// worked if the author put a space after the comma, which is why the old tests
// and every documented example passed.
test "comma groups digits only in threes" {
// Grouped: consumed as one number.
for ([_][]const u8{ "1,000", "1,234,567", "12,345", "123,456,789" }) |source| {
var tok = Tokenizer.init(source, .standard);
const t = tok.next();
try testing.expectEqual(TokenKind.number, t.kind);
try testing.expectEqualStrings(source, t.text(source));
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
}
test "comma with the wrong number of digits is a separate token" {
// One, two or four digits are not a thousands group, so the comma stays a
// comma. This is what makes `log(100,10)` and `max(1,2)` parse as two
// arguments.
for ([_][]const u8{ "100,10", "1,2", "1,00", "1,0000" }) |source| {
var tok = Tokenizer.init(source, .standard);
const first = tok.next();
try testing.expectEqual(TokenKind.number, first.kind);
try testing.expectEqual(TokenKind.comma, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
}
test "comma at the end of input is a separate token" {
var tok = Tokenizer.init("1,", .standard);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.comma, tok.next().kind);
try testing.expectEqual(TokenKind.eof, tok.next().kind);
}
test "comma followed by a non-digit is a separate token" {
var tok = Tokenizer.init("max(1, 2)", .standard);
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
try testing.expectEqual(TokenKind.left_paren, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.comma, tok.next().kind);
try testing.expectEqual(TokenKind.number, tok.next().kind);
try testing.expectEqual(TokenKind.right_paren, tok.next().kind);
}
test "a grouped literal is still exact and keeps its full text" {
// The exact tier re-parses the literal text, so the separators have to remain
// in the token for it to see them.
var tok = Tokenizer.init("9,007,199,254,740,993", .standard);
const t = tok.next();
try testing.expectEqualStrings("9,007,199,254,740,993", t.text("9,007,199,254,740,993"));
}