//! Expression tokenizer for Tally. //! //! Converts an input string into a sequence of tokens for the parser. //! Supports multiple number bases (decimal, hex 0x, octal 0o, binary 0b), //! operators, identifiers (functions/variables), single-quoted ASCII string //! literals, and space/comma/underscore digit separators. const std = @import("std"); const types = @import("types.zig"); const Mode = types.Mode; const Base = types.Base; pub const TokenKind = enum { // Literals number, // Identifiers (function names, variable names, keywords like "to", "rol", "ror") identifier, // Operators plus, minus, star, slash, percent, caret, // ^ (always exponentiation, in both modes: see FR-2.12) star_star, // ** (power in programmer) ampersand, // & pipe, // | tilde, // ~ shift_left, // << shift_right, // >> (arithmetic) shift_right_logical, // >>> // Delimiters left_paren, right_paren, comma, semicolon, equals, // = (assignment) // Special eof, invalid, // String literal (single-quoted, for ASCII byte packing in programmer mode) string_literal, }; pub const Token = struct { kind: TokenKind, /// Byte offset into the source where this token starts. start: usize, /// Byte length of this token in the source. len: usize, /// Extract the token's text from the source. pub fn text(self: Token, source: []const u8) []const u8 { return source[self.start..][0..self.len]; } }; /// Parsed number value from a token. pub const NumberValue = struct { float: f64, /// If the number is a pure integer (no decimal point, no exponent), this /// holds the exact integer value. int_value: ?u64, base: Base, }; /// Parse a number token's text into a value. /// Handles 0x (hex), 0o (octal), 0b (binary), decimal integers, and floats. /// Underscores are ignored as digit separators. pub fn parseNumber(token_text: []const u8) !NumberValue { // Strip underscores for parsing var buf: [128]u8 = undefined; var buf_len: usize = 0; for (token_text) |c| { if (c != '_' and c != ',' and c != ' ') { if (buf_len >= buf.len) return error.InvalidNumber; buf[buf_len] = c; buf_len += 1; } } const clean = buf[0..buf_len]; if (clean.len == 0) return error.InvalidNumber; // Check base prefix if (clean.len >= 2 and clean[0] == '0') { switch (clean[1]) { 'x', 'X' => { const digits = clean[2..]; if (digits.len == 0) return error.InvalidNumber; const val = std.fmt.parseInt(u64, digits, 16) catch return error.InvalidNumber; return .{ .float = @floatFromInt(val), .int_value = val, .base = .hex }; }, 'o', 'O' => { const digits = clean[2..]; if (digits.len == 0) return error.InvalidNumber; const val = std.fmt.parseInt(u64, digits, 8) catch return error.InvalidNumber; return .{ .float = @floatFromInt(val), .int_value = val, .base = .octal }; }, 'b', 'B' => { const digits = clean[2..]; if (digits.len == 0) return error.InvalidNumber; const val = std.fmt.parseInt(u64, digits, 2) catch return error.InvalidNumber; return .{ .float = @floatFromInt(val), .int_value = val, .base = .binary }; }, else => {}, } } // Check if it's a pure integer (no '.', no 'e'/'E') var is_integer = true; for (clean) |c| { if (c == '.' or c == 'e' or c == 'E') { is_integer = false; break; } } if (is_integer) { const val = std.fmt.parseInt(u64, clean, 10) catch { // Could be too large for u64, try as float const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber; return .{ .float = f, .int_value = null, .base = .decimal }; }; return .{ .float = @floatFromInt(val), .int_value = val, .base = .decimal }; } // Float const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber; return .{ .float = f, .int_value = null, .base = .decimal }; } // -- Raw Tokenizer -- pub const Tokenizer = struct { source: []const u8, pos: usize, mode: Mode, pub fn init(source: []const u8, mode: Mode) Tokenizer { return .{ .source = source, .pos = 0, .mode = mode, }; } pub fn next(self: *Tokenizer) Token { self.skipWhitespace(); if (self.pos >= self.source.len) { return .{ .kind = .eof, .start = self.pos, .len = 0 }; } const start = self.pos; const c = self.source[self.pos]; switch (c) { '+' => return self.singleChar(.plus, start), '-' => return self.singleChar(.minus, start), '/' => return self.singleChar(.slash, start), '%' => return self.singleChar(.percent, start), '&' => return self.singleChar(.ampersand, start), '|' => return self.singleChar(.pipe, start), '~' => return self.singleChar(.tilde, start), '(' => return self.singleChar(.left_paren, start), ')' => return self.singleChar(.right_paren, start), ',' => return self.singleChar(.comma, start), ';' => return self.singleChar(.semicolon, start), '=' => return self.singleChar(.equals, start), '^' => return self.singleChar(.caret, start), '*' => { self.pos += 1; if (self.pos < self.source.len and self.source[self.pos] == '*') { self.pos += 1; return .{ .kind = .star_star, .start = start, .len = 2 }; } return .{ .kind = .star, .start = start, .len = 1 }; }, '<' => { self.pos += 1; if (self.pos < self.source.len and self.source[self.pos] == '<') { self.pos += 1; return .{ .kind = .shift_left, .start = start, .len = 2 }; } return .{ .kind = .invalid, .start = start, .len = 1 }; }, '>' => { self.pos += 1; if (self.pos < self.source.len and self.source[self.pos] == '>') { self.pos += 1; if (self.pos < self.source.len and self.source[self.pos] == '>') { self.pos += 1; return .{ .kind = .shift_right_logical, .start = start, .len = 3 }; } return .{ .kind = .shift_right, .start = start, .len = 2 }; } return .{ .kind = .invalid, .start = start, .len = 1 }; }, '0'...'9' => return self.readNumber(start), 'a'...'z', 'A'...'Z', '_' => return self.readIdentifier(start), '.' => { // Could be start of a decimal number like .5 if (self.pos + 1 < self.source.len and self.source[self.pos + 1] >= '0' and self.source[self.pos + 1] <= '9') { return self.readNumber(start); } self.pos += 1; return .{ .kind = .invalid, .start = start, .len = 1 }; }, '\'' => return self.readStringLiteral(start), else => { self.pos += 1; return .{ .kind = .invalid, .start = start, .len = 1 }; }, } } fn singleChar(self: *Tokenizer, kind: TokenKind, start: usize) Token { self.pos += 1; return .{ .kind = kind, .start = start, .len = 1 }; } fn skipWhitespace(self: *Tokenizer) void { while (self.pos < self.source.len) { switch (self.source[self.pos]) { ' ', '\t', '\r', '\n' => self.pos += 1, else => break, } } } fn readNumber(self: *Tokenizer, start: usize) Token { // Check for base prefix: 0x, 0o, 0b if (self.source[self.pos] == '0' and self.pos + 1 < self.source.len) { const next_ch = self.source[self.pos + 1]; switch (next_ch) { 'x', 'X' => { self.pos += 2; self.consumeBaseDigits(isHexDigit); return .{ .kind = .number, .start = start, .len = self.pos - start }; }, 'o', 'O' => { self.pos += 2; self.consumeBaseDigits(isOctalDigit); return .{ .kind = .number, .start = start, .len = self.pos - start }; }, 'b', 'B' => { // Disambiguate: 0b... is binary only if followed by 0 or 1 if (self.pos + 2 < self.source.len and (self.source[self.pos + 2] == '0' or self.source[self.pos + 2] == '1')) { self.pos += 2; self.consumeBaseDigits(isBinaryDigit); return .{ .kind = .number, .start = start, .len = self.pos - start }; } // Otherwise fall through to decimal }, else => {}, } } // Decimal number (possibly floating point) self.consumeDigits(isDecDigit); // Fractional part if (self.pos < self.source.len and self.source[self.pos] == '.') { if (self.pos + 1 < self.source.len and self.source[self.pos + 1] >= '0' and self.source[self.pos + 1] <= '9') { self.pos += 1; // consume '.' self.consumeDigits(isDecDigit); } } // Exponent part (e or E) if (self.pos < self.source.len and (self.source[self.pos] == 'e' or self.source[self.pos] == 'E')) { self.pos += 1; if (self.pos < self.source.len and (self.source[self.pos] == '+' or self.source[self.pos] == '-')) { self.pos += 1; } self.consumeDigits(isDecDigit); } return .{ .kind = .number, .start = start, .len = self.pos - start }; } fn consumeDigits(self: *Tokenizer, predicate: *const fn (u8) bool) void { while (self.pos < self.source.len) { const ch = self.source[self.pos]; if (predicate(ch)) { self.pos += 1; } else if (ch == '_') { // Digit separator self.pos += 1; } else if (ch == ',') { if (self.commaIsGrouping(predicate)) { self.pos += 1; } else { break; } } else { break; } } } /// Decide whether the comma at `self.pos` groups digits or separates /// arguments. /// /// It groups only when followed by EXACTLY three digits, which is what a /// thousands group is. Checking merely for "a digit follows" is not enough: /// it made `log(100,10)` lex as `log(10010)` and return 4.0004 instead of 2, /// and `max(1,2)` lex as `max(12)`, which then failed as an unknown function. /// Every such call worked only if the author happened to put a space after /// the comma, which is why the tests and the help examples all passed. /// /// `max(1,234)` remains ambiguous by construction: three digits follow, so it /// reads as `max(1234)`. FR-1.8 resolves that in favour of the grouping, and a /// space is the way to ask for two arguments. fn commaIsGrouping(self: *Tokenizer, predicate: *const fn (u8) bool) bool { var digits: usize = 0; var i = self.pos + 1; while (i < self.source.len and predicate(self.source[i])) : (i += 1) { digits += 1; // More than a group: not grouping, whatever follows. if (digits > 3) return false; } if (digits != 3) return false; // A fourth digit cannot appear (the loop above would have counted it), so // the group is well formed if what follows is not another digit. What may // follow is another group, a decimal point, an exponent, an operator, or // the end of input. return true; } /// Like consumeDigits but also treats spaces as separators (only when the /// space is followed by a run of valid digits, so "0xFF + 1" stops at the /// space and "0xFF and 1" is not swallowed - "and" has a non-hex letter). /// Used for hex/oct/bin literals which display with space grouping. fn consumeBaseDigits(self: *Tokenizer, predicate: *const fn (u8) bool) void { while (self.pos < self.source.len) { const ch = self.source[self.pos]; if (predicate(ch)) { self.pos += 1; } else if (ch == '_') { self.pos += 1; } else if (ch == ' ') { if (self.spaceContinuesNumber(predicate)) { self.pos += 1; } else { break; } } else { break; } } } /// After a space inside a base literal, decide whether the following text /// is another digit group (continue the number) or a word like a keyword /// operator (stop the number). Returns true only if the maximal /// identifier-run right after the space consists entirely of valid digits. fn spaceContinuesNumber(self: *Tokenizer, predicate: *const fn (u8) bool) bool { var j = self.pos + 1; var saw_any = false; while (j < self.source.len and isIdentChar(self.source[j])) : (j += 1) { if (!predicate(self.source[j])) return false; saw_any = true; } return saw_any; } fn isIdentChar(c: u8) bool { return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or (c >= '0' and c <= '9') or c == '_'; } fn readStringLiteral(self: *Tokenizer, start: usize) Token { self.pos += 1; // consume opening quote while (self.pos < self.source.len and self.source[self.pos] != '\'') { self.pos += 1; } if (self.pos < self.source.len) { self.pos += 1; // consume closing quote } return .{ .kind = .string_literal, .start = start, .len = self.pos - start }; } fn readIdentifier(self: *Tokenizer, start: usize) Token { while (self.pos < self.source.len) { const ch = self.source[self.pos]; if ((ch >= 'a' and ch <= 'z') or (ch >= 'A' and ch <= 'Z') or (ch >= '0' and ch <= '9') or ch == '_') { self.pos += 1; } else { break; } } return .{ .kind = .identifier, .start = start, .len = self.pos - start }; } fn isHexDigit(c: u8) bool { return (c >= '0' and c <= '9') or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F'); } fn isOctalDigit(c: u8) bool { return c >= '0' and c <= '7'; } fn isBinaryDigit(c: u8) bool { return c == '0' or c == '1'; } fn isDecDigit(c: u8) bool { return c >= '0' and c <= '9'; } }; // -- Tests -- const testing = std.testing; test "tokenize simple arithmetic" { var tok = Tokenizer.init("2 + 3 * 4", .standard); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.plus, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.star, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize hex number" { var tok = Tokenizer.init("0xFF", .programmer); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("0xFF", t.text("0xFF")); } test "tokenize binary number" { var tok = Tokenizer.init("0b1010", .programmer); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("0b1010", t.text("0b1010")); } test "tokenize octal number" { var tok = Tokenizer.init("0o777", .programmer); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("0o777", t.text("0o777")); } test "tokenize shift operators" { var tok = Tokenizer.init("x << 3 >> 1 >>> 2", .programmer); try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.shift_left, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.shift_right, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.shift_right_logical, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize star_star" { var tok = Tokenizer.init("2**10", .programmer); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.star_star, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); } test "tokenize number with underscores" { var tok = Tokenizer.init("1_000_000", .standard); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("1_000_000", t.text("1_000_000")); } test "tokenize hex with underscores" { var tok = Tokenizer.init("0xFF_FF", .programmer); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("0xFF_FF", t.text("0xFF_FF")); } test "tokenize number with commas" { var tok = Tokenizer.init("1,000,000", .standard); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("1,000,000", t.text("1,000,000")); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize hex with spaces" { var tok = Tokenizer.init("0xFF FF FF FF", .programmer); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("0xFF FF FF FF", t.text("0xFF FF FF FF")); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize binary with spaces" { var tok = Tokenizer.init("0b1111 0000", .programmer); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("0b1111 0000", t.text("0b1111 0000")); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize octal with spaces" { var tok = Tokenizer.init("0o777 111", .programmer); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("0o777 111", t.text("0o777 111")); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize base literal space before operator stops" { var tok = Tokenizer.init("0b1010 + 1", .programmer); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.plus, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize comma not eaten in function args" { var tok = Tokenizer.init("max(1, 2)", .standard); try testing.expectEqual(TokenKind.identifier, tok.next().kind); // max try testing.expectEqual(TokenKind.left_paren, tok.next().kind); // ( try testing.expectEqual(TokenKind.number, tok.next().kind); // 1 try testing.expectEqual(TokenKind.comma, tok.next().kind); // , try testing.expectEqual(TokenKind.number, tok.next().kind); // 2 try testing.expectEqual(TokenKind.right_paren, tok.next().kind); // ) } test "tokenize function call" { var tok = Tokenizer.init("sin(3.14)", .standard); try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.left_paren, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.right_paren, tok.next().kind); } test "tokenize floating point with exponent" { var tok = Tokenizer.init("1.5e10", .standard); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("1.5e10", t.text("1.5e10")); } test "tokenize negative exponent" { var tok = Tokenizer.init("2.5e-3", .standard); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("2.5e-3", t.text("2.5e-3")); } test "tokenize number starting with dot" { var tok = Tokenizer.init(".5", .standard); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings(".5", t.text(".5")); } test "tokenize assignment" { var tok = Tokenizer.init("X = 42", .standard); try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.equals, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); } test "tokenize all bitwise ops" { var tok = Tokenizer.init("a & b | c ^ ~d", .programmer); try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.ampersand, tok.next().kind); try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.pipe, tok.next().kind); try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.caret, tok.next().kind); try testing.expectEqual(TokenKind.tilde, tok.next().kind); try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize empty string" { var tok = Tokenizer.init("", .standard); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize whitespace only" { var tok = Tokenizer.init(" \t\n ", .standard); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize semicolon" { var tok = Tokenizer.init(";", .standard); try testing.expectEqual(TokenKind.semicolon, tok.next().kind); } test "tokenize bare less-than is invalid" { var tok = Tokenizer.init("<", .programmer); try testing.expectEqual(TokenKind.invalid, tok.next().kind); } test "tokenize bare greater-than is invalid" { var tok = Tokenizer.init(">", .programmer); try testing.expectEqual(TokenKind.invalid, tok.next().kind); } test "tokenize lone dot is invalid" { var tok = Tokenizer.init(".x", .standard); try testing.expectEqual(TokenKind.invalid, tok.next().kind); } test "tokenize unrecognized character is invalid" { // '@' is not handled by any dispatch case, so it hits the else branch var tok = Tokenizer.init("@", .standard); const t = tok.next(); try testing.expectEqual(TokenKind.invalid, t.kind); try testing.expectEqual(@as(usize, 1), t.len); } test "tokenize base literal does not take a comma as a separator" { // Base literals group with spaces and underscores (FR-1.8); commas are the // decimal grouping character. Accepting them here only reintroduced the // argument-separator ambiguity in another place. var tok = Tokenizer.init("0xFF,FF", .programmer); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings("0xFF", t.text("0xFF,FF")); try testing.expectEqual(TokenKind.comma, tok.next().kind); // The trailing "FF" is a bare identifier now, not a continuation of the // literal, which is exactly the point: it is not silently absorbed. try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "tokenize base literal still groups with spaces and underscores" { var tok = Tokenizer.init("0xFF FF", .programmer); try testing.expectEqualStrings("0xFF FF", tok.next().text("0xFF FF")); var underscored = Tokenizer.init("0xFF_FF", .programmer); try testing.expectEqualStrings("0xFF_FF", underscored.next().text("0xFF_FF")); } test "parseNumber huge decimal exceeds u64 and falls back to float here" { // The tokenizer's own integer channel is a u64, so a value this large has no // `int_value` at this layer. That is NOT a precision limit of the engine: // the evaluator re-parses the literal text into an exact rational (see // evaluator.literalToNumber), so `99999999999999999999999999` still // evaluates exactly. This test pins the tokenizer's contract only. const result = try parseNumber("99999999999999999999999999"); try testing.expectEqual(@as(?u64, null), result.int_value); try testing.expectEqual(Base.decimal, result.base); try testing.expect(result.float > 1e25); } // -- parseNumber tests -- test "parseNumber decimal integer" { const result = try parseNumber("42"); try testing.expectEqual(@as(f64, 42.0), result.float); try testing.expectEqual(@as(?u64, 42), result.int_value); try testing.expectEqual(Base.decimal, result.base); } test "parseNumber decimal with underscores" { const result = try parseNumber("1_000_000"); try testing.expectEqual(@as(?u64, 1_000_000), result.int_value); } test "parseNumber hex" { const result = try parseNumber("0xFF"); try testing.expectEqual(@as(?u64, 255), result.int_value); try testing.expectEqual(Base.hex, result.base); } test "parseNumber binary" { const result = try parseNumber("0b1010"); try testing.expectEqual(@as(?u64, 10), result.int_value); try testing.expectEqual(Base.binary, result.base); } test "parseNumber octal" { const result = try parseNumber("0o777"); try testing.expectEqual(@as(?u64, 511), result.int_value); try testing.expectEqual(Base.octal, result.base); } test "parseNumber float" { const result = try parseNumber("3.14"); try testing.expectApproxEqAbs(@as(f64, 3.14), result.float, 1e-10); try testing.expectEqual(@as(?u64, null), result.int_value); try testing.expectEqual(Base.decimal, result.base); } test "parseNumber float with exponent" { const result = try parseNumber("1.5e10"); try testing.expectEqual(@as(f64, 1.5e10), result.float); try testing.expectEqual(@as(?u64, null), result.int_value); } test "parseNumber hex with underscores" { const result = try parseNumber("0xFF_FF"); try testing.expectEqual(@as(?u64, 0xFFFF), result.int_value); try testing.expectEqual(Base.hex, result.base); } test "parseNumber with commas" { const result = try parseNumber("1,000,000"); try testing.expectEqual(@as(?u64, 1_000_000), result.int_value); try testing.expectEqual(Base.decimal, result.base); } // -- ImplicitMulStream tests -- test "no implicit mul: spaces are just whitespace" { // Spaces between tokens don't create implicit multiplication var tok = Tokenizer.init("2 3", .standard); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.eof, tok.next().kind); } // -- The comma grouping rule -- // // A comma groups digits only when exactly three digits follow. The old rule was // "a digit follows", which silently merged function arguments: `log(100,10)` // became `log(10010)` and returned 4.0004 instead of 2. Every affected call // worked if the author put a space after the comma, which is why the old tests // and every documented example passed. test "comma groups digits only in threes" { // Grouped: consumed as one number. for ([_][]const u8{ "1,000", "1,234,567", "12,345", "123,456,789" }) |source| { var tok = Tokenizer.init(source, .standard); const t = tok.next(); try testing.expectEqual(TokenKind.number, t.kind); try testing.expectEqualStrings(source, t.text(source)); try testing.expectEqual(TokenKind.eof, tok.next().kind); } } test "comma with the wrong number of digits is a separate token" { // One, two or four digits are not a thousands group, so the comma stays a // comma. This is what makes `log(100,10)` and `max(1,2)` parse as two // arguments. for ([_][]const u8{ "100,10", "1,2", "1,00", "1,0000" }) |source| { var tok = Tokenizer.init(source, .standard); const first = tok.next(); try testing.expectEqual(TokenKind.number, first.kind); try testing.expectEqual(TokenKind.comma, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.eof, tok.next().kind); } } test "comma at the end of input is a separate token" { var tok = Tokenizer.init("1,", .standard); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.comma, tok.next().kind); try testing.expectEqual(TokenKind.eof, tok.next().kind); } test "comma followed by a non-digit is a separate token" { var tok = Tokenizer.init("max(1, 2)", .standard); try testing.expectEqual(TokenKind.identifier, tok.next().kind); try testing.expectEqual(TokenKind.left_paren, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.comma, tok.next().kind); try testing.expectEqual(TokenKind.number, tok.next().kind); try testing.expectEqual(TokenKind.right_paren, tok.next().kind); } test "a grouped literal is still exact and keeps its full text" { // The exact tier re-parses the literal text, so the separators have to remain // in the token for it to see them. var tok = Tokenizer.init("9,007,199,254,740,993", .standard); const t = tok.next(); try testing.expectEqualStrings("9,007,199,254,740,993", t.text("9,007,199,254,740,993")); }