793 lines
30 KiB
Zig
793 lines
30 KiB
Zig
//! Expression tokenizer for Tally.
|
|
//!
|
|
//! Converts an input string into a sequence of tokens for the parser.
|
|
//! Supports multiple number bases (decimal, hex 0x, octal 0o, binary 0b),
|
|
//! operators, identifiers (functions/variables), single-quoted ASCII string
|
|
//! literals, and space/comma/underscore digit separators.
|
|
|
|
const std = @import("std");
|
|
const types = @import("types.zig");
|
|
const Mode = types.Mode;
|
|
const Base = types.Base;
|
|
|
|
pub const TokenKind = enum {
|
|
// Literals
|
|
number,
|
|
// Identifiers (function names, variable names, keywords like "to", "rol", "ror")
|
|
identifier,
|
|
|
|
// Operators
|
|
plus,
|
|
minus,
|
|
star,
|
|
slash,
|
|
percent,
|
|
caret, // ^ (always exponentiation, in both modes: see FR-2.12)
|
|
star_star, // ** (power in programmer)
|
|
ampersand, // &
|
|
pipe, // |
|
|
tilde, // ~
|
|
shift_left, // <<
|
|
shift_right, // >> (arithmetic)
|
|
shift_right_logical, // >>>
|
|
|
|
// Delimiters
|
|
left_paren,
|
|
right_paren,
|
|
comma,
|
|
semicolon,
|
|
equals, // = (assignment)
|
|
|
|
// Special
|
|
eof,
|
|
invalid,
|
|
// String literal (single-quoted, for ASCII byte packing in programmer mode)
|
|
string_literal,
|
|
};
|
|
|
|
pub const Token = struct {
|
|
kind: TokenKind,
|
|
/// Byte offset into the source where this token starts.
|
|
start: usize,
|
|
/// Byte length of this token in the source.
|
|
len: usize,
|
|
|
|
/// Extract the token's text from the source.
|
|
pub fn text(self: Token, source: []const u8) []const u8 {
|
|
return source[self.start..][0..self.len];
|
|
}
|
|
};
|
|
|
|
/// Parsed number value from a token.
|
|
pub const NumberValue = struct {
|
|
float: f64,
|
|
/// If the number is a pure integer (no decimal point, no exponent), this
|
|
/// holds the exact integer value.
|
|
int_value: ?u64,
|
|
base: Base,
|
|
};
|
|
|
|
/// Parse a number token's text into a value.
|
|
/// Handles 0x (hex), 0o (octal), 0b (binary), decimal integers, and floats.
|
|
/// Underscores are ignored as digit separators.
|
|
pub fn parseNumber(token_text: []const u8) !NumberValue {
|
|
// Strip underscores for parsing
|
|
var buf: [128]u8 = undefined;
|
|
var buf_len: usize = 0;
|
|
for (token_text) |c| {
|
|
if (c != '_' and c != ',' and c != ' ') {
|
|
if (buf_len >= buf.len) return error.InvalidNumber;
|
|
buf[buf_len] = c;
|
|
buf_len += 1;
|
|
}
|
|
}
|
|
const clean = buf[0..buf_len];
|
|
|
|
if (clean.len == 0) return error.InvalidNumber;
|
|
|
|
// Check base prefix
|
|
if (clean.len >= 2 and clean[0] == '0') {
|
|
switch (clean[1]) {
|
|
'x', 'X' => {
|
|
const digits = clean[2..];
|
|
if (digits.len == 0) return error.InvalidNumber;
|
|
const val = std.fmt.parseInt(u64, digits, 16) catch return error.InvalidNumber;
|
|
return .{ .float = @floatFromInt(val), .int_value = val, .base = .hex };
|
|
},
|
|
'o', 'O' => {
|
|
const digits = clean[2..];
|
|
if (digits.len == 0) return error.InvalidNumber;
|
|
const val = std.fmt.parseInt(u64, digits, 8) catch return error.InvalidNumber;
|
|
return .{ .float = @floatFromInt(val), .int_value = val, .base = .octal };
|
|
},
|
|
'b', 'B' => {
|
|
const digits = clean[2..];
|
|
if (digits.len == 0) return error.InvalidNumber;
|
|
const val = std.fmt.parseInt(u64, digits, 2) catch return error.InvalidNumber;
|
|
return .{ .float = @floatFromInt(val), .int_value = val, .base = .binary };
|
|
},
|
|
else => {},
|
|
}
|
|
}
|
|
|
|
// Check if it's a pure integer (no '.', no 'e'/'E')
|
|
var is_integer = true;
|
|
for (clean) |c| {
|
|
if (c == '.' or c == 'e' or c == 'E') {
|
|
is_integer = false;
|
|
break;
|
|
}
|
|
}
|
|
|
|
if (is_integer) {
|
|
const val = std.fmt.parseInt(u64, clean, 10) catch {
|
|
// Could be too large for u64, try as float
|
|
const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber;
|
|
return .{ .float = f, .int_value = null, .base = .decimal };
|
|
};
|
|
return .{ .float = @floatFromInt(val), .int_value = val, .base = .decimal };
|
|
}
|
|
|
|
// Float
|
|
const f = std.fmt.parseFloat(f64, clean) catch return error.InvalidNumber;
|
|
return .{ .float = f, .int_value = null, .base = .decimal };
|
|
}
|
|
|
|
// -- Raw Tokenizer --
|
|
|
|
pub const Tokenizer = struct {
|
|
source: []const u8,
|
|
pos: usize,
|
|
mode: Mode,
|
|
|
|
pub fn init(source: []const u8, mode: Mode) Tokenizer {
|
|
return .{
|
|
.source = source,
|
|
.pos = 0,
|
|
.mode = mode,
|
|
};
|
|
}
|
|
|
|
pub fn next(self: *Tokenizer) Token {
|
|
self.skipWhitespace();
|
|
|
|
if (self.pos >= self.source.len) {
|
|
return .{ .kind = .eof, .start = self.pos, .len = 0 };
|
|
}
|
|
|
|
const start = self.pos;
|
|
const c = self.source[self.pos];
|
|
|
|
switch (c) {
|
|
'+' => return self.singleChar(.plus, start),
|
|
'-' => return self.singleChar(.minus, start),
|
|
'/' => return self.singleChar(.slash, start),
|
|
'%' => return self.singleChar(.percent, start),
|
|
'&' => return self.singleChar(.ampersand, start),
|
|
'|' => return self.singleChar(.pipe, start),
|
|
'~' => return self.singleChar(.tilde, start),
|
|
'(' => return self.singleChar(.left_paren, start),
|
|
')' => return self.singleChar(.right_paren, start),
|
|
',' => return self.singleChar(.comma, start),
|
|
';' => return self.singleChar(.semicolon, start),
|
|
'=' => return self.singleChar(.equals, start),
|
|
'^' => return self.singleChar(.caret, start),
|
|
'*' => {
|
|
self.pos += 1;
|
|
if (self.pos < self.source.len and self.source[self.pos] == '*') {
|
|
self.pos += 1;
|
|
return .{ .kind = .star_star, .start = start, .len = 2 };
|
|
}
|
|
return .{ .kind = .star, .start = start, .len = 1 };
|
|
},
|
|
'<' => {
|
|
self.pos += 1;
|
|
if (self.pos < self.source.len and self.source[self.pos] == '<') {
|
|
self.pos += 1;
|
|
return .{ .kind = .shift_left, .start = start, .len = 2 };
|
|
}
|
|
return .{ .kind = .invalid, .start = start, .len = 1 };
|
|
},
|
|
'>' => {
|
|
self.pos += 1;
|
|
if (self.pos < self.source.len and self.source[self.pos] == '>') {
|
|
self.pos += 1;
|
|
if (self.pos < self.source.len and self.source[self.pos] == '>') {
|
|
self.pos += 1;
|
|
return .{ .kind = .shift_right_logical, .start = start, .len = 3 };
|
|
}
|
|
return .{ .kind = .shift_right, .start = start, .len = 2 };
|
|
}
|
|
return .{ .kind = .invalid, .start = start, .len = 1 };
|
|
},
|
|
'0'...'9' => return self.readNumber(start),
|
|
'a'...'z', 'A'...'Z', '_' => return self.readIdentifier(start),
|
|
'.' => {
|
|
// Could be start of a decimal number like .5
|
|
if (self.pos + 1 < self.source.len and
|
|
self.source[self.pos + 1] >= '0' and self.source[self.pos + 1] <= '9')
|
|
{
|
|
return self.readNumber(start);
|
|
}
|
|
self.pos += 1;
|
|
return .{ .kind = .invalid, .start = start, .len = 1 };
|
|
},
|
|
'\'' => return self.readStringLiteral(start),
|
|
else => {
|
|
self.pos += 1;
|
|
return .{ .kind = .invalid, .start = start, .len = 1 };
|
|
},
|
|
}
|
|
}
|
|
|
|
fn singleChar(self: *Tokenizer, kind: TokenKind, start: usize) Token {
|
|
self.pos += 1;
|
|
return .{ .kind = kind, .start = start, .len = 1 };
|
|
}
|
|
|
|
fn skipWhitespace(self: *Tokenizer) void {
|
|
while (self.pos < self.source.len) {
|
|
switch (self.source[self.pos]) {
|
|
' ', '\t', '\r', '\n' => self.pos += 1,
|
|
else => break,
|
|
}
|
|
}
|
|
}
|
|
|
|
fn readNumber(self: *Tokenizer, start: usize) Token {
|
|
// Check for base prefix: 0x, 0o, 0b
|
|
if (self.source[self.pos] == '0' and self.pos + 1 < self.source.len) {
|
|
const next_ch = self.source[self.pos + 1];
|
|
switch (next_ch) {
|
|
'x', 'X' => {
|
|
self.pos += 2;
|
|
self.consumeBaseDigits(isHexDigit);
|
|
return .{ .kind = .number, .start = start, .len = self.pos - start };
|
|
},
|
|
'o', 'O' => {
|
|
self.pos += 2;
|
|
self.consumeBaseDigits(isOctalDigit);
|
|
return .{ .kind = .number, .start = start, .len = self.pos - start };
|
|
},
|
|
'b', 'B' => {
|
|
// Disambiguate: 0b... is binary only if followed by 0 or 1
|
|
if (self.pos + 2 < self.source.len and
|
|
(self.source[self.pos + 2] == '0' or self.source[self.pos + 2] == '1'))
|
|
{
|
|
self.pos += 2;
|
|
self.consumeBaseDigits(isBinaryDigit);
|
|
return .{ .kind = .number, .start = start, .len = self.pos - start };
|
|
}
|
|
// Otherwise fall through to decimal
|
|
},
|
|
else => {},
|
|
}
|
|
}
|
|
|
|
// Decimal number (possibly floating point)
|
|
self.consumeDigits(isDecDigit);
|
|
|
|
// Fractional part
|
|
if (self.pos < self.source.len and self.source[self.pos] == '.') {
|
|
if (self.pos + 1 < self.source.len and
|
|
self.source[self.pos + 1] >= '0' and self.source[self.pos + 1] <= '9')
|
|
{
|
|
self.pos += 1; // consume '.'
|
|
self.consumeDigits(isDecDigit);
|
|
}
|
|
}
|
|
|
|
// Exponent part (e or E)
|
|
if (self.pos < self.source.len and
|
|
(self.source[self.pos] == 'e' or self.source[self.pos] == 'E'))
|
|
{
|
|
self.pos += 1;
|
|
if (self.pos < self.source.len and
|
|
(self.source[self.pos] == '+' or self.source[self.pos] == '-'))
|
|
{
|
|
self.pos += 1;
|
|
}
|
|
self.consumeDigits(isDecDigit);
|
|
}
|
|
|
|
return .{ .kind = .number, .start = start, .len = self.pos - start };
|
|
}
|
|
|
|
fn consumeDigits(self: *Tokenizer, predicate: *const fn (u8) bool) void {
|
|
while (self.pos < self.source.len) {
|
|
const ch = self.source[self.pos];
|
|
if (predicate(ch)) {
|
|
self.pos += 1;
|
|
} else if (ch == '_') {
|
|
// Digit separator
|
|
self.pos += 1;
|
|
} else if (ch == ',') {
|
|
if (self.commaIsGrouping(predicate)) {
|
|
self.pos += 1;
|
|
} else {
|
|
break;
|
|
}
|
|
} else {
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Decide whether the comma at `self.pos` groups digits or separates
|
|
/// arguments.
|
|
///
|
|
/// It groups only when followed by EXACTLY three digits, which is what a
|
|
/// thousands group is. Checking merely for "a digit follows" is not enough:
|
|
/// it made `log(100,10)` lex as `log(10010)` and return 4.0004 instead of 2,
|
|
/// and `max(1,2)` lex as `max(12)`, which then failed as an unknown function.
|
|
/// Every such call worked only if the author happened to put a space after
|
|
/// the comma, which is why the tests and the help examples all passed.
|
|
///
|
|
/// `max(1,234)` remains ambiguous by construction: three digits follow, so it
|
|
/// reads as `max(1234)`. FR-1.8 resolves that in favour of the grouping, and a
|
|
/// space is the way to ask for two arguments.
|
|
fn commaIsGrouping(self: *Tokenizer, predicate: *const fn (u8) bool) bool {
|
|
var digits: usize = 0;
|
|
var i = self.pos + 1;
|
|
while (i < self.source.len and predicate(self.source[i])) : (i += 1) {
|
|
digits += 1;
|
|
// More than a group: not grouping, whatever follows.
|
|
if (digits > 3) return false;
|
|
}
|
|
if (digits != 3) return false;
|
|
// A fourth digit cannot appear (the loop above would have counted it), so
|
|
// the group is well formed if what follows is not another digit. What may
|
|
// follow is another group, a decimal point, an exponent, an operator, or
|
|
// the end of input.
|
|
return true;
|
|
}
|
|
|
|
/// Like consumeDigits but also treats spaces as separators (only when the
|
|
/// space is followed by a run of valid digits, so "0xFF + 1" stops at the
|
|
/// space and "0xFF and 1" is not swallowed - "and" has a non-hex letter).
|
|
/// Used for hex/oct/bin literals which display with space grouping.
|
|
fn consumeBaseDigits(self: *Tokenizer, predicate: *const fn (u8) bool) void {
|
|
while (self.pos < self.source.len) {
|
|
const ch = self.source[self.pos];
|
|
if (predicate(ch)) {
|
|
self.pos += 1;
|
|
} else if (ch == '_') {
|
|
self.pos += 1;
|
|
} else if (ch == ' ') {
|
|
if (self.spaceContinuesNumber(predicate)) {
|
|
self.pos += 1;
|
|
} else {
|
|
break;
|
|
}
|
|
} else {
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
/// After a space inside a base literal, decide whether the following text
|
|
/// is another digit group (continue the number) or a word like a keyword
|
|
/// operator (stop the number). Returns true only if the maximal
|
|
/// identifier-run right after the space consists entirely of valid digits.
|
|
fn spaceContinuesNumber(self: *Tokenizer, predicate: *const fn (u8) bool) bool {
|
|
var j = self.pos + 1;
|
|
var saw_any = false;
|
|
while (j < self.source.len and isIdentChar(self.source[j])) : (j += 1) {
|
|
if (!predicate(self.source[j])) return false;
|
|
saw_any = true;
|
|
}
|
|
return saw_any;
|
|
}
|
|
|
|
fn isIdentChar(c: u8) bool {
|
|
return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or
|
|
(c >= '0' and c <= '9') or c == '_';
|
|
}
|
|
|
|
fn readStringLiteral(self: *Tokenizer, start: usize) Token {
|
|
self.pos += 1; // consume opening quote
|
|
while (self.pos < self.source.len and self.source[self.pos] != '\'') {
|
|
self.pos += 1;
|
|
}
|
|
if (self.pos < self.source.len) {
|
|
self.pos += 1; // consume closing quote
|
|
}
|
|
return .{ .kind = .string_literal, .start = start, .len = self.pos - start };
|
|
}
|
|
|
|
fn readIdentifier(self: *Tokenizer, start: usize) Token {
|
|
while (self.pos < self.source.len) {
|
|
const ch = self.source[self.pos];
|
|
if ((ch >= 'a' and ch <= 'z') or
|
|
(ch >= 'A' and ch <= 'Z') or
|
|
(ch >= '0' and ch <= '9') or
|
|
ch == '_')
|
|
{
|
|
self.pos += 1;
|
|
} else {
|
|
break;
|
|
}
|
|
}
|
|
return .{ .kind = .identifier, .start = start, .len = self.pos - start };
|
|
}
|
|
|
|
fn isHexDigit(c: u8) bool {
|
|
return (c >= '0' and c <= '9') or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F');
|
|
}
|
|
|
|
fn isOctalDigit(c: u8) bool {
|
|
return c >= '0' and c <= '7';
|
|
}
|
|
|
|
fn isBinaryDigit(c: u8) bool {
|
|
return c == '0' or c == '1';
|
|
}
|
|
|
|
fn isDecDigit(c: u8) bool {
|
|
return c >= '0' and c <= '9';
|
|
}
|
|
};
|
|
|
|
// -- Tests --
|
|
|
|
const testing = std.testing;
|
|
|
|
test "tokenize simple arithmetic" {
|
|
var tok = Tokenizer.init("2 + 3 * 4", .standard);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.plus, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.star, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize hex number" {
|
|
var tok = Tokenizer.init("0xFF", .programmer);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("0xFF", t.text("0xFF"));
|
|
}
|
|
|
|
test "tokenize binary number" {
|
|
var tok = Tokenizer.init("0b1010", .programmer);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("0b1010", t.text("0b1010"));
|
|
}
|
|
|
|
test "tokenize octal number" {
|
|
var tok = Tokenizer.init("0o777", .programmer);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("0o777", t.text("0o777"));
|
|
}
|
|
|
|
test "tokenize shift operators" {
|
|
var tok = Tokenizer.init("x << 3 >> 1 >>> 2", .programmer);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.shift_left, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.shift_right, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.shift_right_logical, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize star_star" {
|
|
var tok = Tokenizer.init("2**10", .programmer);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.star_star, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize number with underscores" {
|
|
var tok = Tokenizer.init("1_000_000", .standard);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("1_000_000", t.text("1_000_000"));
|
|
}
|
|
|
|
test "tokenize hex with underscores" {
|
|
var tok = Tokenizer.init("0xFF_FF", .programmer);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("0xFF_FF", t.text("0xFF_FF"));
|
|
}
|
|
|
|
test "tokenize number with commas" {
|
|
var tok = Tokenizer.init("1,000,000", .standard);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("1,000,000", t.text("1,000,000"));
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize hex with spaces" {
|
|
var tok = Tokenizer.init("0xFF FF FF FF", .programmer);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("0xFF FF FF FF", t.text("0xFF FF FF FF"));
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize binary with spaces" {
|
|
var tok = Tokenizer.init("0b1111 0000", .programmer);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("0b1111 0000", t.text("0b1111 0000"));
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize octal with spaces" {
|
|
var tok = Tokenizer.init("0o777 111", .programmer);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("0o777 111", t.text("0o777 111"));
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize base literal space before operator stops" {
|
|
var tok = Tokenizer.init("0b1010 + 1", .programmer);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.plus, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize comma not eaten in function args" {
|
|
var tok = Tokenizer.init("max(1, 2)", .standard);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind); // max
|
|
try testing.expectEqual(TokenKind.left_paren, tok.next().kind); // (
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind); // 1
|
|
try testing.expectEqual(TokenKind.comma, tok.next().kind); // ,
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind); // 2
|
|
try testing.expectEqual(TokenKind.right_paren, tok.next().kind); // )
|
|
}
|
|
|
|
test "tokenize function call" {
|
|
var tok = Tokenizer.init("sin(3.14)", .standard);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.left_paren, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.right_paren, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize floating point with exponent" {
|
|
var tok = Tokenizer.init("1.5e10", .standard);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("1.5e10", t.text("1.5e10"));
|
|
}
|
|
|
|
test "tokenize negative exponent" {
|
|
var tok = Tokenizer.init("2.5e-3", .standard);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("2.5e-3", t.text("2.5e-3"));
|
|
}
|
|
|
|
test "tokenize number starting with dot" {
|
|
var tok = Tokenizer.init(".5", .standard);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings(".5", t.text(".5"));
|
|
}
|
|
|
|
test "tokenize assignment" {
|
|
var tok = Tokenizer.init("X = 42", .standard);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.equals, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize all bitwise ops" {
|
|
var tok = Tokenizer.init("a & b | c ^ ~d", .programmer);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.ampersand, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.pipe, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.caret, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.tilde, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize empty string" {
|
|
var tok = Tokenizer.init("", .standard);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize whitespace only" {
|
|
var tok = Tokenizer.init(" \t\n ", .standard);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize semicolon" {
|
|
var tok = Tokenizer.init(";", .standard);
|
|
try testing.expectEqual(TokenKind.semicolon, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize bare less-than is invalid" {
|
|
var tok = Tokenizer.init("<", .programmer);
|
|
try testing.expectEqual(TokenKind.invalid, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize bare greater-than is invalid" {
|
|
var tok = Tokenizer.init(">", .programmer);
|
|
try testing.expectEqual(TokenKind.invalid, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize lone dot is invalid" {
|
|
var tok = Tokenizer.init(".x", .standard);
|
|
try testing.expectEqual(TokenKind.invalid, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize unrecognized character is invalid" {
|
|
// '@' is not handled by any dispatch case, so it hits the else branch
|
|
var tok = Tokenizer.init("@", .standard);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.invalid, t.kind);
|
|
try testing.expectEqual(@as(usize, 1), t.len);
|
|
}
|
|
|
|
test "tokenize base literal does not take a comma as a separator" {
|
|
// Base literals group with spaces and underscores (FR-1.8); commas are the
|
|
// decimal grouping character. Accepting them here only reintroduced the
|
|
// argument-separator ambiguity in another place.
|
|
var tok = Tokenizer.init("0xFF,FF", .programmer);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings("0xFF", t.text("0xFF,FF"));
|
|
try testing.expectEqual(TokenKind.comma, tok.next().kind);
|
|
// The trailing "FF" is a bare identifier now, not a continuation of the
|
|
// literal, which is exactly the point: it is not silently absorbed.
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "tokenize base literal still groups with spaces and underscores" {
|
|
var tok = Tokenizer.init("0xFF FF", .programmer);
|
|
try testing.expectEqualStrings("0xFF FF", tok.next().text("0xFF FF"));
|
|
var underscored = Tokenizer.init("0xFF_FF", .programmer);
|
|
try testing.expectEqualStrings("0xFF_FF", underscored.next().text("0xFF_FF"));
|
|
}
|
|
|
|
test "parseNumber huge decimal exceeds u64 and falls back to float here" {
|
|
// The tokenizer's own integer channel is a u64, so a value this large has no
|
|
// `int_value` at this layer. That is NOT a precision limit of the engine:
|
|
// the evaluator re-parses the literal text into an exact rational (see
|
|
// evaluator.literalToNumber), so `99999999999999999999999999` still
|
|
// evaluates exactly. This test pins the tokenizer's contract only.
|
|
const result = try parseNumber("99999999999999999999999999");
|
|
try testing.expectEqual(@as(?u64, null), result.int_value);
|
|
try testing.expectEqual(Base.decimal, result.base);
|
|
try testing.expect(result.float > 1e25);
|
|
}
|
|
|
|
// -- parseNumber tests --
|
|
|
|
test "parseNumber decimal integer" {
|
|
const result = try parseNumber("42");
|
|
try testing.expectEqual(@as(f64, 42.0), result.float);
|
|
try testing.expectEqual(@as(?u64, 42), result.int_value);
|
|
try testing.expectEqual(Base.decimal, result.base);
|
|
}
|
|
|
|
test "parseNumber decimal with underscores" {
|
|
const result = try parseNumber("1_000_000");
|
|
try testing.expectEqual(@as(?u64, 1_000_000), result.int_value);
|
|
}
|
|
|
|
test "parseNumber hex" {
|
|
const result = try parseNumber("0xFF");
|
|
try testing.expectEqual(@as(?u64, 255), result.int_value);
|
|
try testing.expectEqual(Base.hex, result.base);
|
|
}
|
|
|
|
test "parseNumber binary" {
|
|
const result = try parseNumber("0b1010");
|
|
try testing.expectEqual(@as(?u64, 10), result.int_value);
|
|
try testing.expectEqual(Base.binary, result.base);
|
|
}
|
|
|
|
test "parseNumber octal" {
|
|
const result = try parseNumber("0o777");
|
|
try testing.expectEqual(@as(?u64, 511), result.int_value);
|
|
try testing.expectEqual(Base.octal, result.base);
|
|
}
|
|
|
|
test "parseNumber float" {
|
|
const result = try parseNumber("3.14");
|
|
try testing.expectApproxEqAbs(@as(f64, 3.14), result.float, 1e-10);
|
|
try testing.expectEqual(@as(?u64, null), result.int_value);
|
|
try testing.expectEqual(Base.decimal, result.base);
|
|
}
|
|
|
|
test "parseNumber float with exponent" {
|
|
const result = try parseNumber("1.5e10");
|
|
try testing.expectEqual(@as(f64, 1.5e10), result.float);
|
|
try testing.expectEqual(@as(?u64, null), result.int_value);
|
|
}
|
|
|
|
test "parseNumber hex with underscores" {
|
|
const result = try parseNumber("0xFF_FF");
|
|
try testing.expectEqual(@as(?u64, 0xFFFF), result.int_value);
|
|
try testing.expectEqual(Base.hex, result.base);
|
|
}
|
|
|
|
test "parseNumber with commas" {
|
|
const result = try parseNumber("1,000,000");
|
|
try testing.expectEqual(@as(?u64, 1_000_000), result.int_value);
|
|
try testing.expectEqual(Base.decimal, result.base);
|
|
}
|
|
|
|
// -- ImplicitMulStream tests --
|
|
|
|
test "no implicit mul: spaces are just whitespace" {
|
|
// Spaces between tokens don't create implicit multiplication
|
|
var tok = Tokenizer.init("2 3", .standard);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
// -- The comma grouping rule --
|
|
//
|
|
// A comma groups digits only when exactly three digits follow. The old rule was
|
|
// "a digit follows", which silently merged function arguments: `log(100,10)`
|
|
// became `log(10010)` and returned 4.0004 instead of 2. Every affected call
|
|
// worked if the author put a space after the comma, which is why the old tests
|
|
// and every documented example passed.
|
|
|
|
test "comma groups digits only in threes" {
|
|
// Grouped: consumed as one number.
|
|
for ([_][]const u8{ "1,000", "1,234,567", "12,345", "123,456,789" }) |source| {
|
|
var tok = Tokenizer.init(source, .standard);
|
|
const t = tok.next();
|
|
try testing.expectEqual(TokenKind.number, t.kind);
|
|
try testing.expectEqualStrings(source, t.text(source));
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
}
|
|
|
|
test "comma with the wrong number of digits is a separate token" {
|
|
// One, two or four digits are not a thousands group, so the comma stays a
|
|
// comma. This is what makes `log(100,10)` and `max(1,2)` parse as two
|
|
// arguments.
|
|
for ([_][]const u8{ "100,10", "1,2", "1,00", "1,0000" }) |source| {
|
|
var tok = Tokenizer.init(source, .standard);
|
|
const first = tok.next();
|
|
try testing.expectEqual(TokenKind.number, first.kind);
|
|
try testing.expectEqual(TokenKind.comma, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
}
|
|
|
|
test "comma at the end of input is a separate token" {
|
|
var tok = Tokenizer.init("1,", .standard);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.comma, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.eof, tok.next().kind);
|
|
}
|
|
|
|
test "comma followed by a non-digit is a separate token" {
|
|
var tok = Tokenizer.init("max(1, 2)", .standard);
|
|
try testing.expectEqual(TokenKind.identifier, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.left_paren, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.comma, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.number, tok.next().kind);
|
|
try testing.expectEqual(TokenKind.right_paren, tok.next().kind);
|
|
}
|
|
|
|
test "a grouped literal is still exact and keeps its full text" {
|
|
// The exact tier re-parses the literal text, so the separators have to remain
|
|
// in the token for it to see them.
|
|
var tok = Tokenizer.init("9,007,199,254,740,993", .standard);
|
|
const t = tok.next();
|
|
try testing.expectEqualStrings("9,007,199,254,740,993", t.text("9,007,199,254,740,993"));
|
|
}
|