tally/engine/src/programmer.zig

504 lines
19 KiB
Zig

//! Programmer mode evaluator for Tally.
//!
//! All operations use exact integer arithmetic (u128 storage), with results masked
//! to the configured width. No floating-point involved. The bitwise operators, the
//! shifts and the rotations live in `bitwise.zig`, which standard mode shares; what
//! is here is the wrapping arithmetic and the walk over the tree.
const std = @import("std");
const Allocator = std.mem.Allocator;
const ast = @import("ast.zig");
const Expr = ast.Expr;
const BinaryOp = ast.BinaryOp;
const Integer = @import("Integer.zig");
const BitWidth = Integer.BitWidth;
const Signedness = Integer.Signedness;
const parser_mod = @import("parser.zig");
const Parser = parser_mod.Parser;
const bitwise = @import("bitwise.zig");
/// What programmer-mode evaluation can fail with: the parse, the fixed-width
/// operators, and its own arithmetic and name errors.
pub const Error = error{
DivisionByZero,
DomainError,
InvalidNumber,
InvalidOperandType,
Overflow,
UnknownFunction,
UnknownVariable,
} || parser_mod.Error || bitwise.Error;
/// What programmer mode needs to know: the integer type to compute in, and one
/// display preference that does not affect arithmetic.
pub const Config = struct {
width: BitWidth = .bits64,
signedness: Signedness = .signed,
/// Byte order for the HEX and ASCII rows only. Defaults to big-endian so the
/// HEX row reads as the number itself (matching DEC/OCT/BIN); the little-endian
/// view (x86 memory layout) is available via the toggle.
display_endian: std.builtin.Endian = .big,
/// A value of the configured integer type. Every literal and every result goes
/// through here, so the width and signedness are attached once rather than
/// tracked alongside a bare pattern.
pub fn value(self: Config, raw: u128) Integer {
return .{ .raw = raw & self.width.mask(), .width = self.width, .signedness = self.signedness };
}
};
/// Evaluate an AST in programmer mode, producing an exact integer result.
pub fn evalProgrammer(config: Config, expr: *const Expr) Error!Integer {
return evalExpr(config, expr);
}
/// Recursively evaluate an expression.
fn evalExpr(config: Config, expr: *const Expr) Error!Integer {
switch (expr.*) {
.number => |n| {
if (n.int_value) |int_val| {
return config.value(int_val);
}
// Float literal in programmer mode: truncate to integer.
// (Number literals are always non-negative; unary minus is a
// separate operator handled below.)
//
// The range check is not optional: `@intFromFloat` on an out-of-range
// value is illegal behaviour, and `tally -p '1e40'` aborted the process
// before this guard existed.
const float_value = n.float_value;
if (!std.math.isFinite(float_value) or float_value < 0) return Error.DomainError;
if (float_value >= 340282366920938463463374607431768211456.0) return Error.Overflow;
return config.value(@intFromFloat(float_value));
},
.string_literal => |text| {
// Pack ASCII bytes into integer.
// Big-endian packing: first char -> most significant used byte.
const max_bytes = @as(usize, config.width.bits()) / 8;
if (text.len > max_bytes) return Error.Overflow;
var result: u128 = 0;
for (text) |byte| {
if (byte > 0x7F) return Error.InvalidNumber;
result = (result << 8) | byte;
}
return config.value(result);
},
.variable => {
return Error.UnknownVariable;
},
.assignment => {
return Error.InvalidOperandType;
},
.unary => |u| {
const operand = try evalExpr(config, u.operand);
return switch (u.op) {
.negate => bitwise.negate(operand),
.bitwise_not => bitwise.not(operand),
};
},
.binary => |b| {
const left = try evalExpr(config, b.left);
const right = try evalExpr(config, b.right);
return evalBinaryOp(b.op, left, right);
},
.call => {
// No function calls in programmer mode
return Error.UnknownFunction;
},
}
}
/// Evaluate a binary operation on two values of the configured type.
///
/// The bitwise operators, shifts and rotations are not here: they live in
/// `bitwise.zig`, which standard mode uses too, so the two modes cannot drift
/// apart again. What remains is the arithmetic, which genuinely differs between the
/// modes: it wraps at the width here and is exact rational arithmetic there.
fn evalBinaryOp(op: BinaryOp, left_in: Integer, right_in: Integer) Error!Integer {
const mask = left_in.width.mask();
const left = left_in.masked();
const right = right_in.masked();
const raw: u128 = switch (op) {
.add => left +% right,
.sub => left -% right,
.mul => left *% right,
.div => blk: {
if (right == 0) return Error.DivisionByZero;
break :blk left / right;
},
.mod => blk: {
if (right == 0) return Error.DivisionByZero;
break :blk left % right;
},
.pow => blk: {
// Integer exponentiation
var base = left;
var exp = right;
var acc: u128 = 1;
while (exp > 0) : (exp >>= 1) {
if (exp & 1 != 0) acc = (acc *% base) & mask;
base = (base *% base) & mask;
}
break :blk acc;
},
// The fixed-width operators, in the shared implementation. `inline else`
// resolves the operator at comptime, so an operator added to `BinaryOp`
// that `bitwise.fromBinaryOp` does not know is a compile error here.
inline else => |fixed_op| return try bitwise.apply(
comptime bitwise.fromBinaryOp(fixed_op).?,
left_in,
right_in,
),
};
return left_in.withRaw(raw);
}
/// High-level: parse and evaluate a string in programmer mode.
pub fn evalProgrammerString(allocator: Allocator, source: []const u8, config: Config) Error!Integer {
var p = Parser.init(allocator, source);
const expr = try p.parse();
// Same ownership rule as evalStringInfo: the tree is ours to release, and the
// returned Integer does not borrow from it.
defer parser_mod.freeExpr(allocator, expr);
return evalProgrammer(config, expr);
}
// -- Tests --
const testing = std.testing;
fn testProg(source: []const u8) !Integer {
return testProgWith(source, .{});
}
/// Tests care about the width and signedness, never about the display byte order,
/// so they pass a partial config.
fn testProgWith(source: []const u8, config: Config) !Integer {
var arena = std.heap.ArenaAllocator.init(std.heap.page_allocator);
defer _ = arena.deinit();
return evalProgrammerString(arena.allocator(), source, config);
}
test "prog: simple number" {
const result = try testProg("42");
try testing.expectEqual(@as(u128, 42), result.unsignedValue());
}
test "prog: hex number" {
const result = try testProg("0xFF");
try testing.expectEqual(@as(u128, 255), result.unsignedValue());
}
test "prog: binary number" {
const result = try testProg("0b1010");
try testing.expectEqual(@as(u128, 10), result.unsignedValue());
}
test "prog: addition" {
const result = try testProg("10 + 20");
try testing.expectEqual(@as(u128, 30), result.unsignedValue());
}
test "prog: subtraction wrapping" {
// 5 - 10 in 8-bit unsigned wraps
const result = try testProgWith("5 - 10", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 251), result.unsignedValue()); // 256 - 5
try testing.expectEqual(@as(i128, -5), result.signedValue());
}
test "prog: multiplication" {
const result = try testProg("6 * 7");
try testing.expectEqual(@as(u128, 42), result.unsignedValue());
}
test "prog: multiplication overflow 8-bit" {
const result = try testProgWith("200 * 2", .{ .width = .bits8 });
// 400 & 0xFF = 144
try testing.expectEqual(@as(u128, 144), result.unsignedValue());
}
test "prog: division" {
const result = try testProg("100 / 4");
try testing.expectEqual(@as(u128, 25), result.unsignedValue());
}
test "prog: division by zero" {
const result = testProg("10 / 0");
try testing.expectError(Error.DivisionByZero, result);
}
test "prog: modulo" {
const result = try testProg("10 % 3");
try testing.expectEqual(@as(u128, 1), result.unsignedValue());
}
test "prog: power" {
const result = try testProg("2 ** 10");
try testing.expectEqual(@as(u128, 1024), result.unsignedValue());
}
test "prog: bitwise AND" {
const result = try testProg("0xFF & 0x0F");
try testing.expectEqual(@as(u128, 0x0F), result.unsignedValue());
}
test "prog: bitwise OR" {
const result = try testProg("0xF0 | 0x0F");
try testing.expectEqual(@as(u128, 0xFF), result.unsignedValue());
}
test "prog: bitwise XOR" {
const result = try testProg("0xFF xor 0x0F");
try testing.expectEqual(@as(u128, 0xF0), result.unsignedValue());
}
test "prog: caret is power not XOR" {
// 0x2 ^ 0x3 = 2^3 = 8 (power), NOT 1 (XOR)
const result = try testProg("0x2 ^ 0x3");
try testing.expectEqual(@as(u128, 8), result.unsignedValue());
}
test "prog: and/or/not keywords" {
const a = try testProg("0xFF and 0x0F");
try testing.expectEqual(@as(u128, 0x0F), a.unsignedValue());
const o = try testProg("0xF0 or 0x0F");
try testing.expectEqual(@as(u128, 0xFF), o.unsignedValue());
const n = try testProgWith("not 0x0F", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0xF0), n.unsignedValue());
}
test "prog: bitwise NOT 8-bit" {
const result = try testProgWith("~0x0F", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0xF0), result.unsignedValue());
}
test "prog: bitwise NOT 16-bit" {
const result = try testProgWith("~0x00FF", .{ .width = .bits16 });
try testing.expectEqual(@as(u128, 0xFF00), result.unsignedValue());
}
test "prog: bitwise NOT 32-bit" {
const result = try testProgWith("~0", .{ .width = .bits32 });
try testing.expectEqual(@as(u128, 0xFFFF_FFFF), result.unsignedValue());
}
test "prog: shift left" {
const result = try testProg("1 << 8");
try testing.expectEqual(@as(u128, 256), result.unsignedValue());
}
test "prog: shift left past the width shifts everything out" {
// The distance used to be clamped to width - 1, so this gave 128.
const result = try testProgWith("1 << 8", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0), result.unsignedValue());
// One less than the width still keeps the bit.
const edge = try testProgWith("1 << 7", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 128), edge.unsignedValue());
}
test "prog: logical shift right" {
const result = try testProgWith("0x80 >>> 4", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0x08), result.unsignedValue());
}
test "prog: arithmetic shift right (sign bit preserved)" {
// 0x80 in 8-bit is -128; >> 1 should give 0xC0 (-64)
const result = try testProgWith("0x80 >> 1", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0xC0), result.unsignedValue());
try testing.expectEqual(@as(i128, -64), result.signedValue());
}
test "prog: arithmetic shift right (positive)" {
// 0x40 in 8-bit is positive; >> 1 should give 0x20
const result = try testProgWith("0x40 >> 1", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0x20), result.unsignedValue());
}
test "prog: rotate left 8-bit" {
// 0x81 rol 1 in 8-bit: bit 7 wraps to bit 0 -> 0x03
const result = try testProgWith("0x81 rol 1", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0x03), result.unsignedValue());
}
test "prog: rotate right 8-bit" {
// 0x81 ror 1 in 8-bit: bit 0 wraps to bit 7 -> 0xC0
const result = try testProgWith("0x81 ror 1", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0xC0), result.unsignedValue());
}
test "prog: negation two's complement" {
const result = try testProgWith("-1", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0xFF), result.unsignedValue());
try testing.expectEqual(@as(i128, -1), result.signedValue());
}
test "prog: negation 16-bit" {
const result = try testProgWith("-42", .{ .width = .bits16 });
try testing.expectEqual(@as(i128, -42), result.signedValue());
}
test "prog: complex expression" {
// (0xFF & 0x0F) | (1 << 4) = 0x0F | 0x10 = 0x1F
const result = try testProg("(0xFF & 0x0F) | (1 << 4)");
try testing.expectEqual(@as(u128, 0x1F), result.unsignedValue());
}
test "prog: precedence AND before OR" {
// 0xF0 | 0xFF & 0x0F = 0xF0 | (0xFF & 0x0F) = 0xF0 | 0x0F = 0xFF
const result = try testProg("0xF0 | 0xFF & 0x0F");
try testing.expectEqual(@as(u128, 0xFF), result.unsignedValue());
}
test "prog: chained shifts" {
const result = try testProg("1 << 4 << 2");
// Left-associative: (1 << 4) << 2 = 16 << 2 = 64
try testing.expectEqual(@as(u128, 64), result.unsignedValue());
}
test "prog: mask applied to input" {
// 0x1FF in 8-bit mode should be masked to 0xFF
const result = try testProgWith("0x1FF", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0xFF), result.unsignedValue());
}
test "prog: 32-bit operations" {
const result = try testProgWith("0xFFFF_FFFF + 1", .{ .width = .bits32 });
try testing.expectEqual(@as(u128, 0), result.unsignedValue());
}
test "prog: 64-bit max" {
const result = try testProgWith("~0", .{ .width = .bits64 });
try testing.expectEqual(@as(u128, 0xFFFF_FFFF_FFFF_FFFF), result.unsignedValue());
}
test "prog: negative float input" {
// -5.0 as a float in programmer mode should become two's complement
const result = try testProgWith("-5", .{ .width = .bits8 });
try testing.expectEqual(@as(i128, -5), result.signedValue());
}
test "prog: variable reference errors" {
var arena = std.heap.ArenaAllocator.init(std.heap.page_allocator);
defer _ = arena.deinit();
const result = evalProgrammerString(arena.allocator(), "x", .{});
try testing.expectError(Error.UnknownVariable, result);
}
test "prog: assignment errors" {
var arena = std.heap.ArenaAllocator.init(std.heap.page_allocator);
defer _ = arena.deinit();
const result = evalProgrammerString(arena.allocator(), "X = 5", .{});
try testing.expectError(Error.InvalidOperandType, result);
}
test "prog: function call errors" {
var arena = std.heap.ArenaAllocator.init(std.heap.page_allocator);
defer _ = arena.deinit();
const result = evalProgrammerString(arena.allocator(), "sin(1)", .{});
try testing.expectError(Error.UnknownFunction, result);
}
test "prog: ASCII literal single char" {
const result = try testProg("'A'");
try testing.expectEqual(@as(u128, 0x41), result.unsignedValue());
}
test "prog: ASCII literal multi char" {
const result = try testProg("'ELF'");
try testing.expectEqual(@as(u128, 0x454C46), result.unsignedValue());
}
test "prog: ASCII literal full word" {
const result = try testProg("'ascii'");
try testing.expectEqual(@as(u128, 0x6173636969), result.unsignedValue());
}
test "prog: ASCII literal in expression" {
const result = try testProg("'A' | 0x20");
// 0x41 | 0x20 = 0x61 = 'a'
try testing.expectEqual(@as(u128, 0x61), result.unsignedValue());
}
test "prog: ASCII literal overflow 8-bit" {
var arena = std.heap.ArenaAllocator.init(std.heap.page_allocator);
defer _ = arena.deinit();
const result = evalProgrammerString(arena.allocator(), "'AB'", .{ .width = .bits8 });
try testing.expectError(Error.Overflow, result);
}
test "prog: float literal truncates to integer" {
// 3.14 has no int_value, so the float branch truncates to 3
const result = try testProg("3.14");
try testing.expectEqual(@as(u128, 3), result.unsignedValue());
}
test "prog: arithmetic shift right past the width leaves the sign fill" {
// 0xFF in 8-bit is -1; shifting a negative value all the way out leaves every
// bit set, which is still -1. The distance used to be clamped to width - 1,
// which reached the same answer here for the wrong reason.
const result = try testProgWith("0xFF >> 20", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0xFF), result.unsignedValue());
try testing.expectEqual(@as(i128, -1), result.signedValue());
}
test "prog: logical shift right past the width shifts everything out" {
// Used to clamp the distance to 7 and give 1.
const result = try testProgWith("0xFF >>> 20", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0), result.unsignedValue());
// One less than the width still keeps the bottom bit.
const edge = try testProgWith("0xFF >>> 7", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 1), edge.unsignedValue());
}
test "prog: arithmetic shift right past the width, positive value" {
// 0x40 in 8-bit is positive, so the fill is zeros and everything shifts out.
const result = try testProgWith("0x40 >> 20", .{ .width = .bits8 });
try testing.expectEqual(@as(u128, 0), result.unsignedValue());
}
test "no leak: evalProgrammerString releases the parsed tree" {
// testing.allocator rather than an arena, so a retained AST fails the test.
const config: Config = .{};
const good = [_][]const u8{ "0xFF and 0x0F", "1 << 8", "not 0", "0b1010 xor 0b0101", "5 rol 2" };
for (good) |source| {
_ = try evalProgrammerString(std.testing.allocator, source, config);
}
const bad = [_][]const u8{ "0xFF and", "1 <<", "(1 | 2" };
for (bad) |source| {
_ = evalProgrammerString(std.testing.allocator, source, config) catch continue;
return error.TestUnexpectedResult;
}
}
test "programmer mode: a float literal out of range errors instead of aborting" {
const config: Config = .{};
// `tally -p '1e40'` used to abort the process here: @intFromFloat on a value
// past u128 is illegal behaviour, and only 3.14 was ever tested.
try std.testing.expectError(
Error.Overflow,
evalProgrammerString(std.testing.allocator, "1e40", config),
);
try std.testing.expectError(
Error.Overflow,
evalProgrammerString(std.testing.allocator, "1e100", config),
);
// Still truncates the values that do fit.
const small = try evalProgrammerString(std.testing.allocator, "3.99", config);
try std.testing.expectEqual(@as(u128, 3), small.unsignedValue());
const large = try evalProgrammerString(std.testing.allocator, "1e30", config);
try std.testing.expect(large.unsignedValue() != 0);
}
test "programmer mode: an infinite or NaN literal is a domain error" {
const config: Config = .{};
// 10^400 overflows the exact tier's float projection to infinity.
try std.testing.expectError(
Error.DomainError,
evalProgrammerString(std.testing.allocator, "1e400", config),
);
}