remove top bit, rename masked, introduce canonical format option

This commit is contained in:
Emil Lerch 2026-08-05 17:41:33 -07:00
parent 5e46d9e706
commit ee33c679bc
Signed by: lobo
GPG key ID: A7B62D657EF764F8
6 changed files with 96 additions and 95 deletions

View file

@ -862,7 +862,7 @@ Each job went to the type that owns it:
screen, so there is no budget to pass. The dead `formatAmount` went away with it,
and the null-on-failure contract became `error.WriteFailed`, so the nine CLI
`orelse return unformattableResult()` sites became `catch return`.
- **`Integer.zig` got `displayWidthFor`**, the one thing in the file that formatted
- **`Integer.zig` got it as `BitWidth.displayFor`**, the one thing in the file that formatted
nothing: it returns a `BitWidth`, which is `Integer`'s own type.
Two rules stayed in the engine because they are facts about values rather than
@ -921,8 +921,9 @@ being rendered at a width it did not come from.
evaluation only objects to a genuine dependency cycle), but it would point the
low-level type at the high-level display module.
- `formatter.zig` keeps what is not a fixed-width integer: floats, exact `Number`s,
scientific notation, money, and `displayWidthFor`. It re-exports the two grouping
functions the TUI uses on partially typed input.
scientific notation, money, and the display width. It re-exports the two grouping
functions the TUI uses on partially typed input. (Task 5.17 later distributed all
of that and deleted the file.)
Also in this task, the nested `IntType` went away: see design 2.5. `bitwise.apply`
takes two `Integer`s and returns one, `programmer.evalExpr` threads `Integer` instead

View file

@ -25,33 +25,28 @@ width: BitWidth = .bits64,
/// Whether the top bit is a sign.
signedness: Signedness = .signed,
/// The top bit's position, whether or not this value treats it as a sign.
pub fn topBit(self: Integer) u128 {
return @as(u128, 1) << @intCast(self.width.bits() - 1);
}
/// The pattern, truncated to the width.
pub fn masked(self: Integer) u128 {
return self.raw & self.width.mask();
}
/// True when the pattern denotes a negative number. An unsigned value has no
/// negative range, so `>>` is a zero fill there and a large shift distance is large
/// rather than negative.
pub fn isNegative(self: Integer) bool {
return self.signedness == .signed and self.masked() & self.topBit() != 0;
const sign_bit = @as(u128, 1) << @intCast(self.width.bits() - 1);
return self.signedness == .signed and self.unsignedValue() & sign_bit != 0;
}
/// Interpret as a signed value, sign-extended from the width.
pub fn signedValue(self: Integer) i128 {
const m = self.masked();
if (self.isNegative()) return @bitCast(m | ~self.width.mask());
return @intCast(m);
const pattern = self.unsignedValue();
if (self.isNegative()) return @bitCast(pattern | ~self.width.mask());
return @intCast(pattern);
}
/// Interpret as an unsigned value, which is just the mask.
/// Interpret as an unsigned value.
///
/// Also the bit pattern itself, truncated to the width, since an unsigned reading of
/// two's complement is the pattern: code doing bit work rather than arithmetic calls
/// this too. `raw` may carry bits above the width, so this is what readers use.
pub fn unsignedValue(self: Integer) u128 {
return self.masked();
return self.raw & self.width.mask();
}
/// The same width and signedness, a different pattern. How operations return their
@ -121,9 +116,6 @@ pub const FormatOptions = struct {
/// Big-endian reads as the number itself; little-endian is the x86 memory view.
/// A prefixed form is always canonical, since `0x...` names the value.
endian: std.builtin.Endian = .big,
/// What the clipboard wants: no separators, and a prefix where one exists.
pub const plain: FormatOptions = .{ .separators = false, .prefix = true };
};
/// This value in `notation`, ready to print.
@ -139,7 +131,7 @@ pub const Format = struct {
/// Render into `w`. This is the `{f}` implementation, so `int.fmt(.hex, .{})`
/// prints directly.
pub fn format(self: Format, w: *Writer) Writer.Error!void {
const value = self.value.masked();
const value = self.value.unsignedValue();
switch (self.notation) {
.hex => {
if (self.options.prefix) try w.writeAll("0x");
@ -228,7 +220,7 @@ pub const Format = struct {
const count = self.byteCount();
const msb_index = if (self.options.endian == .big) k else count - 1 - k;
const shift: u7 = @intCast((count - 1 - msb_index) * 8);
return @intCast((self.value.masked() >> shift) & 0xFF);
return @intCast((self.value.unsignedValue() >> shift) & 0xFF);
}
fn byteCount(self: Format) usize {
@ -266,31 +258,31 @@ pub const BitWidth = enum(u8) {
pub fn bits(self: BitWidth) u8 {
return @intFromEnum(self);
}
};
/// The width to print the hex, octal and binary rows at, for a standard-mode value
/// that has no configured width (FR-1.9).
///
/// The narrowest of the standard widths that holds `value`, so a small number does
/// not come out padded to 64 bits: 1 prints as `01`, and 256 steps up to `01 00`.
/// It rounds up to a width a reader recognises rather than to a bit count, which is
/// what makes the hex row read as whole bytes.
///
/// Unsigned only: it counts significant bits of the pattern, so a negative value's
/// two's complement form would always report the full width. Callers reach this
/// only for non-negative integers, which is also what FR-1.9 promises. Zero has no
/// significant bits and prints at the narrowest width.
///
/// Programmer mode does not use this; there the width is the user's setting.
pub fn displayWidthFor(value: u128) BitWidth {
return switch (128 - @clz(value)) {
0...8 => .bits8,
9...16 => .bits16,
17...32 => .bits32,
33...64 => .bits64,
else => .bits128,
};
}
/// The width to print the hex, octal and binary rows at, for a standard-mode
/// value that has no configured width (FR-1.9).
///
/// The narrowest of the standard widths that holds `value`, so a small number
/// does not come out padded to 64 bits: 1 prints as `01`, and 256 steps up to
/// `01 00`. It rounds up to a width a reader recognises rather than to a bit
/// count, which is what makes the hex row read as whole bytes.
///
/// Unsigned only: it counts significant bits of the pattern, so a negative
/// value's two's complement form would always report the full width. Callers
/// reach this only for non-negative integers, which is also what FR-1.9
/// promises. Zero has no significant bits and prints at the narrowest width.
///
/// Programmer mode does not use this; there the width is the user's setting.
pub fn displayFor(value: u128) BitWidth {
return switch (128 - @clz(value)) {
0...8 => .bits8,
9...16 => .bits16,
17...32 => .bits32,
33...64 => .bits64,
else => .bits128,
};
}
};
/// Whether the top bit of a pattern is a sign.
pub const Signedness = enum {
@ -302,6 +294,14 @@ pub const Signedness = enum {
const testing = std.testing;
/// The no-separators, prefixed form: `0xFF` rather than `FF`.
///
/// This was `FormatOptions.plain`, documented as what the clipboard wants, until a
/// look at the callers found all 24 of them in the tests below and none in a
/// frontend. The TUI has no yank command yet (tasks.md open item 14), so it lives
/// here until something outside the tests asks for it.
const canonical: FormatOptions = .{ .separators = false, .prefix = true };
test "BitWidth.mask" {
try testing.expectEqual(@as(u128, 0xFF), BitWidth.bits8.mask());
try testing.expectEqual(@as(u128, 0xFFFF), BitWidth.bits16.mask());
@ -404,21 +404,21 @@ test "hex: display groups per byte, plain form carries the prefix" {
var buf: [256]u8 = undefined;
try testing.expectEqualStrings("FF", try show(&buf, at(0xFF, .bits8), .hex, .{}));
try testing.expectEqualStrings("0xFF", try show(&buf, at(0xFF, .bits8), .hex, .plain));
try testing.expectEqualStrings("0xFF", try show(&buf, at(0xFF, .bits8), .hex, canonical));
try testing.expectEqualStrings("AB CD", try show(&buf, at(0xABCD, .bits16), .hex, .{}));
try testing.expectEqualStrings("0xABCD", try show(&buf, at(0xABCD, .bits16), .hex, .plain));
try testing.expectEqualStrings("0xABCD", try show(&buf, at(0xABCD, .bits16), .hex, canonical));
try testing.expectEqualStrings("DE AD BE EF", try show(&buf, at(0xDEADBEEF, .bits32), .hex, .{}));
try testing.expectEqualStrings("0xDEADBEEF", try show(&buf, at(0xDEADBEEF, .bits32), .hex, .plain));
try testing.expectEqualStrings("0xDEADBEEF", try show(&buf, at(0xDEADBEEF, .bits32), .hex, canonical));
const quad = at(0x0123456789ABCDEF, .bits64);
try testing.expectEqualStrings("01 23 45 67 89 AB CD EF", try show(&buf, quad, .hex, .{}));
try testing.expectEqualStrings("0x0123456789ABCDEF", try show(&buf, quad, .hex, .plain));
try testing.expectEqualStrings("0x0123456789ABCDEF", try show(&buf, quad, .hex, canonical));
// Zero still fills the width.
try testing.expectEqualStrings("00 00 00 00", try show(&buf, at(0, .bits32), .hex, .{}));
try testing.expectEqualStrings("0x00000000", try show(&buf, at(0, .bits32), .hex, .plain));
try testing.expectEqualStrings("0x00000000", try show(&buf, at(0, .bits32), .hex, canonical));
}
test "hex: little-endian reverses the display, never the prefixed form" {
@ -443,7 +443,7 @@ test "binary: display groups per nibble" {
var buf: [512]u8 = undefined;
try testing.expectEqualStrings("1111 1111", try show(&buf, at(0xFF, .bits8), .binary, .{}));
try testing.expectEqualStrings("0b11111111", try show(&buf, at(0xFF, .bits8), .binary, .plain));
try testing.expectEqualStrings("0b11111111", try show(&buf, at(0xFF, .bits8), .binary, canonical));
try testing.expectEqualStrings("1010 0101", try show(&buf, at(0b1010_0101, .bits8), .binary, .{}));
try testing.expectEqualStrings(
@ -452,7 +452,7 @@ test "binary: display groups per nibble" {
);
try testing.expectEqualStrings(
"0b1111000010101100",
try show(&buf, at(0xF0AC, .bits16), .binary, .plain),
try show(&buf, at(0xF0AC, .bits16), .binary, canonical),
);
}
@ -461,7 +461,7 @@ test "octal: zero-padded to the width, grouped in threes from the right" {
// 16 bits needs 6 octal digits, which groups evenly.
try testing.expectEqualStrings("000 777", try show(&buf, at(0o777, .bits16), .octal, .{}));
try testing.expectEqualStrings("0o000777", try show(&buf, at(0o777, .bits16), .octal, .plain));
try testing.expectEqualStrings("0o000777", try show(&buf, at(0o777, .bits16), .octal, canonical));
// 32 bits needs 11, so the leftmost group is short.
try testing.expectEqualStrings(
@ -470,11 +470,11 @@ test "octal: zero-padded to the width, grouped in threes from the right" {
);
try testing.expectEqualStrings(
"0o00007777777",
try show(&buf, at(0o7777777, .bits32), .octal, .plain),
try show(&buf, at(0o7777777, .bits32), .octal, canonical),
);
try testing.expectEqualStrings("000", try show(&buf, at(0, .bits8), .octal, .{}));
try testing.expectEqualStrings("0o000", try show(&buf, at(0, .bits8), .octal, .plain));
try testing.expectEqualStrings("0o000", try show(&buf, at(0, .bits8), .octal, canonical));
}
test "decimal: the reading is chosen by the notation, not by the value's signedness" {
@ -494,11 +494,11 @@ test "decimal: display groups in thousands, plain does not" {
const large = at(4294967295, .bits32);
try testing.expectEqualStrings("4,294,967,295", try show(&buf, large, .decimal_unsigned, .{}));
try testing.expectEqualStrings("4294967295", try show(&buf, large, .decimal_unsigned, .plain));
try testing.expectEqualStrings("4294967295", try show(&buf, large, .decimal_unsigned, canonical));
const negative = at(@bitCast(@as(i128, -1234567)), .bits32);
try testing.expectEqualStrings("-1,234,567", try show(&buf, negative, .decimal_signed, .{}));
try testing.expectEqualStrings("-1234567", try show(&buf, negative, .decimal_signed, .plain));
try testing.expectEqualStrings("-1234567", try show(&buf, negative, .decimal_signed, canonical));
// Three digits or fewer have nothing to group.
try testing.expectEqualStrings("255", try show(&buf, at(255, .bits8), .decimal_unsigned, .{}));
@ -511,7 +511,7 @@ test "decimal: minInt(i128) formats instead of panicking" {
const min = at(@as(u128, 1) << 127, .bits128);
try testing.expectEqualStrings(
"-170141183460469231731687303715884105728",
try show(&buf, min, .decimal_signed, .plain),
try show(&buf, min, .decimal_signed, canonical),
);
try testing.expectEqualStrings(
"-170,141,183,460,469,231,731,687,303,715,884,105,728",
@ -524,7 +524,7 @@ test "decimal: minInt(i128) formats instead of panicking" {
try show(&buf, max, .decimal_signed, .{}),
);
try testing.expectEqualStrings("0", try show(&buf, at(0, .bits128), .decimal_signed, .plain));
try testing.expectEqualStrings("0", try show(&buf, at(0, .bits128), .decimal_signed, canonical));
}
test "ascii: printable bytes render, everything else is a dot" {
@ -532,27 +532,27 @@ test "ascii: printable bytes render, everything else is a dot" {
// "..asciii" packed into 64 bits: two zero bytes then the letters.
const word = at(0x0000_6173_6369_6969, .bits64);
try testing.expectEqualStrings("..asciii", try show(&buf, word, .ascii, .plain));
try testing.expectEqualStrings("..asciii", try show(&buf, word, .ascii, canonical));
// Two columns per byte so each glyph sits under its hex pair.
try testing.expectEqualStrings(" . . a s c i i i", try show(&buf, word, .ascii, .{}));
try testing.expectEqualStrings("A", try show(&buf, at('A', .bits8), .ascii, .plain));
try testing.expectEqualStrings("A", try show(&buf, at('A', .bits8), .ascii, canonical));
try testing.expectEqualStrings(" A", try show(&buf, at('A', .bits8), .ascii, .{}));
// Control and high bytes are dots.
try testing.expectEqualStrings(".", try show(&buf, at(0x00, .bits8), .ascii, .plain));
try testing.expectEqualStrings(".", try show(&buf, at(0x80, .bits8), .ascii, .plain));
try testing.expectEqualStrings(".", try show(&buf, at(0x00, .bits8), .ascii, canonical));
try testing.expectEqualStrings(".", try show(&buf, at(0x80, .bits8), .ascii, canonical));
// The boundaries of the printable range, and one past it.
try testing.expectEqualStrings(" ", try show(&buf, at(0x20, .bits8), .ascii, .plain));
try testing.expectEqualStrings("~", try show(&buf, at(0x7E, .bits8), .ascii, .plain));
try testing.expectEqualStrings(".", try show(&buf, at(0x7F, .bits8), .ascii, .plain));
try testing.expectEqualStrings(" ", try show(&buf, at(0x20, .bits8), .ascii, canonical));
try testing.expectEqualStrings("~", try show(&buf, at(0x7E, .bits8), .ascii, canonical));
try testing.expectEqualStrings(".", try show(&buf, at(0x7F, .bits8), .ascii, canonical));
}
test "ascii: little-endian reverses the byte order, as hex does" {
var buf: [128]u8 = undefined;
const value = at(0x4142_4344, .bits32);
try testing.expectEqualStrings("ABCD", try show(&buf, value, .ascii, .plain));
try testing.expectEqualStrings("ABCD", try show(&buf, value, .ascii, canonical));
try testing.expectEqualStrings(
"DCBA",
try show(&buf, value, .ascii, .{ .separators = false, .endian = .little }),
@ -587,8 +587,8 @@ test "display: every row has the width's worth of digits, at every width" {
// ASCII: two columns per byte plus a space between bytes.
try testing.expectEqual(bytes * 3 - 1, (try show(&buf, value, .ascii, .{})).len);
// The prefixed forms are the prefix plus one digit per unit of the base.
try testing.expectEqual(2 + bits_count / 4, (try show(&buf, value, .hex, .plain)).len);
try testing.expectEqual(2 + bits_count, (try show(&buf, value, .binary, .plain)).len);
try testing.expectEqual(2 + bits_count / 4, (try show(&buf, value, .hex, canonical)).len);
try testing.expectEqual(2 + bits_count, (try show(&buf, value, .binary, canonical)).len);
}
}
@ -639,28 +639,28 @@ test "the default format is the number, read by its own signedness" {
try testing.expectEqualStrings("5 and -2", pair.buffered());
}
// -- displayWidthFor tests --
// -- BitWidth.displayFor tests --
//
// Moved here with the function, from `formatter.zig`, where it was the one thing
// in the file that formatted nothing: it returns a `BitWidth`, which is this
// file's type.
test "displayWidthFor: the narrowest standard width that holds the value" {
try testing.expectEqual(BitWidth.bits8, displayWidthFor(0));
try testing.expectEqual(BitWidth.bits8, displayWidthFor(255));
try testing.expectEqual(BitWidth.bits16, displayWidthFor(256));
try testing.expectEqual(BitWidth.bits16, displayWidthFor(65535));
try testing.expectEqual(BitWidth.bits32, displayWidthFor(65536));
try testing.expectEqual(BitWidth.bits64, displayWidthFor(0x1_0000_0000));
try testing.expectEqual(BitWidth.bits128, displayWidthFor(0x1_0000_0000_0000_0000));
try testing.expectEqual(BitWidth.bits128, displayWidthFor(std.math.maxInt(u128)));
test "BitWidth.displayFor: the narrowest standard width that holds the value" {
try testing.expectEqual(BitWidth.bits8, BitWidth.displayFor(0));
try testing.expectEqual(BitWidth.bits8, BitWidth.displayFor(255));
try testing.expectEqual(BitWidth.bits16, BitWidth.displayFor(256));
try testing.expectEqual(BitWidth.bits16, BitWidth.displayFor(65535));
try testing.expectEqual(BitWidth.bits32, BitWidth.displayFor(65536));
try testing.expectEqual(BitWidth.bits64, BitWidth.displayFor(0x1_0000_0000));
try testing.expectEqual(BitWidth.bits128, BitWidth.displayFor(0x1_0000_0000_0000_0000));
try testing.expectEqual(BitWidth.bits128, BitWidth.displayFor(std.math.maxInt(u128)));
}
test "displayWidthFor: the chosen width holds the value and sizes the rows" {
test "BitWidth.displayFor: the chosen width holds the value and sizes the rows" {
// What the width is for: the hex row of a small number is one byte, not eight.
var buf: [256]u8 = undefined;
for ([_]u128{ 0, 1, 255, 256, 0xFFFF, 0x1_0000, std.math.maxInt(u64), std.math.maxInt(u128) }) |value| {
const width = displayWidthFor(value);
const width = BitWidth.displayFor(value);
try testing.expectEqual(value, value & width.mask());
const int: Integer = .{ .raw = value, .width = width, .signedness = .unsigned };
const hex = try int.fmt(.hex, .{}).render(&buf);

View file

@ -83,7 +83,7 @@ const Distance = union(enum) {
/// mode used to reduce it modulo 64, so `8 >> -1` quietly became `8 >> 63`.
fn distance(right: Integer) Error!Distance {
if (right.isNegative()) return Error.DomainError;
const value = right.masked();
const value = right.unsignedValue();
if (value >= right.width.bits()) return .past_width;
return .{ .within = @intCast(value) };
}
@ -94,7 +94,7 @@ fn distance(right: Integer) Error!Distance {
/// saturated: rotating a 64-bit value by 65 is rotating it by 1.
fn rotation(right: Integer) Error!u7 {
if (right.isNegative()) return Error.DomainError;
return @intCast(right.masked() % right.width.bits());
return @intCast(right.unsignedValue() % right.width.bits());
}
/// Apply a fixed-width operation to two values of the same integer type.
@ -105,8 +105,8 @@ pub fn apply(op: Op, left_in: Integer, right_in: Integer) Error!Integer {
std.debug.assert(left_in.sameTypeAs(right_in));
const mask = left_in.width.mask();
const left = left_in.masked();
const right = right_in.masked();
const left = left_in.unsignedValue();
const right = right_in.unsignedValue();
const raw: u128 = switch (op) {
.bit_and => left & right,
@ -155,12 +155,12 @@ pub fn apply(op: Op, left_in: Integer, right_in: Integer) Error!Integer {
/// Bitwise complement within the width.
pub fn not(value: Integer) Integer {
return value.withRaw(~value.masked());
return value.withRaw(~value.unsignedValue());
}
/// Two's complement negation within the width.
pub fn negate(value: Integer) Integer {
return value.withRaw(~value.masked() +% 1);
return value.withRaw(~value.unsignedValue() +% 1);
}
// -- Tests --
@ -314,7 +314,7 @@ test "results stay inside the width, at every width and operator" {
const op = @field(Op, field.name);
// 1 is a safe distance for the shifts and a legal operand for the rest.
const result = try apply(op, all_ones, all_ones.withRaw(1));
try testing.expectEqual(result.raw, result.masked());
try testing.expectEqual(result.raw, result.unsignedValue());
try testing.expect(result.sameTypeAs(all_ones));
}
}

View file

@ -116,8 +116,8 @@ fn evalExpr(config: Config, expr: *const Expr) Error!Integer {
/// modes: it wraps at the width here and is exact rational arithmetic there.
fn evalBinaryOp(op: BinaryOp, left_in: Integer, right_in: Integer) Error!Integer {
const mask = left_in.width.mask();
const left = left_in.masked();
const right = right_in.masked();
const left = left_in.unsignedValue();
const right = right_in.unsignedValue();
const raw: u128 = switch (op) {
.add => left +% right,

View file

@ -500,7 +500,7 @@ fn formatStandardMultiBase(buf: []u8, dec_display: []const u8, value: f64) CliRe
const int_val: u128 = @intFromFloat(value);
const int: engine.Integer = .{
.raw = int_val,
.width = engine.Integer.displayWidthFor(int_val),
.width = engine.BitWidth.displayFor(int_val),
.signedness = .unsigned,
};

View file

@ -1074,7 +1074,7 @@ pub const App = struct {
const int_val: u128 = @intFromFloat(as_float);
const int: engine.Integer = .{
.raw = int_val,
.width = engine.Integer.displayWidthFor(int_val),
.width = engine.BitWidth.displayFor(int_val),
.signedness = .unsigned,
};
details = .{