human review: Rational.zig
This commit is contained in:
parent
880c2e657f
commit
7853d8aad4
2 changed files with 174 additions and 57 deletions
|
|
@ -30,11 +30,27 @@ pub const Error = error{
|
||||||
InvalidNumber,
|
InvalidNumber,
|
||||||
/// The exponent of an integer power did not fit the supported range.
|
/// The exponent of an integer power did not fit the supported range.
|
||||||
ExponentTooLarge,
|
ExponentTooLarge,
|
||||||
|
/// More digits than `parseDecimal` will hold. Distinct from `InvalidNumber`
|
||||||
|
/// because the text is a perfectly good number; it is this parser that declines.
|
||||||
|
TooManyDigits,
|
||||||
/// Square root of a negative value. Raised by the numeric model rather than
|
/// Square root of a negative value. Raised by the numeric model rather than
|
||||||
/// checked by each caller, so the domain rule lives in one place.
|
/// checked by each caller, so the domain rule lives in one place.
|
||||||
NegativeRoot,
|
NegativeRoot,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// Mantissa digits `parseDecimal` will hold. Past this the value is rejected rather
|
||||||
|
/// than truncated: a silently shortened literal is a wrong answer.
|
||||||
|
const max_parse_digits = 512;
|
||||||
|
|
||||||
|
/// Largest decimal exponent `parseDecimal` accepts. `1e100000` already needs a
|
||||||
|
/// 42KB numerator, so this is generous rather than restrictive.
|
||||||
|
const max_parse_exponent = 100_000;
|
||||||
|
|
||||||
|
/// Digit separators, accepted anywhere in numeric text.
|
||||||
|
fn isSeparator(c: u8) bool {
|
||||||
|
return c == '_' or c == ',' or c == ' ';
|
||||||
|
}
|
||||||
|
|
||||||
/// Numerator. Carries the sign of the value.
|
/// Numerator. Carries the sign of the value.
|
||||||
num: Managed,
|
num: Managed,
|
||||||
/// Denominator. Always strictly positive.
|
/// Denominator. Always strictly positive.
|
||||||
|
|
@ -160,76 +176,89 @@ pub fn parse(allocator: Allocator, text: []const u8) Error!Rational {
|
||||||
/// Parse a decimal numeric literal exactly.
|
/// Parse a decimal numeric literal exactly.
|
||||||
///
|
///
|
||||||
/// Accepts an optional sign, digits with an optional fractional part, and an
|
/// Accepts an optional sign, digits with an optional fractional part, and an
|
||||||
/// optional decimal exponent: `42`, `-1.25`, `1e5`, `1.5e-3`. Underscores,
|
/// optional decimal exponent: `42`, `-1.25`, `1e5`, `1.5e-3`. Underscores, commas
|
||||||
/// commas and spaces are ignored as digit separators, matching the
|
/// and spaces are ignored anywhere, which is looser than the tokenizer's grouping
|
||||||
/// tokenizer's accepted forms.
|
/// rules (FR-1.8): this sees text the tokenizer has already accepted, plus the
|
||||||
|
/// hand-written factor strings in `units.zig`.
|
||||||
///
|
///
|
||||||
/// The result is exact: `0.1` becomes `1/10`, not the nearest binary float.
|
/// The result is exact: `0.1` becomes `1/10`, not the nearest binary float. Going
|
||||||
|
/// through `std.fmt.parseFloat` would defeat the purpose of this tier, since the
|
||||||
|
/// binary approximation is what the exact tier exists to avoid.
|
||||||
|
///
|
||||||
|
/// One pass over the text. It used to make two copies, one to strip separators and
|
||||||
|
/// another to strip the decimal point, in two 512-byte buffers.
|
||||||
pub fn parseDecimal(allocator: Allocator, text: []const u8) Error!Rational {
|
pub fn parseDecimal(allocator: Allocator, text: []const u8) Error!Rational {
|
||||||
var buf: [512]u8 = undefined;
|
// The mantissa's digits, with separators and the point removed.
|
||||||
var len: usize = 0;
|
var digits: [max_parse_digits]u8 = undefined;
|
||||||
for (text) |c| {
|
var digit_count: usize = 0;
|
||||||
if (c == '_' or c == ',' or c == ' ') continue;
|
|
||||||
if (len >= buf.len) return Error.InvalidNumber;
|
|
||||||
buf[len] = c;
|
|
||||||
len += 1;
|
|
||||||
}
|
|
||||||
var s = buf[0..len];
|
|
||||||
if (s.len == 0) return Error.InvalidNumber;
|
|
||||||
|
|
||||||
var negative = false;
|
|
||||||
if (s[0] == '+' or s[0] == '-') {
|
|
||||||
negative = s[0] == '-';
|
|
||||||
s = s[1..];
|
|
||||||
if (s.len == 0) return Error.InvalidNumber;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Split off the exponent.
|
|
||||||
var exponent: i64 = 0;
|
|
||||||
var mantissa = s;
|
|
||||||
if (std.mem.indexOfAny(u8, s, "eE")) |e_idx| {
|
|
||||||
mantissa = s[0..e_idx];
|
|
||||||
const exp_text = s[e_idx + 1 ..];
|
|
||||||
if (exp_text.len == 0) return Error.InvalidNumber;
|
|
||||||
exponent = std.fmt.parseInt(i64, exp_text, 10) catch return Error.InvalidNumber;
|
|
||||||
if (exponent > 100_000 or exponent < -100_000) return Error.ExponentTooLarge;
|
|
||||||
}
|
|
||||||
if (mantissa.len == 0) return Error.InvalidNumber;
|
|
||||||
|
|
||||||
// Split the mantissa on the decimal point.
|
|
||||||
var digits_buf: [512]u8 = undefined;
|
|
||||||
var digits_len: usize = 0;
|
|
||||||
var frac_digits: usize = 0;
|
var frac_digits: usize = 0;
|
||||||
var seen_dot = false;
|
var seen_dot = false;
|
||||||
for (mantissa) |c| {
|
var negative = false;
|
||||||
|
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < text.len and isSeparator(text[i])) i += 1;
|
||||||
|
if (i < text.len and (text[i] == '+' or text[i] == '-')) {
|
||||||
|
negative = text[i] == '-';
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Mantissa, stopping at an exponent marker.
|
||||||
|
while (i < text.len) : (i += 1) {
|
||||||
|
const c = text[i];
|
||||||
|
if (isSeparator(c)) continue;
|
||||||
|
if (c == 'e' or c == 'E') break;
|
||||||
if (c == '.') {
|
if (c == '.') {
|
||||||
if (seen_dot) return Error.InvalidNumber;
|
if (seen_dot) return Error.InvalidNumber;
|
||||||
seen_dot = true;
|
seen_dot = true;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (c < '0' or c > '9') return Error.InvalidNumber;
|
if (!std.ascii.isDigit(c)) return Error.InvalidNumber;
|
||||||
if (digits_len >= digits_buf.len) return Error.InvalidNumber;
|
if (digit_count == digits.len) return Error.TooManyDigits;
|
||||||
digits_buf[digits_len] = c;
|
digits[digit_count] = c;
|
||||||
digits_len += 1;
|
digit_count += 1;
|
||||||
if (seen_dot) frac_digits += 1;
|
if (seen_dot) frac_digits += 1;
|
||||||
}
|
}
|
||||||
if (digits_len == 0) return Error.InvalidNumber;
|
if (digit_count == 0) return Error.InvalidNumber;
|
||||||
|
|
||||||
|
// Exponent, accumulated directly rather than through a second buffer. The
|
||||||
|
// bound is checked as it grows, so the accumulator cannot overflow.
|
||||||
|
var exponent: i64 = 0;
|
||||||
|
if (i < text.len) {
|
||||||
|
i += 1; // the 'e' or 'E'
|
||||||
|
var exponent_negative = false;
|
||||||
|
while (i < text.len and isSeparator(text[i])) i += 1;
|
||||||
|
if (i < text.len and (text[i] == '+' or text[i] == '-')) {
|
||||||
|
exponent_negative = text[i] == '-';
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
var exponent_digits: usize = 0;
|
||||||
|
while (i < text.len) : (i += 1) {
|
||||||
|
const c = text[i];
|
||||||
|
if (isSeparator(c)) continue;
|
||||||
|
if (!std.ascii.isDigit(c)) return Error.InvalidNumber;
|
||||||
|
exponent = exponent * 10 + (c - '0');
|
||||||
|
if (exponent > max_parse_exponent) return Error.ExponentTooLarge;
|
||||||
|
exponent_digits += 1;
|
||||||
|
}
|
||||||
|
if (exponent_digits == 0) return Error.InvalidNumber;
|
||||||
|
if (exponent_negative) exponent = -exponent;
|
||||||
|
}
|
||||||
|
|
||||||
var num = try Managed.init(allocator);
|
var num = try Managed.init(allocator);
|
||||||
errdefer num.deinit();
|
errdefer num.deinit();
|
||||||
num.setString(10, digits_buf[0..digits_len]) catch return Error.InvalidNumber;
|
num.setString(10, digits[0..digit_count]) catch return Error.InvalidNumber;
|
||||||
|
|
||||||
var den = try Managed.initSet(allocator, 1);
|
var den = try Managed.initSet(allocator, 1);
|
||||||
errdefer den.deinit();
|
errdefer den.deinit();
|
||||||
|
|
||||||
// value = digits / 10^frac_digits * 10^exponent
|
// value = digits / 10^frac_digits * 10^exponent, so one scale factor does both.
|
||||||
|
// The exponent is bounded above and `frac_digits` cannot exceed the digit
|
||||||
|
// buffer, so the magnitude fits a u32 with four orders of magnitude to spare.
|
||||||
const net_exp: i64 = exponent - @as(i64, @intCast(frac_digits));
|
const net_exp: i64 = exponent - @as(i64, @intCast(frac_digits));
|
||||||
if (net_exp != 0) {
|
if (net_exp != 0) {
|
||||||
const magnitude: u64 = @intCast(@abs(net_exp));
|
|
||||||
if (magnitude > std.math.maxInt(u32)) return Error.ExponentTooLarge;
|
|
||||||
var scale = try Managed.initSet(allocator, 10);
|
var scale = try Managed.initSet(allocator, 10);
|
||||||
defer scale.deinit();
|
defer scale.deinit();
|
||||||
try scale.pow(&scale, @intCast(magnitude));
|
try scale.pow(&scale, @intCast(@abs(net_exp)));
|
||||||
if (net_exp > 0) {
|
if (net_exp > 0) {
|
||||||
try num.mul(&num, &scale);
|
try num.mul(&num, &scale);
|
||||||
} else {
|
} else {
|
||||||
|
|
@ -371,18 +400,16 @@ pub fn div(allocator: Allocator, a: Rational, b: Rational) Error!Rational {
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn negate(allocator: Allocator, a: Rational) Error!Rational {
|
pub fn negate(allocator: Allocator, a: Rational) Error!Rational {
|
||||||
var result = try a.clone();
|
var result = try a.cloneWith(allocator);
|
||||||
errdefer result.deinit();
|
errdefer result.deinit();
|
||||||
result.num.negate();
|
result.num.negate();
|
||||||
_ = allocator;
|
|
||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn abs(allocator: Allocator, a: Rational) Error!Rational {
|
pub fn abs(allocator: Allocator, a: Rational) Error!Rational {
|
||||||
var result = try a.clone();
|
var result = try a.cloneWith(allocator);
|
||||||
errdefer result.deinit();
|
errdefer result.deinit();
|
||||||
result.num.abs();
|
result.num.abs();
|
||||||
_ = allocator;
|
|
||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -706,10 +733,8 @@ pub fn toScientificString(
|
||||||
// It can be off by one either way, which the loop below corrects.
|
// It can be off by one either way, which the loop below corrects.
|
||||||
const num_bits: i64 = @intCast(magnitude.bitCountAbs());
|
const num_bits: i64 = @intCast(magnitude.bitCountAbs());
|
||||||
const den_bits: i64 = @intCast(self.den.bitCountAbs());
|
const den_bits: i64 = @intCast(self.den.bitCountAbs());
|
||||||
const log10_of_2 = 0.30102999566398120;
|
const bit_difference: f64 = @floatFromInt(num_bits - den_bits);
|
||||||
var exponent: i64 = @intFromFloat(
|
var exponent: i64 = @intFromFloat(@floor(bit_difference * @log10(2.0)));
|
||||||
@floor(@as(f64, @floatFromInt(num_bits - den_bits)) * log10_of_2),
|
|
||||||
);
|
|
||||||
|
|
||||||
const wanted: i64 = @intCast(significant_digits);
|
const wanted: i64 = @intCast(significant_digits);
|
||||||
// The exponent has to be settled on the TRUNCATED scaling, not the rounded
|
// The exponent has to be settled on the TRUNCATED scaling, not the rounded
|
||||||
|
|
@ -982,6 +1007,72 @@ test "parseDecimal: rejects malformed input" {
|
||||||
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "abc"));
|
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "abc"));
|
||||||
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "-"));
|
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "-"));
|
||||||
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "1e"));
|
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "1e"));
|
||||||
|
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "1e+"));
|
||||||
|
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "1e2x"));
|
||||||
|
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "."));
|
||||||
|
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "+"));
|
||||||
|
// Separators alone are not a number.
|
||||||
|
try testing.expectError(Error.InvalidNumber, Rational.parseDecimal(alloc, "_,_"));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "parseDecimal: a valid number with too many digits says so" {
|
||||||
|
// Distinct from InvalidNumber: the text is a number, this parser just will not
|
||||||
|
// hold it. Reporting "invalid number" for 600 good digits was a wording bug.
|
||||||
|
var long: [max_parse_digits + 1]u8 = undefined;
|
||||||
|
@memset(&long, '9');
|
||||||
|
try testing.expectError(Error.TooManyDigits, Rational.parseDecimal(alloc, &long));
|
||||||
|
|
||||||
|
// One digit fewer is fine, and exact.
|
||||||
|
var at_limit: [max_parse_digits]u8 = undefined;
|
||||||
|
@memset(&at_limit, '9');
|
||||||
|
var value = try Rational.parseDecimal(alloc, &at_limit);
|
||||||
|
defer value.deinit();
|
||||||
|
try testing.expect(value.isInteger());
|
||||||
|
|
||||||
|
// Separators do not count against the limit, since they are not digits.
|
||||||
|
var with_separators: [max_parse_digits * 2]u8 = undefined;
|
||||||
|
for (0..max_parse_digits) |k| {
|
||||||
|
with_separators[k * 2] = '9';
|
||||||
|
with_separators[k * 2 + 1] = '_';
|
||||||
|
}
|
||||||
|
var separated = try Rational.parseDecimal(alloc, &with_separators);
|
||||||
|
defer separated.deinit();
|
||||||
|
try testing.expect(separated.isInteger());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "parseDecimal: the exponent is bounded, and separators inside it are ignored" {
|
||||||
|
var ten_billion = try Rational.parseDecimal(alloc, "1e1_0");
|
||||||
|
defer ten_billion.deinit();
|
||||||
|
try expectFrac("10000000000", ten_billion);
|
||||||
|
|
||||||
|
var spaced = try Rational.parseDecimal(alloc, "1e 5");
|
||||||
|
defer spaced.deinit();
|
||||||
|
try expectFrac("100000", spaced);
|
||||||
|
|
||||||
|
// Leading zeros in the exponent do not overflow the accumulator.
|
||||||
|
var padded = try Rational.parseDecimal(alloc, "1e0000000000000000005");
|
||||||
|
defer padded.deinit();
|
||||||
|
try expectFrac("100000", padded);
|
||||||
|
|
||||||
|
try testing.expectError(
|
||||||
|
Error.ExponentTooLarge,
|
||||||
|
Rational.parseDecimal(alloc, "1e100001"),
|
||||||
|
);
|
||||||
|
try testing.expectError(
|
||||||
|
Error.ExponentTooLarge,
|
||||||
|
Rational.parseDecimal(alloc, "1e-100001"),
|
||||||
|
);
|
||||||
|
// Absurd exponents are caught while accumulating, not by overflowing an i64.
|
||||||
|
try testing.expectError(
|
||||||
|
Error.ExponentTooLarge,
|
||||||
|
Rational.parseDecimal(alloc, "1e99999999999999999999999999"),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "parseDecimal: a sign after separators still applies" {
|
||||||
|
var negative = try Rational.parseDecimal(alloc, " -1.5");
|
||||||
|
defer negative.deinit();
|
||||||
|
try expectFrac("-3/2", negative);
|
||||||
}
|
}
|
||||||
|
|
||||||
test "add: the classic binary float failure is exact here" {
|
test "add: the classic binary float failure is exact here" {
|
||||||
|
|
@ -1105,6 +1196,29 @@ test "negate and abs" {
|
||||||
try expectFrac("3/4", b);
|
try expectFrac("3/4", b);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
test "negate and abs allocate from the allocator they are given" {
|
||||||
|
// They used to call `clone()`, which copies into the operand's allocator, and
|
||||||
|
// discarded the parameter with `_ = allocator`. The leak detector is the
|
||||||
|
// assertion here: nothing deinits the results, so the arena has to own them. If
|
||||||
|
// they came from `alloc` instead, this fails as a leak.
|
||||||
|
var arena_state = std.heap.ArenaAllocator.init(alloc);
|
||||||
|
defer arena_state.deinit();
|
||||||
|
const arena = arena_state.allocator();
|
||||||
|
|
||||||
|
var value = try Rational.initRatio(alloc, 3, 4);
|
||||||
|
defer value.deinit();
|
||||||
|
|
||||||
|
const negated = try Rational.negate(arena, value);
|
||||||
|
try expectFrac("-3/4", negated);
|
||||||
|
|
||||||
|
const magnitude = try Rational.abs(arena, negated);
|
||||||
|
try expectFrac("3/4", magnitude);
|
||||||
|
|
||||||
|
// The operand is untouched: these return a fresh value rather than mutating one,
|
||||||
|
// which is what makes them safe to call on a stored variable or on `Ans`.
|
||||||
|
try expectFrac("3/4", value);
|
||||||
|
}
|
||||||
|
|
||||||
test "powInt: positive exponent" {
|
test "powInt: positive exponent" {
|
||||||
var a = try Rational.initInt(alloc, 2);
|
var a = try Rational.initInt(alloc, 2);
|
||||||
defer a.deinit();
|
defer a.deinit();
|
||||||
|
|
|
||||||
|
|
@ -86,11 +86,14 @@ pub fn phrase(err: Error) []const u8 {
|
||||||
error.DomainError => "domain error",
|
error.DomainError => "domain error",
|
||||||
error.Overflow => "overflow",
|
error.Overflow => "overflow",
|
||||||
error.InvalidOperandType => "invalid operand type",
|
error.InvalidOperandType => "invalid operand type",
|
||||||
// These two used to be flattened into Overflow and DomainError by a mapping
|
// These used to be flattened into Overflow and DomainError by a mapping
|
||||||
// between error sets. The specific wording is the whole reason the numeric
|
// between error sets. The specific wording is the whole reason the numeric
|
||||||
// tier bothered to distinguish them.
|
// tier bothered to distinguish them.
|
||||||
error.ExponentTooLarge => "the exponent is too large to compute",
|
error.ExponentTooLarge => "the exponent is too large to compute",
|
||||||
error.NegativeRoot => "square root of a negative number",
|
error.NegativeRoot => "square root of a negative number",
|
||||||
|
// A valid number with more digits than the exact parser will hold, which is
|
||||||
|
// not the same as a malformed one.
|
||||||
|
error.TooManyDigits => "too many digits",
|
||||||
|
|
||||||
// Units
|
// Units
|
||||||
error.UnknownUnit => "unknown unit",
|
error.UnknownUnit => "unknown unit",
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue