From 19f7b5e2e2760f45592d735090bdad78d5d5eefa Mon Sep 17 00:00:00 2001 From: gingerBill Date: Tue, 29 Sep 2026 14:53:28 +0100 Subject: [PATCH] Add f16/f32 precision checks; Fix macOS `MSIZE` macro collision --- src/big_rat.cpp | 49 +++++++++++++++--------- src/check_expr.cpp | 29 ++++++++++++-- tests/internal/test_number_literals.odin | 37 ++++++++++++++++++ 3 files changed, 93 insertions(+), 22 deletions(-) diff --git a/src/big_rat.cpp b/src/big_rat.cpp index 8f4383a72..0acf64d70 100644 --- a/src/big_rat.cpp +++ b/src/big_rat.cpp @@ -1,11 +1,8 @@ -// An exact rational value: num/den with den > 0, kept in lowest terms. struct BigRat { mp_int num; // signed mp_int den; // > 0 }; -// Guards against a tiny literal requesting an enormous 10^N (e.g. `1.0e999999999`). Far beyond any -// representable float (f64 range is ~1e+-324); a literal past this is rejected as malformed. i64 const BIG_RAT_MAX_DECIMAL_EXP = 65536; // Reduce num/den to lowest terms with den > 0 (0 becomes 0/1). @@ -112,14 +109,18 @@ gb_internal bool big_rat_from_decimal_string(String const &s, mp_int *num, mp_in return true; } -// Convert the exact rational `a/b` (b != 0) to the nearest f64 (round-to-nearest, ties-to-even). +// Convert the exact rational `a/b` (b != 0) to the nearest value of a target IEEE-754 binary float, +// with round-to-nearest, ties-to-even. `mantissa_bits`/`ebias` select the target format: +// f16: 10 / 15, f32: 23 / 127, f64: 52 / 1023. +// The result is returned as an f64 that exactly equals that target value (target subnormals and +// overflow-to-infinity included), so it can be stored in an f64 and re-emitted losslessly. // NOTE(bill): Ported from core:math/big `internal_rat_to_float`. -gb_internal f64 big_rat_to_f64(mp_int const *a_in, mp_int const *b_in) { - int const MSIZE = 52; // explicit mantissa bits - int const MSIZE1 = MSIZE + 1; // 53, incl. the implicit bit - int const MSIZE2 = MSIZE + 2; // 54, one guard bit - int const EBIAS = 1023; - int const EMIN = 1 - EBIAS; // -1022 +gb_internal f64 big_rat_to_float(mp_int const *a_in, mp_int const *b_in, int mantissa_bits, int ebias) { + // NOTE: lowercase locals on purpose: `MSIZE` is a system macro on some platforms (arm/param.h). + int const msize = mantissa_bits; // explicit mantissa bits + int const msize1 = msize + 1; // incl. the implicit bit + int const msize2 = msize + 2; // one guard bit + int const emin = 1 - ebias; int alen = mp_count_bits(a_in); if (alen == 0) { @@ -135,7 +136,7 @@ gb_internal f64 big_rat_to_f64(mp_int const *a_in, mp_int const *b_in) { mp_abs(a_in, &a2); mp_abs(b_in, &b2); - int shift = MSIZE2 - exp; + int shift = msize2 - exp; if (shift > 0) { mp_mul_2d(&a2, shift, &a2); } else if (shift < 0) { @@ -146,26 +147,26 @@ gb_internal f64 big_rat_to_f64(mp_int const *a_in, mp_int const *b_in) { bool has_rem = !mp_iszero(&r); u64 mantissa = mp_get_mag_u64(&q); - if ((mantissa >> MSIZE2) == 1) { + if ((mantissa >> msize2) == 1) { if (mantissa & 1) has_rem = true; mantissa >>= 1; exp += 1; } - // mantissa is now in [2^53, 2^54): 53 significant bits plus one guard bit. + // mantissa is now in [2^msize1, 2^msize2): msize1 significant bits plus one guard bit. - if (EMIN - MSIZE <= exp && exp <= EMIN) { + if (emin - msize <= exp && exp <= emin) { // Denormalise: fold the bits that fall below the subnormal grid into the guard/sticky. - unsigned sh = cast(unsigned)(EMIN - (exp - 1)); + unsigned sh = cast(unsigned)(emin - (exp - 1)); u64 lost = mantissa & ((cast(u64)1 << sh) - 1); has_rem = has_rem || (lost != 0); mantissa >>= sh; - exp = 2 - EBIAS; + exp = 2 - ebias; } if (mantissa & 1) { if (has_rem || (mantissa & 2)) { // round half to even mantissa += 1; - if (mantissa >= (cast(u64)1 << MSIZE2)) { + if (mantissa >= (cast(u64)1 << msize2)) { mantissa >>= 1; exp += 1; } @@ -173,9 +174,21 @@ gb_internal f64 big_rat_to_f64(mp_int const *a_in, mp_int const *b_in) { } mantissa >>= 1; // drop the guard bit - f64 f = ldexp(cast(f64)mantissa, exp - MSIZE1); + f64 f = ldexp(cast(f64)mantissa, exp - msize1); + // Materialise the target format's overflow-to-infinity (exact otherwise: `f` already has the + // target's mantissa width and exponent, so the narrowing cast does not round). + if (msize == 23) { + f = cast(f64)cast(f32)f; + } else if (msize == 10) { + f = cast(f64)f16_to_f32(f32_to_f16(cast(f32)f)); + } if (has_sign) { f = -f; } return f; +} + +// Convert the exact rational `a/b` (b != 0) to the nearest f64 (round-to-nearest, ties-to-even). +gb_internal f64 big_rat_to_f64(mp_int const *a_in, mp_int const *b_in) { + return big_rat_to_float(a_in, b_in, 52, 1023); } \ No newline at end of file diff --git a/src/check_expr.cpp b/src/check_expr.cpp index 45f5c504a..8c75dd8c0 100644 --- a/src/check_expr.cpp +++ b/src/check_expr.cpp @@ -2511,11 +2511,32 @@ gb_internal bool check_representable_as_constant(CheckerContext *c, ExactValue i default: GB_PANIC("Compiler error: Unknown integer type!"); break; } } else if (is_type_float(type)) { - ExactValue v = exact_value_to_float(in_value); - if (v.kind != ExactValue_Float) { - return false; + ExactValue v; + if (in_value.kind == ExactValue_Rational) { + // Round the exact rational directly to the target format (single rounding, ties-to-even), + // rather than rational -> f64 -> f16/f32 which would round twice. + + + int mantissa_bits = 52; + int ebias = 1023; + switch (type->Basic.kind) { + case Basic_f16: case Basic_f16le: case Basic_f16be: + mantissa_bits = 10; + ebias = 15; + break; + case Basic_f32: case Basic_f32le: case Basic_f32be: + mantissa_bits = 23; + ebias = 127; + break; + } + v = exact_value_float(big_rat_to_float(&in_value.value_rational->num, &in_value.value_rational->den, mantissa_bits, ebias)); + } else { + v = exact_value_to_float(in_value); + if (v.kind != ExactValue_Float) { + return false; + } + check_update_float_precision(&v, type); } - check_update_float_precision(&v, type); // An exact finite constant (integer or rational) that overflows the target float's range is not // representable by it; without this it would silently become +/-Inf (e.g. `x: f64 = 1.0e400`). diff --git a/tests/internal/test_number_literals.odin b/tests/internal/test_number_literals.odin index 98f221933..415b4227f 100644 --- a/tests/internal/test_number_literals.odin +++ b/tests/internal/test_number_literals.odin @@ -1,6 +1,7 @@ package test_internal import "core:testing" +import "core:strconv" // Regression tests for numeric literal parsing and constant folding: // * an integer literal with a large exponent (e.g. `98765e309`) is an exact arbitrary-precision @@ -56,3 +57,39 @@ large_integer_literals :: proc(t: ^testing.T) { testing.expect(t, u128(1e38) > u128(1e37), "1e38 > 1e37 as u128") testing.expect_value(t, u128(1e38) / u128(1e19), u128(1e19)) } + +@(test) +float_literal_f16_f32_precision :: proc(t: ^testing.T) { + // f16/f32 constants are rounded once, directly from the exact value, so a constant-folded literal + // matches the correctly-rounded runtime parse (rather than double-rounding via f64). + c32 :: proc(t: ^testing.T, got: f32, lit: string) { + want, _ := strconv.parse_f32(lit) + testing.expectf(t, transmute(u32)got == transmute(u32)want, + "f32 %s: got %08x, want %08x", lit, transmute(u32)got, transmute(u32)want) + } + c32(t, 0.1, "0.1") + c32(t, 0.2, "0.2") + c32(t, 0.3, "0.3") + c32(t, f32(1.0/3.0), "0.3333333333333333") + c32(t, 3.14159265358979323846, "3.14159265358979323846") + c32(t, 1.1, "1.1") + c32(t, 1e-40, "1e-40") // subnormal f32 + c32(t, 1.5e-45, "1.5e-45") + c32(t, 8388609.0, "8388609.0") // 2^23 + 1 + + // f16 spot checks against an f64-parsed reference (53 bits is exact relative to f16's 11). + c16 :: proc(t: ^testing.T, got: f16, lit: string) { + w64, _ := strconv.parse_f64(lit) + want := f16(w64) + testing.expectf(t, transmute(u16)got == transmute(u16)want, + "f16 %s: got %04x, want %04x", lit, transmute(u16)got, transmute(u16)want) + } + c16(t, 0.1, "0.1") + c16(t, f16(1.0/3.0), "0.3333333333333333") + c16(t, 3.14159265358979323846, "3.14159265358979323846") + c16(t, 6e-8, "6e-8") // subnormal f16 + + // f16/f32 overflow of an exact constant is rejected, not silently +Inf (compile-time #assert + // can't test a reject, but these confirm large finite values still fold correctly). + testing.expect_value(t, f32(1e38), strconv.parse_f32("1e38") or_else 0) +}