A u64 converts to and from a float on x86 as it does under LLVM, across the top half of its range

This commit is contained in:
Joseph Ferano 2026-09-25 10:15:36 +07:00
parent 96be1193c0
commit cf04bfc8fd
6 changed files with 152 additions and 10 deletions

View File

@ -1092,9 +1092,12 @@ build included, reaches =emit_globals_init= as =Emit.startup_plan='s
straight into their symbol are =Tast.const_init= ones, which neither transfer straight into their symbol are =Tast.const_init= ones, which neither transfer
nor read anything. Nothing to change. nor read anything. Nothing to change.
** TODO A u64 converted to f64 is signed on x86 ** DONE A u64 converted to f64 is signed on x86
=(f64 (u64 18446744073709551615))= prints =1.84467e+19= under LLVM and =-1= under CLOSED: [2026-09-25]
=--x86=, with a literal or a run-time =u64= alike. =--x86= converts =u64= to and from =f64= and =f32= the way LLVM's =uitofp= and
=fptoui= do: the halve-and-double sequence one way, subtract 2^63 and set the top
bit the other, ties rounding to even. =test/programs/u64-float.flan= runs on both
backends. A cast that misses =u64='s range reports it as =[0 18446744073709551615]=.
** DONE An aggregate built in place never reads its own destination ** DONE An aggregate built in place never reads its own destination
CLOSED: [2026-09-25] CLOSED: [2026-09-25]

View File

@ -1333,6 +1333,7 @@ let option_lay f (t : Types.t) =
let cc_e = 4 and cc_ne = 5 let cc_e = 4 and cc_ne = 5
let cc_b = 2 and cc_ae = 3 and cc_be = 6 and cc_a = 7 let cc_b = 2 and cc_ae = 3 and cc_be = 6 and cc_a = 7
let cc_l = 12 and cc_ge = 13 and cc_le = 14 and cc_g = 15 let cc_l = 12 and cc_ge = 13 and cc_le = 14 and cc_g = 15
let cc_s = 8
let int_cc ~signed (p : Tast.prim) = let int_cc ~signed (p : Tast.prim) =
match p, signed with match p, signed with
@ -3419,9 +3420,29 @@ and cast f (a : Tast.expr) (target : Types.t) dst =
cvtss2sd f.b ~dst:xmm0 ~src:xmm0; cvtss2sd f.b ~dst:xmm0 ~src:xmm0;
fstore f.b ~src:xmm0 ~mm:(lmem f dst ~scratch:r11) ~f64:(f64_of dst_t) fstore f.b ~src:xmm0 ~mm:(lmem f dst ~scratch:r11) ~f64:(f64_of dst_t)
| false, true -> | false, true ->
let f64 = f64_of dst_t in
load_loc f ~reg:rax l src_t; load_loc f ~reg:rax l src_t;
cvtsi2f f.b ~f64:(f64_of dst_t) ~dst:xmm0 ~src:rax; if src_t = Types.Int Types.U64 then begin
fstore f.b ~src:xmm0 ~mm:(lmem f dst ~scratch:r11) ~f64:(f64_of dst_t) (* [cvtsi2sd] reads the register as signed, so a u64 with the top bit
set is halved first and doubled after. The bit shifted out is ORed
back into the lowest one, which keeps the halved value from landing
exactly on a tie it was not on, so the one rounding the conversion
does is the rounding of the whole value. *)
let big = new_label f "u2fbig" and done_ = new_label f "u2fdone" in
test_rr f.b ~a:rax ~c:rax;
jcc_lbl f.b ~cc:cc_s big;
cvtsi2f f.b ~f64 ~dst:xmm0 ~src:rax;
jmp_lbl f.b done_;
lbl f.b big;
mov_rr f.b ~dst:rcx ~src:rax;
shr_imm f.b ~dst:rcx ~n:1;
and_imm f.b ~dst:rax 1;
or_rr f.b ~dst:rcx ~src:rax;
cvtsi2f f.b ~f64 ~dst:xmm0 ~src:rcx;
farith f.b ~op:0x58 ~f64 ~dst:xmm0 ~src:xmm0;
lbl f.b done_
end else cvtsi2f f.b ~f64 ~dst:xmm0 ~src:rax;
fstore f.b ~src:xmm0 ~mm:(lmem f dst ~scratch:r11) ~f64
| true, false -> | true, false ->
fload f.b ~dst:xmm0 ~mm:(lmem f l ~scratch:r11) ~f64:(f64_of src_t); fload f.b ~dst:xmm0 ~mm:(lmem f l ~scratch:r11) ~f64:(f64_of src_t);
(* [cvttsd2si] answers a fixed "integer indefinite" for a value out of (* [cvttsd2si] answers a fixed "integer indefinite" for a value out of
@ -3431,7 +3452,24 @@ and cast f (a : Tast.expr) (target : Types.t) dst =
(match src_t, dst_t with (match src_t, dst_t with
| Types.Float sk, Types.Int k -> check_cast f a.Tast.loc sk k | Types.Float sk, Types.Int k -> check_cast f a.Tast.loc sk k
| _ -> ()); | _ -> ());
cvttf2si f.b ~f64:(f64_of src_t) ~dst:rax ~src:xmm0; let f64 = f64_of src_t in
if dst_t = Types.Int Types.U64 then begin
(* [cvttsd2si] answers only for the signed range, so a value at 2^63 or
above has 2^63 taken off before the conversion and the top bit set
after it. A NaN compares unordered and takes the first branch. *)
let big = new_label f "f2ubig" and done_ = new_label f "f2udone" in
fload f.b ~dst:1 ~mm:(Sym (float_const f (ldexp 1.0 63) ~f64, 0)) ~f64;
ucomis f.b ~f64 ~a:xmm0 ~c:1;
jcc_lbl f.b ~cc:cc_ae big;
cvttf2si f.b ~f64 ~dst:rax ~src:xmm0;
jmp_lbl f.b done_;
lbl f.b big;
farith f.b ~op:0x5c ~f64 ~dst:xmm0 ~src:1;
cvttf2si f.b ~f64 ~dst:rax ~src:xmm0;
imm_into f ~reg:rcx Int64.min_int;
xor_rr f.b ~dst:rax ~src:rcx;
lbl f.b done_
end else cvttf2si f.b ~f64 ~dst:rax ~src:xmm0;
store_loc f ~reg:rax dst dst_t store_loc f ~reg:rax dst dst_t
(* ── Call frame information ──────────────────────────────────────────── *) (* ── Call frame information ──────────────────────────────────────────── *)

View File

@ -1020,7 +1020,15 @@ static void flan_arith_fail(const uint8_t *loc, int64_t loclen, int32_t op,
"to\n", "to\n",
(int)loclen, (const char *)loc); (int)loclen, (const char *)loc);
break; break;
/* An unsigned type's range starts at zero and a signed one's below it, so
* the lower bound says how to read the upper one: u64's is all ones. */
default: default:
if (lhs == 0)
fprintf(stderr,
"%.*s: this value does not fit the integer type it is cast to, "
"which holds [0 %llu]\n",
(int)loclen, (const char *)loc, (unsigned long long)rhs);
else
fprintf(stderr, fprintf(stderr,
"%.*s: this value does not fit the integer type it is cast to, " "%.*s: this value does not fit the integer type it is cast to, "
"which holds [%lld %lld]\n", "which holds [%lld %lld]\n",

View File

@ -82,6 +82,9 @@
(= n 10) (print (i32 (/ (f64 1.0) (f64 0.0)))) (= n 10) (print (i32 (/ (f64 1.0) (f64 0.0))))
(= n 11) (print (u8 (/ (f32 -1.0) (f32 0.0)))) (= n 11) (print (u8 (/ (f32 -1.0) (f32 0.0))))
(= n 12) (print (i32 (/ (f32 0.0) (f32 0.0)))) (= n 12) (print (i32 (/ (f32 0.0) (f32 0.0))))
;; A u64 destination, whose upper bound is all ones and is printed as
;; the unsigned number it is.
(= n 13) (print (u64 (* huge 1e-280)))
:else (println "?")) :else (println "?"))
0)) 0))

View File

@ -0,0 +1,65 @@
;;;; Conversions between the unsigned integers and the floats, across the top
;;;; half of u64's range, where the value does not fit a signed register. Both
;;;; backends print the same lines. Each case is written twice: once through a
;;;; global, so the conversion happens at run time, and once on a literal.
;;;;
;;;; A float is printed by converting it back to an integer, because the
;;;; printed float has six digits and every case here differs past the sixth.
(defonce top u64 18446744073709551615) ; 2^64 - 1
(defonce half1 u64 9223372036854775809) ; 2^63 + 1
(defonce tie0 u64 9223372036854776832) ; 2^63 + 1024, a tie that rounds down to even
(defonce tie1 u64 9223372036854778880) ; 2^63 + 3072, a tie that rounds up to even
(defonce above u64 9223372036854776833) ; 2^63 + 1025, just past a tie
(defonce small u64 12345)
(defonce ftie0 u64 9223372586610589696) ; 2^63 + 2^39, an f32 tie that rounds down
(defonce fup u64 9223372586610589697) ; 2^63 + 2^39 + 1
(defonce ftie1 u64 9223373686122217472) ; 2^63 + 3*2^39, an f32 tie that rounds up
(defonce u32top u32 4294967295)
(defonce d63 f64 9223372036854775808.0) ; 2^63
(defonce dmax f64 18446744073709549568.0) ; the largest f64 below 2^64
(defonce dbelow f64 9223372036854774784.0) ; the largest f64 below 2^63
(defonce d15 f64 1.5e19)
(defonce dsmall f64 7.9)
(defonce s63 f32 9223372036854775808.0)
(defonce smax f32 18446742974197923840.0) ; the largest f32 below 2^64
(defonce d32 f64 4294967295.0)
(defonce d3e9 f64 3e9)
(defn main [] i32
;; u64 to f64.
(println (f64 top))
(println (u64 (* (f64 top) 0.5)))
(println (u64 (f64 half1)))
(println (u64 (f64 tie0)))
(println (u64 (f64 tie1)))
(println (u64 (f64 above)))
(println (u64 (f64 small)))
(println (f64 (u64 18446744073709551615)))
(println (u64 (f64 (u64 9223372036854778880))))
;; u64 to f32.
(println (u64 (* (f32 top) (f32 0.5))))
(println (u64 (f32 ftie0)))
(println (u64 (f32 fup)))
(println (u64 (f32 ftie1)))
(println (u64 (f32 small)))
(println (u64 (f32 (u64 9223372586610589697))))
;; u32 to both.
(println (u64 (f64 u32top)))
(println (u64 (f32 u32top)))
;; f64 and f32 to u64.
(println (u64 d63))
(println (u64 dmax))
(println (u64 dbelow))
(println (u64 d15))
(println (u64 dsmall))
(println (u64 s63))
(println (u64 smax))
(println (u64 18446744073709549568.0))
(println (u64 (f32 1.5e19)))
;; f64 to u32.
(println (u32 d32))
(println (u32 d3e9))
(println (u32 4294967295.0))
0)

View File

@ -2639,6 +2639,8 @@ let () =
"this value is infinite, which has no integer value to cast to"; "this value is infinite, which has no integer value to cast to";
traps "an f32 NaN cast to an integer" "12" traps "an f32 NaN cast to an integer" "12"
"this value is NaN, which has no integer value to cast to"; "this value is NaN, which has no integer value to cast to";
traps "a float too large for a u64" "13"
"which holds [0 18446744073709551615]";
(try Sys.remove exe with Sys_error _ -> ()) (try Sys.remove exe with Sys_error _ -> ())
in in
arith (); arith ();
@ -2772,6 +2774,29 @@ let () =
outputs ~x86:true outputs ~x86:true
"an aggregate that reads its destination sees the old value, --x86" "an aggregate that reads its destination sees the old value, --x86"
"programs/self-read.flan" self_read_out; "programs/self-read.flan" self_read_out;
(* Conversions between u64 and the floats across the top half of u64's
range, where [cvtsi2sd] and [cvttsd2si] read the register as signed:
x86 answered -1 for (f64 (u64 18446744073709551615)). The expected
lines are what LLVM's [uitofp] and [fptoui] give, ties included. *)
let u64_float_out =
"1.84467e+19\n9223372036854775808\n9223372036854775808\n\
9223372036854775808\n9223372036854779904\n9223372036854777856\n\
12345\n1.84467e+19\n9223372036854779904\n9223372036854775808\n\
9223372036854775808\n9223373136366403584\n9223374235878031360\n\
12345\n9223373136366403584\n4294967295\n4294967296\n\
9223372036854775808\n18446744073709549568\n9223372036854774784\n\
15000000000000000000\n7\n9223372036854775808\n\
18446742974197923840\n18446744073709549568\n15000000520515485696\n\
4294967295\n3000000000\n4294967295\n"
in
outputs "u64 and the floats convert across the whole range"
"programs/u64-float.flan" u64_float_out;
outputs ~opt:"-O0"
"u64 and the floats convert across the whole range, -O0"
"programs/u64-float.flan" u64_float_out;
outputs ~x86:true
"u64 and the floats convert across the whole range, --x86"
"programs/u64-float.flan" u64_float_out;
(* ── Packages: the link follows the program ──────────────────────── (* ── Packages: the link follows the program ────────────────────────
A package's C and linker arguments used to come with the import, A package's C and linker arguments used to come with the import,