diff --git a/calc-me.flan b/calc-me.flan index 058dbec..e0e17a0 100644 --- a/calc-me.flan +++ b/calc-me.flan @@ -34,8 +34,10 @@ (while (= (peek c) \space) (advance c))) -(defn digit? [b u8] bool - (and (>= b \0) (<= b \9))) +;; digit? was written here until the prelude grew one. There is a single +;; top-level namespace, so a second definition is now an error rather than a +;; shadow — which is the rule working: two byte-identical digit? functions +;; that later drift apart is exactly what it exists to prevent. ;; ── number := digit+ ("." digit+)? ──────────────────────────────────── (defn parse-number [c (Ptr Cursor)] (Option f64) diff --git a/lib/prelude.ml b/lib/prelude.ml index fcdea6e..da923f8 100644 --- a/lib/prelude.ml +++ b/lib/prelude.ml @@ -194,9 +194,13 @@ let source = {flan| ;; ── Numbers ─────────────────────────────────────────────────────────── ;; -;; Only the two that encode a decision. clamp is (min hi (max lo x)) over two +;; Only the ones that encode a decision. clamp is (min hi (max lo x)) over two ;; builtins and abs is (max x (- 0 x)); a wrapper over those is a function -;; emitted into every program to save a caller nothing. +;; emitted into every program to save a caller nothing. The one honest caveat +;; on that abs: at the least representable integer it answers itself, because +;; the negation wraps. That is what every two's-complement abs does, a +;; function here would do it too, and the only fix is not to hand it that +;; value — so it is written down rather than wrapped. ;; Zero for zero, and zero for NaN — neither is positive nor negative, so ;; neither comparison fires. A caller that needs to know which it got should @@ -233,6 +237,106 @@ let source = {flan| ;; [lo, hi), because rand-f32 never reaches 1.0. (defn rand-f32-range [lo f32 hi f32] f32 (+ lo (* (rand-f32) (- hi lo)))) +;; ── Byte classes ────────────────────────────────────────────────────── +;; +;; ASCII only, and deliberately: a byte is a byte here, there is no code point +;; type, and a UTF-8 continuation byte is not a digit under any locale. Both +;; exist because something below needs them — parse-f64 the first, trim the +;; second — and both are what a caller writing a tokenizer reaches for anyway. + +(defn digit? [b u8] bool + (and (>= b \0) (<= b \9))) + +(defn space? [b u8] bool + (or (= b \space) (= b \tab) (= b \newline) (= b \return))) + +;; ── More of the bytes family ────────────────────────────────────────── + +;; Substring search, first occurrence. The length test is first and returns +;; before the loop, so a needle longer than the haystack answers None rather +;; than building a slice that runs off the end. An empty needle is Some 0, +;; which is the answer that makes (index-of-bytes s p) agree with +;; (starts-with? s p) on every p. +;; +;; Naive, O(n·m), and that is the deliberate choice: Boyer–Moore wants a skip +;; table, which is an array sized by the needle, which is an allocation. +(defn index-of-bytes [s [u8] p [u8]] (Option i32) + (when (> (len p) (len s)) + (return None)) + (let [last (- (len s) (len p)) + i 0] + (while (<= i last) + (when (bytes=? (slice s i (+ i (len p))) p) + (return (Some i))) + (set i (+ i 1)))) + None) + +;; Returns a slice *of the input*, which is the whole reason trim can exist +;; without an allocator: there is no new storage, only a narrower view of the +;; caller's. It follows that the result dies with its owner, and that trimming +;; does not modify anything. +;; +;; The two loops both test (< lo hi), so an all-whitespace input walks lo up +;; to hi and stops there, and the result is the empty slice. Without that test +;; lo would pass hi and (slice s lo hi) would be a reversed range, which traps. +(defn trim [s [u8]] [u8] + (let [lo 0 + hi (len s)] + (while (and (< lo hi) (space? (at s lo))) + (set lo (+ lo 1))) + (while (and (< lo hi) (space? (at s (- hi 1)))) + (set hi (- hi 1))) + (slice s lo hi))) + +;; The grammar is Flan's and the rounding is libc's, which is a split and not +;; a dodge. parse-i64 is entirely Flan because strtoll's *answers* are wrong +;; for a caller — 0 for "", 0 for "abc", 12 for "12x" — and reproducing +;; correct-to-the-last-bit decimal-to-binary conversion is a different problem +;; from rejecting junk. So this validates the whole slice first, and only a +;; slice that is entirely a number is handed to bytes->f64; every string this +;; returns Some for is one strtod converts exactly, correctly rounded, and +;; identically everywhere, because that much IEEE-754 requires. +;; +;; The locale worry that keeps parse-i64 in Flan does apply to strtod's +;; decimal point — and is moot here because nothing in the runtime calls +;; setlocale, so the program stays in the C locale for its whole life. If that +;; ever stops being true this function is the thing that breaks. +;; +;; Accepts [+-]? digits [. digits] [eE [+-] digits], needing at least one +;; mantissa digit; refuses "", ".", "1e", "nan", "0x10", " 1" and "1 ". The +;; 511 cap is flan_bytes_to_f64's buffer: past it the shim truncates, and a +;; validator that said yes to 600 digits would be approving a different +;; number than the one strtod reads. +(defn parse-f64 [s [u8]] (Option f64) + (let [i 0 + digits 0] + (when (or (= (len s) 0) (> (len s) 511)) + (return None)) + (when (or (= (at s 0) \-) (= (at s 0) \+)) + (set i 1)) + (while (and (< i (len s)) (digit? (at s i))) + (set i (+ i 1)) + (set digits (+ digits 1))) + (when (and (< i (len s)) (= (at s i) \.)) + (set i (+ i 1)) + (while (and (< i (len s)) (digit? (at s i))) + (set i (+ i 1)) + (set digits (+ digits 1)))) + (when (= digits 0) + (return None)) ; "." and "+" and "e5" are not numbers + (when (and (< i (len s)) (or (= (at s i) \e) (= (at s i) \E))) + (set i (+ i 1)) + (when (and (< i (len s)) (or (= (at s i) \-) (= (at s i) \+))) + (set i (+ i 1))) + (let [e 0] + (while (and (< i (len s)) (digit? (at s i))) + (set i (+ i 1)) + (set e (+ e 1))) + (when (= e 0) + (return None)))) ; a lone exponent marker + ;; Trailing junk is the case strtod is silent about, so the position has + ;; to land exactly on the end. + (if (= i (len s)) (Some (bytes->f64 s)) None))) |flan} let file = "" diff --git a/test/programs/bytes2.flan b/test/programs/bytes2.flan new file mode 100644 index 0000000..810e585 --- /dev/null +++ b/test/programs/bytes2.flan @@ -0,0 +1,95 @@ +;;;; index-of-bytes, trim, the two byte classes, and parse-f64. +;;;; +;;;; The search cases are the ones a naive loop gets wrong rather than the +;;;; ones it gets right: a needle that matches only at the very end, one that +;;;; matches only at index 0, one whose first byte occurs repeatedly before +;;;; the real match ("aab" in "aaab"), a needle longer than the haystack +;;;; (which must answer None and must not trap building the window), the +;;;; empty needle, and a near-miss that shares every byte but the last. +;;;; +;;;; The parse-f64 cases are every shape strtod answers a plausible number +;;;; for and a caller cannot tell from a real one: "", "abc", "1x", ".", +;;;; "1e", " 1", "0x10" and "nan". Each must be None. + +(defn show-idx [o (Option i32)] + (print-i64 (i64 (match o (Some i) i None -1))) + (print-str " ")) + +(defn show-bool [b bool] + (print-str (if b "t" "f"))) + +;; Brackets around the result so an empty trim is visible as [] rather than +;; as nothing at all — the all-whitespace case is otherwise indistinguishable +;; from a trim that printed the wrong slice of length zero. +(defn show-trim [s string] + (print-str "[") + (print-bytes (trim (bytes s))) + (print-str "]")) + +(defn main [] i32 + (show-idx (index-of-bytes (bytes "hello world") (bytes "world"))) ; 6, at the end + (show-idx (index-of-bytes (bytes "hello world") (bytes "hello"))) ; 0, at the start + (show-idx (index-of-bytes (bytes "hello world") (bytes "o w"))) ; 4, in the middle + (show-idx (index-of-bytes (bytes "banana") (bytes "na"))) ; 2, first of two + (show-idx (index-of-bytes (bytes "aaab") (bytes "aab"))) ; 1, after false starts + (newline) + (show-idx (index-of-bytes (bytes "hello") (bytes "hellp"))) ; -1, last byte differs + (show-idx (index-of-bytes (bytes "hi") (bytes "hiya"))) ; -1, longer, no trap + (show-idx (index-of-bytes (bytes "") (bytes "a"))) ; -1, empty haystack + (show-idx (index-of-bytes (bytes "hello") (bytes ""))) ; 0, empty needle + (show-idx (index-of-bytes (bytes "") (bytes ""))) ; 0, both empty + (show-idx (index-of-bytes (bytes "hello") (bytes "hello"))) ; 0, whole string + (newline) + + (show-trim " hi ") ; [hi] + (show-trim "hi") ; [hi] nothing to remove + (show-trim "\thi\n") ; [hi] tab and newline count + (show-trim " ") ; [] all whitespace, must not run backwards + (show-trim "") ; [] + (show-trim " a b ") ; [a b] the inner space survives + (show-trim " x") ; [x] one-sided + (show-trim "x ") ; [x] + (newline) + + (show-bool (digit? \0)) (show-bool (digit? \9)) (show-bool (digit? \/)) + (show-bool (digit? \:)) (show-bool (digit? \a)) + (newline) + (show-bool (space? \space)) (show-bool (space? \tab)) + (show-bool (space? \newline)) (show-bool (space? \return)) + (show-bool (space? \a)) (show-bool (space? \0)) + (newline) + + ;; Accepted. The last is the round trip through %g that proves the value and + ;; not merely the acceptance is right. + (print-f64 (match (parse-f64 (bytes "0")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "3.5")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "-3.5")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "+0.25")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "1e3")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "1.5E-2")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "12")) (Some v) v None -999.0)) + (newline) + ;; Refused. Every one of these is a number out of strtod, which is the point. + (print-f64 (match (parse-f64 (bytes "")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "abc")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "1x")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes ".")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "1e")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "1e+")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes " 1")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "1 ")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "0x10")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "nan")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes "+")) (Some v) v None -999.0)) + (newline) + ;; A trailing dot with no fraction is a C float literal and is accepted; a + ;; leading one is too. Both are here because they are the boundary the + ;; digit counter, not the position, decides. + (print-f64 (match (parse-f64 (bytes "1.")) (Some v) v None -999.0)) (print-str " ") + (print-f64 (match (parse-f64 (bytes ".5")) (Some v) v None -999.0)) + (newline) + + ;; Parsing a trimmed field, which is why both exist. + (print-f64 (match (parse-f64 (trim (bytes " 2.25 "))) (Some v) v None -999.0)) + (newline) + 0) diff --git a/test/test_acceptance.ml b/test/test_acceptance.ml index a3c5e84..28345c7 100644 --- a/test/test_acceptance.ml +++ b/test/test_acceptance.ml @@ -143,6 +143,30 @@ let () = outputs "bytes, parsing and numbers" "programs/text.flan" text_out; outputs ~opt:"-O0" "bytes, parsing and numbers, -O0" "programs/text.flan" text_out; + (* index-of-bytes, trim, the byte classes and parse-f64. The search cases + are the ones that separate a correct loop from a lucky one: a match + only at the end, "aab" in "aaab" (where the first byte matches twice + before the needle does), a needle longer than the haystack, which must + answer None without building a window off the end, and the empty needle + at Some 0. trim prints inside brackets so the all-whitespace answer is + visible as [] — that input is also the one that would build a reversed + slice and trap. And parse-f64's refusals are every shape strtod hands + back a plausible number for: "", "abc", "1x", ".", "1e", " 1", "1 ", + "0x10", "nan". *) + let bytes2_out = + "6 0 4 2 1 \n\ + -1 -1 -1 0 0 0 \n\ + [hi][hi][hi][][][a b][x][x]\n\ + ttfff\n\ + ttttff\n\ + 0 3.5 -3.5 0.25 1000 0.015 12\n\ + -999 -999 -999 -999 -999 -999 -999 -999 -999 -999 -999\n\ + 1 0.5\n\ + 2.25\n" + in + outputs "substring, trim and parse-f64" "programs/bytes2.flan" bytes2_out; + outputs ~opt:"-O0" "substring, trim and parse-f64, -O0" "programs/bytes2.flan" + bytes2_out; (* handler-bind and signal, spec-conditions.md §1 and §2: signal returns Unit and carries on, an unhandled one is a no-op, a nested frame does not displace the one outside it, and the stack is restored after. *)