From 2acb70b8db0128954c255aaa5378f17fe876647f Mon Sep 17 00:00:00 2001 From: Joseph Ferano Date: Sat, 12 Sep 2026 03:49:52 +0700 Subject: [PATCH] Decoding is the only part of a string library that needs no allocator Odin's core/strings and all of core/fmt take an allocator; core/unicode/utf8 does not, because decoding is classification and every answer is a number. That line is where the port stops, and the refusals at the foot of the file say so by name rather than leaving a caller to find out. The accept_sizes table becomes a cond over the lead byte. Its four awkward rows are the ones a hand-written decoder gets wrong one at a time, so they are written out: 0xc0/0xc1 lead nothing, 0xe0 and 0xf0 have a raised second-byte floor against overlongs, 0xed has a lowered ceiling against the surrogates. Two divergences from Odin, both the parse-i64 argument again. A malformed sequence carries ok:false instead of decoding to U+FFFD, which is a real code point a caller cannot tell from a failure; and encode-rune! answers None rather than silently substituting U+FFFD for a rune it was not given. Width stays 1 on a bad byte, which is Odin's rule and load-bearing: every loop here advances by it, and a 0 would hang rather than answer wrong. split cannot return a sequence it would have to own, so the cursor is what survives. It follows the allocating strings.split rather than Odin's own iterator, which drops a trailing empty field and disagrees with it. Case conversion is byte-wise and not in place: a literal is emitted into read-only memory, so lowering (bytes "Hi") would type check and segfault. --- lib/prelude.ml | 281 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 281 insertions(+) diff --git a/lib/prelude.ml b/lib/prelude.ml index 7955089..1e4e72a 100644 --- a/lib/prelude.ml +++ b/lib/prelude.ml @@ -407,6 +407,287 @@ let source = {flan| ;; Trailing junk is the case strtod is silent about, so the position has ;; to land exactly on the end. (if (= i (len s)) (Some (bytes->f64 s)) None))) + +;; ── UTF-8 ───────────────────────────────────────────────────────────── +;; +;; Ported from Odin's core/unicode/utf8/utf8.odin, which is the one corner of +;; a string library that is allocation-free by construction: decoding is +;; classification, and every answer it gives is a number. Everything else in +;; Odin's core/strings and all of core/fmt takes `allocator := +;; context.allocator`, and is therefore refused below rather than ported. +;; +;; Odin's 256-entry accept_sizes table becomes a cond over the lead byte here. +;; The table is the cache-friendly form and the cond is the one you can check +;; by reading, and nothing in a game decodes UTF-8 in a hot loop — DrawText +;; hands the bytes straight to raylib. +;; +;; The four rules that table encodes, and which a hand-written decoder gets +;; wrong one at a time: +;; +;; 0x80..0xc1 never a lead byte. 0x80..0xbf are continuation bytes, and +;; 0xc0 and 0xc1 could only ever begin an *overlong* two-byte +;; spelling of an ASCII character — the encoding that lets +;; "\xc0\xaf" smuggle a "/" past a check for one. +;; 0xe0 second byte 0xa0..0xbf and not 0x80..0xbf; the low half is +;; the overlong three-byte range. +;; 0xed second byte 0x80..0x9f. The high half is U+D800..U+DFFF, +;; the UTF-16 surrogates, which are not scalar values. +;; 0xf0, 0xf4 second byte 0x90..0xbf and 0x80..0x8f: overlong below, +;; and past U+10FFFF above. 0xf5..0xff lead nothing at all. +;; +;; A rune is an i32 and not a type of its own. That is Odin's answer too — +;; its `rune` is a four-byte integer distinguished only by a flag on the +;; basic-type row (src/types.cpp, the Basic_rune entry) — so nothing in the +;; checker has to learn a new type for any of this. + +;; One deliberate divergence from Odin, and it is the parse-i64 argument over +;; again. Odin's decode_rune answers RUNE_ERROR — U+FFFD — for malformed +;; bytes, and U+FFFD is a perfectly real code point that a well-formed string +;; may contain, so a caller cannot tell a decoded replacement character from a +;; failure to decode. This carries `ok` instead, and leaves `code` 0 when it +;; is false. +;; +;; `width` is 1 on a malformed byte and 0 only for an empty input. That is +;; Odin's rule and it is load-bearing rather than cosmetic: every loop below +;; advances by `width`, so a 0 there on a bad byte is an infinite loop, not a +;; wrong number. +(defstruct Rune [code i32 width i32 ok bool]) + +(defn rune-start? [b u8] bool + (!= (bit-and b 0xc0) 0x80)) + +(defn decode-rune [s [u8]] Rune + (when (= (len s) 0) + (return (Rune {:code 0 :width 0 :ok false}))) + (let [b0 (at s 0)] + (when (< b0 0x80) + (return (Rune {:code (i32 b0) :width 1 :ok true}))) + ;; size 0 means "this byte cannot lead"; lo/hi are the *second* byte's + ;; accepted range, which is the only place the overlong and surrogate + ;; rules live. Bytes three and four are always 0x80..0xbf. + (let [size 0 + lo (u8 0x80) + hi (u8 0xbf)] + (cond + (< b0 0xc2) (set size 0) + (<= b0 0xdf) (set size 2) + (= b0 0xe0) (do (set size 3) (set lo (u8 0xa0))) + (<= b0 0xec) (set size 3) + (= b0 0xed) (do (set size 3) (set hi (u8 0x9f))) + (<= b0 0xef) (set size 3) + (= b0 0xf0) (do (set size 4) (set lo (u8 0x90))) + (<= b0 0xf3) (set size 4) + (= b0 0xf4) (do (set size 4) (set hi (u8 0x8f))) + :else (set size 0)) + (when (= size 0) + (return (Rune {:code 0 :width 1 :ok false}))) + ;; A sequence cut off by the end of the slice. Width 1, so a caller + ;; scanning a buffer boundary makes progress instead of stalling. + (when (> size (len s)) + (return (Rune {:code 0 :width 1 :ok false}))) + (let [b1 (at s 1)] + (when (or (< b1 lo) (> b1 hi)) + (return (Rune {:code 0 :width 1 :ok false}))) + (when (= size 2) + (return (Rune {:code (bit-or (<< (i32 (bit-and b0 0x1f)) 6) + (i32 (bit-and b1 0x3f))) + :width 2 :ok true}))) + (let [b2 (at s 2)] + (when (or (< b2 0x80) (> b2 0xbf)) + (return (Rune {:code 0 :width 1 :ok false}))) + (when (= size 3) + (return (Rune {:code (bit-or (bit-or (<< (i32 (bit-and b0 0x0f)) 12) + (<< (i32 (bit-and b1 0x3f)) 6)) + (i32 (bit-and b2 0x3f))) + :width 3 :ok true}))) + (let [b3 (at s 3)] + (when (or (< b3 0x80) (> b3 0xbf)) + (return (Rune {:code 0 :width 1 :ok false}))) + (Rune {:code (bit-or (bit-or (<< (i32 (bit-and b0 0x07)) 18) + (bit-or (<< (i32 (bit-and b1 0x3f)) 12) + (<< (i32 (bit-and b2 0x3f)) 6))) + (i32 (bit-and b3 0x3f))) + :width 4 :ok true}))))))) + +;; Decode at a byte offset. None when the offset is not on a rune boundary or +;; the bytes there are malformed, which is stricter than Odin's rune_at — that +;; one hands back RUNE_ERROR and the caller carries on with a wrong character. +(defn rune-at [s [u8] i i32] (Option i32) + (if (or (< i 0) (>= i (len s))) + None + (let [r (decode-rune (slice s i (len s)))] + (if (.ok r) (Some (.code r)) None)))) + +;; Counted through decode-rune rather than through a second walk of its own. +;; Odin keeps a separate rune_count_in_bytes that re-implements the size +;; table; two copies of that classification is two places for the surrogate +;; rule to be right in only one of them. +;; +;; A malformed byte counts as one, which is what a replacement-character +;; renderer would draw, so this agrees with what the screen shows. +(defn rune-count [s [u8]] i32 + (let [i 0 + n 0] + (while (< i (len s)) + (let [r (decode-rune (slice s i (len s)))] + (set i (+ i (.width r))) + (set n (+ n 1)))) + n)) + +(defn valid-utf8? [s [u8]] bool + (let [i 0] + (while (< i (len s)) + (let [r (decode-rune (slice s i (len s)))] + (when (not (.ok r)) + (return false)) + (set i (+ i (.width r))))) + true)) + +;; How many bytes this code point encodes to, or None if it is not a scalar +;; value. Odin's rune_size answers -1 for the refusals; a sentinel index is +;; exactly what index-of-i32 avoids above, so this is an Option like the rest +;; of the file. +(defn rune-size [code i32] (Option i32) + (cond + (< code 0) None + (<= code 0x7f) (Some 1) + (<= code 0x7ff) (Some 2) + (and (>= code 0xd800) (<= code 0xdfff)) None + (<= code 0xffff) (Some 3) + (<= code 0x10ffff) (Some 4) + :else None)) + +;; Encoding is the one operation here whose result is not a slice of its +;; input, because the bytes it makes existed nowhere before. With no allocator +;; the only shape left is Odin's own allocation-free one — strings.Builder +;; built by builder_from_bytes over a caller's backing array (builder.odin, +;; builder_from_bytes: "Uses Nil Allocator - Does NOT allocate") — reduced to +;; its essential case: write into a buffer the caller owns, and say how much +;; was written. +;; +;; None rather than a partial write when the buffer is short, and None rather +;; than Odin's silent substitution of U+FFFD for an invalid rune. Odin's +;; encode_rune rewrites a surrogate or an out-of-range value to the +;; replacement character and reports success; the caller then finds three +;; bytes of U+FFFD in its buffer and no indication that it asked for something +;; else. Nothing is written at all when this answers None. +(defn encode-rune! [dst [u8] code i32] (Option i32) + (match (rune-size code) + None None + (Some w) + (if (> w (len dst)) + None + (do + (cond + (= w 1) + (set (at dst 0) (u8 code)) + (= w 2) + (do (set (at dst 0) (u8 (bit-or 0xc0 (>> code 6)))) + (set (at dst 1) (u8 (bit-or 0x80 (bit-and code 0x3f))))) + (= w 3) + (do (set (at dst 0) (u8 (bit-or 0xe0 (>> code 12)))) + (set (at dst 1) (u8 (bit-or 0x80 (bit-and (>> code 6) 0x3f)))) + (set (at dst 2) (u8 (bit-or 0x80 (bit-and code 0x3f))))) + :else + (do (set (at dst 0) (u8 (bit-or 0xf0 (>> code 18)))) + (set (at dst 1) (u8 (bit-or 0x80 (bit-and (>> code 12) 0x3f)))) + (set (at dst 2) (u8 (bit-or 0x80 (bit-and (>> code 6) 0x3f)))) + (set (at dst 3) (u8 (bit-or 0x80 (bit-and code 0x3f)))))) + (Some w))))) + +;; ── Splitting ───────────────────────────────────────────────────────── +;; +;; `split` returning a sequence of fields must allocate the sequence, and +;; there is no allocator — so it is refused by name at the bottom of this +;; file, and this is the shape that survives. It is Odin's +;; split_by_byte_iterator (strings.odin): a cursor holding the rest of the +;; input, handing back one field at a time. Every field is a slice *of the +;; caller's bytes*; nothing is copied and nothing is owned. +;; +;; One divergence, and it is a wart of Odin's rather than a decision. Odin's +;; iterator stops on an empty final field, so "a,b," iterates a and b and the +;; trailing empty field is lost — while Odin's own allocating strings.split +;; returns ["a", "b", ""] for the same input. The two disagree. This follows +;; split: n separators always yield n+1 fields, an empty input yields one +;; empty field, and `rest` is exhausted only after the last one is taken. That +;; is the rule you can state without exceptions, and the one a caller counting +;; comma-separated columns needs. +(defstruct Split [rest [u8] sep u8 more bool]) + +(defn split-on-byte [s [u8] sep u8] Split + (Split {:rest s :sep sep :more true})) + +(defn split-next! [it (Ptr Split)] (Option [u8]) + (when (not (.more it)) + (return None)) + (match (index-of-byte (.rest it) (.sep it)) + (Some i) + (let [field (slice (.rest it) 0 i)] + (set (.rest it) (slice (.rest it) (+ i 1) (len (.rest it)))) + (Some field)) + None + (let [field (.rest it)] + (set (.more it) false) + (set (.rest it) (slice (.rest it) (len (.rest it)) (len (.rest it)))) + (Some field)))) + +;; ── ASCII case ──────────────────────────────────────────────────────── +;; +;; Byte in, byte out, and *not* a function over a slice. Odin's to_lower and +;; to_upper both allocate a new string (core/strings/conversion.odin), which +;; is not available here; the obvious substitute — lowering a [u8] in place — +;; is a trap, and it is worth saying why rather than shipping it. A string +;; literal is emitted `private unnamed_addr constant` (emit.ml), so it lives +;; in read-only memory, and (bytes "Hello") is a [u8] pointing straight at it. +;; An in-place lower-ascii! would type check against that slice and segfault +;; on the store. Given a byte function, a caller that really does own its +;; buffer writes the two-line loop itself and can see what it is writing to. +;; +;; ASCII only, and only the 26 letters: case outside ASCII is not a byte +;; operation at all — it is per-code-point, it is not length-preserving (ß +;; upcases to SS), and it is locale-dependent (Turkish dotless ı). A byte +;; table that pretended otherwise would be wrong in the quiet way. +(defn lower-ascii [b u8] u8 + (if (and (>= b \A) (<= b \Z)) (+ b 32) b)) + +(defn upper-ascii [b u8] u8 + (if (and (>= b \a) (<= b \z)) (- b 32) b)) + +;; Case-insensitive comparison as a fold over both inputs, which is the useful +;; half of to_lower and needs no storage at all: comparing two lowered copies +;; is what a caller wanted, and this is that answer without either copy. +(defn bytes-ci=? [a [u8] b [u8]] bool + (if (!= (len a) (len b)) + false + (do + (dotimes [i (len a)] + (when (!= (lower-ascii (at a i)) (lower-ascii (at b i))) + (return false))) + true))) + +;; ── Refused, by name ────────────────────────────────────────────────── +;; +;; Every one of these needs to produce bytes that did not exist in its input, +;; and there is no allocator, so each is absent rather than approximated. +;; None of them is hard to write once `(Vec u8)` and an allocator exist; all +;; of them are impossible to write honestly today. +;; +;; join, concat build one buffer out of several inputs. +;; to-lower, to-upper a new string, per Odin's conversion.odin. The +;; byte-wise and folding-comparison forms above are +;; what is available without one. +;; split the *sequence* of fields is itself an allocation. +;; split-on-byte / split-next! above is the same +;; information with no sequence to own. +;; replace, repeat, pad same reason as join. +;; string-from-bytes a [u8] cannot become a `string` here even though +;; the layouts are identical; see the report. +;; format, sprintf Odin's fmt.aprintf family, all allocating. +;; Builder strings.Builder is (defstruct Builder [buf +;; (Vec u8)]), which spec-memory.md already makes +;; move-only by the rule that a struct containing a +;; Vec is move-only. It needs the Vec, not a spec +;; change. |flan} let file = ""