The half the shadow stack was built for. A slot's entry in the frame is its address, null until the binding that fills it has run, so "not bound yet at this point" is a null and needs no liveness analysis. The daemon compiles a thunk that renders the types it already knows -- Tast.fn.slots, with snames beside them -- at the addresses the stopped program supplies, and reads the text back the way C-x C-e does. Nothing is copied out, because a value with no header is bytes with no meaning anywhere but in the program that holds it. That is render.ml's walk with its root changed, which is the pointer-rooted thunk NEXT.md said this needed, and one new arm in the backend: a cast from one pointer type to another, which emits nothing. Only named slots are recorded. A recorded slot escapes and stops being promotable, and the slots that would cost most are the ones with nothing to show -- dotimes' bound, the temporaries min and max use, the walk's own scratch. They are refused by name rather than shown under an invented one. Recording every slot was built and timed and is inside the noise, so the rule stands on what it shows. Four refusals, each by name and with its reason: a slot nobody named, a slot the program has not reached, a type the printer has no arm for, and two whole frames -- an evaluation's thunk, and a frame running a body that has been redefined since, where every slot index would be a guess. Measured, minimum of nine runs: +61% on call-heavy code over globals against +33% for the frames alone, 0.06% of a frame at 60fps.
405 lines
17 KiB
C
405 lines
17 KiB
C
/* flan_dev — the part of the host ABI that only a dev build has.
|
|
*
|
|
* A redefinition module reaches the host's functions and globals through
|
|
* symbols the host already exports: a cell for each function, the storage for
|
|
* each global. That covers everything the program was *built* with. It does
|
|
* not cover a name the module introduces — a defn or a defvar typed into the
|
|
* REPL after the process started — because there is no symbol in the host to
|
|
* bind to and ELF cannot grow one.
|
|
*
|
|
* So a name that is new at run time is keyed by string instead. This file is
|
|
* the two lookups that make that work, and deliberately nothing else:
|
|
*
|
|
* flan_dev_cell(name) the cell a new function lives in
|
|
* flan_dev_global(name, size, init) the storage a new global lives in
|
|
* flan_dev_emit(...) where an evaluated expression's rendering
|
|
* goes, piece by piece, to be read back
|
|
*
|
|
* Both are idempotent: the second module to mention a name gets what the first
|
|
* one got. That is the whole point. Two modules that each define their own
|
|
* copy of a new function would each call their own, and redefining it would
|
|
* update one of them.
|
|
*
|
|
* The table never moves. A module holds the address of a cell for as long as
|
|
* it is loaded, so a growable table would leave those addresses pointing into
|
|
* a freed allocation. Fixed capacity and a loud failure instead.
|
|
*
|
|
* Never dlclose a module. A cell holds an address inside that module's text,
|
|
* and unloading it leaves every call site pointing at unmapped memory. There
|
|
* is no unload path here on purpose.
|
|
*/
|
|
|
|
#include <stdint.h>
|
|
#include <stdio.h>
|
|
#include <stdlib.h>
|
|
#include <string.h>
|
|
|
|
#define FLAN_DEV_MAX 4096
|
|
|
|
typedef struct {
|
|
const char *name; /* strdup'd: the module that passed it may go away */
|
|
void *cell; /* a function's cell, or a global's storage */
|
|
size_t size; /* a global's size; 0 for a function */
|
|
} entry;
|
|
|
|
static entry table[FLAN_DEV_MAX];
|
|
static size_t used;
|
|
|
|
static void die(const char *what, const char *name) {
|
|
fprintf(stderr, "flan_dev: %s: %s\n", what, name);
|
|
fflush(stderr);
|
|
abort();
|
|
}
|
|
|
|
static entry *find(const char *name) {
|
|
for (size_t i = 0; i < used; i++)
|
|
if (strcmp(table[i].name, name) == 0) return &table[i];
|
|
return NULL;
|
|
}
|
|
|
|
static entry *intern(const char *name) {
|
|
if (used == FLAN_DEV_MAX) die("out of dev name slots", name);
|
|
entry *e = &table[used++];
|
|
e->name = strdup(name);
|
|
if (e->name == NULL) die("out of memory", name);
|
|
e->cell = NULL;
|
|
e->size = 0;
|
|
return e;
|
|
}
|
|
|
|
/* The cell a run-time-introduced function is called through. One indirection
|
|
* more than a function the host was built with, whose cell is a symbol the
|
|
* module can name directly — the compiler picks per name, so the common case
|
|
* stays a single load. */
|
|
void **flan_dev_cell(const char *name) {
|
|
entry *e = find(name);
|
|
if (e == NULL) e = intern(name);
|
|
return &e->cell;
|
|
}
|
|
|
|
/* Storage for a run-time-introduced global, allocated once.
|
|
*
|
|
* [init] is its declared initial value, or NULL for all-zero. It is copied on
|
|
* the allocation and ignored on every call after it, which is where "a reload
|
|
* must not reset the program's state" lives: the second module to mention this
|
|
* name is a redefinition, and re-running an initialiser would throw away
|
|
* exactly what the reload exists to preserve. Doing it here rather than by a
|
|
* branch in the caller means the rule cannot be got wrong at one call site.
|
|
*
|
|
* A size mismatch is the layout-drift failure, caught at its first chance: the
|
|
* running process has already laid this memory out, and handing back the old
|
|
* allocation for a differently shaped type means the new body reads fields at
|
|
* the wrong offsets and nothing ever says so. Retyping a var needs a restart. */
|
|
void *flan_dev_global(const char *name, uint64_t size, const void *init) {
|
|
entry *e = find(name);
|
|
if (e == NULL) {
|
|
e = intern(name);
|
|
e->cell = calloc(1, size ? (size_t)size : 1);
|
|
if (e->cell == NULL) die("out of memory", name);
|
|
e->size = (size_t)size;
|
|
if (init != NULL && size > 0) memcpy(e->cell, init, (size_t)size);
|
|
return e->cell;
|
|
}
|
|
if (e->size != (size_t)size) die("size changed; restart to retype", name);
|
|
return e->cell;
|
|
}
|
|
|
|
/* ── The value of an evaluated expression ──────────────────────────── */
|
|
|
|
/* C-x C-e compiles a thunk that renders one expression and emits it here, a
|
|
* piece at a time. It is not written to stdout: stdout belongs to the program,
|
|
* it is in the hot path for anything that prints, and a dev-only feature must
|
|
* not put a branch in it. The daemon reads this back over the agent's socket.
|
|
*
|
|
* Emitting piece by piece rather than returning one string is what makes a
|
|
* composite renderer possible at all — a struct is its fields with punctuation
|
|
* between them, and concatenating that in the generated IR would mean an
|
|
* allocator the language does not have.
|
|
*
|
|
* The output bound lives here and nowhere else. A slice of a million elements
|
|
* renders with a loop the compiler cannot bound, so [emit] truncates and
|
|
* [end] says so with an ellipsis. One place enforcing it means no renderer has
|
|
* to carry a budget.
|
|
*
|
|
* [generation] is what makes the read safe without a handshake. The thunk runs
|
|
* on the game thread at a frame boundary, whenever that happens to be; the
|
|
* daemon waits for the counter to move rather than guessing it has.
|
|
*
|
|
* It is a *seqlock*, and it has to be a real one, because the reader is the
|
|
* agent's listener thread and the writer is the game thread and neither waits
|
|
* for the other. The counter is odd for exactly as long as a value is being
|
|
* written, so a reader that sees an odd count, or a different count either
|
|
* side of its copy, has read a value that was being overwritten underneath it
|
|
* and reads again. A count of 2k means k complete values; the count the
|
|
* outside world is given is that k, so that the daemon's "has it moved" keeps
|
|
* meaning "is there a new value".
|
|
*
|
|
* The copy is what makes it safe, and the API is shaped around that: a reader
|
|
* gets *bytes of its own*, not a pointer into [result]. The pointer version of
|
|
* this was the bug — it read the generation, then a length, then handed back
|
|
* the buffer itself, and the caller sent it down a socket some time later
|
|
* while the game thread was free to be a hundred bytes into the next value.
|
|
* A seqlock cannot validate a read that happens after it returns. */
|
|
|
|
#define RESULT_MAX 4096
|
|
static char result[RESULT_MAX];
|
|
static size_t result_len;
|
|
static int result_full;
|
|
static uint64_t generation;
|
|
|
|
void flan_dev_result_begin(void) {
|
|
/* Odd first, and only then the reset: the counter has to say "in progress"
|
|
* before the buffer stops being the value it used to be.
|
|
*
|
|
* The fence is the half of that a release *store* cannot do. A release store
|
|
* orders what comes before it, not what comes after, so the writes below —
|
|
* and every memcpy in [emit] — would be free to become visible ahead of the
|
|
* odd count, and a reader could see an even count either side of a copy it
|
|
* made while the buffer was being overwritten. Which is the bug this
|
|
* replaced, with more ceremony. So: mark it relaxed, fence, then write.
|
|
*
|
|
* Setting the low bit rather than incrementing, because a [begin] with no
|
|
* [end] is reachable and must not poison the counter for the life of the
|
|
* process. A render thunk that signals is stopped inside this window, and a
|
|
* restart taken from that break transfers past the thunk — [end] never runs.
|
|
* Repairing it here costs nothing in the ordinary case (2k, 2k+1, 2k+2) and
|
|
* means an abandoned write is over as soon as the next evaluation starts,
|
|
* rather than leaving every later read reporting "in progress" forever.
|
|
*
|
|
* What it does not fix, because the buffer cannot: an evaluation that runs
|
|
* while another is stopped mid-render shares this one buffer, so the inner
|
|
* value is the one that survives and the outer thunk, if it is ever resumed,
|
|
* appends to it. That was true before the counter was a seqlock. */
|
|
__atomic_store_n(&generation, generation | 1, __ATOMIC_RELAXED);
|
|
__atomic_thread_fence(__ATOMIC_RELEASE);
|
|
result_len = 0;
|
|
result_full = 0;
|
|
}
|
|
|
|
void flan_dev_emit(const uint8_t *bytes, int64_t len) {
|
|
size_t n = len < 0 ? 0 : (size_t)len;
|
|
if (result_len + n > RESULT_MAX) {
|
|
n = RESULT_MAX - result_len;
|
|
result_full = 1;
|
|
}
|
|
memcpy(result + result_len, bytes, n);
|
|
result_len += n;
|
|
}
|
|
|
|
static void emit_cstr(const char *s) {
|
|
flan_dev_emit((const uint8_t *)s, (int64_t)strlen(s));
|
|
}
|
|
|
|
/* Rendered in C so that u64 is not a lie: the language's own i64->bytes is
|
|
* signed, and anything past 2^63 would come back negative. */
|
|
void flan_dev_emit_u64(uint64_t x) {
|
|
char buf[32];
|
|
snprintf(buf, sizeof buf, "%llu", (unsigned long long)x);
|
|
emit_cstr(buf);
|
|
}
|
|
|
|
void flan_dev_emit_i64(int64_t x) {
|
|
char buf[32];
|
|
snprintf(buf, sizeof buf, "%lld", (long long)x);
|
|
emit_cstr(buf);
|
|
}
|
|
|
|
void flan_dev_emit_f64(double x) {
|
|
char buf[64];
|
|
snprintf(buf, sizeof buf, "%g", x);
|
|
emit_cstr(buf);
|
|
}
|
|
|
|
/* Quoted and escaped, in C, because doing it in the generated IR would be a
|
|
* loop per string and the language has no allocator to build the result in.
|
|
* A string whose content is not escaped does not round-trip and reads as a
|
|
* framing bug rather than as the value it is. */
|
|
void flan_dev_emit_str(const uint8_t *bytes, int64_t len) {
|
|
size_t n = len < 0 ? 0 : (size_t)len;
|
|
emit_cstr("\"");
|
|
for (size_t i = 0; i < n; i++) {
|
|
unsigned char c = bytes[i];
|
|
switch (c) {
|
|
case '"': emit_cstr("\\\""); break;
|
|
case '\\': emit_cstr("\\\\"); break;
|
|
case '\n': emit_cstr("\\n"); break;
|
|
case '\t': emit_cstr("\\t"); break;
|
|
case '\r': emit_cstr("\\r"); break;
|
|
default:
|
|
if (c < 0x20) {
|
|
char buf[8];
|
|
snprintf(buf, sizeof buf, "\\x%02x", c);
|
|
emit_cstr(buf);
|
|
} else {
|
|
flan_dev_emit(&c, 1);
|
|
}
|
|
}
|
|
}
|
|
emit_cstr("\"");
|
|
}
|
|
|
|
void flan_dev_result_end(void) {
|
|
if (result_full) {
|
|
/* Room is made for it rather than assumed: the buffer is full by
|
|
* definition when this fires. */
|
|
const char *ell = "...";
|
|
size_t k = strlen(ell);
|
|
if (result_len > RESULT_MAX - k) result_len = RESULT_MAX - k;
|
|
memcpy(result + result_len, ell, k);
|
|
result_len += k;
|
|
}
|
|
/* Last, and back to even, so a reader that sees the new generation sees the
|
|
* whole value. [| 1] first for the same reason [begin] sets rather than
|
|
* increments: this must land on an even count whatever state an abandoned
|
|
* write left behind. */
|
|
__atomic_store_n(&generation, (generation | 1) + 1, __ATOMIC_RELEASE);
|
|
}
|
|
|
|
/* Copy the current value out, with the counter that says which one it is.
|
|
*
|
|
* Returns 1 having copied a value that was complete for the whole of the copy,
|
|
* 0 if the game thread was in the middle of writing one — in which case [gen]
|
|
* is the last *complete* value's number and [len] is 0, so a caller polling
|
|
* for a new one keeps polling instead of being handed half of it. Spinning
|
|
* here is bounded: the writer is a render thunk between frames, not a loop,
|
|
* and the reader is the listener thread, which has nothing better to do.
|
|
*
|
|
* [cap] is the caller's buffer. A value longer than it is truncated, which is
|
|
* the only failure this can have and is a clamp rather than an overrun; the
|
|
* agent sizes its buffer at RESULT_MAX so it does not arise. */
|
|
/* What a caller's buffer has to be for the copy never to be truncated. The
|
|
* bound is declared in one place and asked for rather than written down twice:
|
|
* the agent's buffer and this one agreeing is the whole of "never truncated",
|
|
* and two literals in two files is how that stops being true. */
|
|
uint64_t flan_dev_result_cap(void) { return RESULT_MAX; }
|
|
|
|
int flan_dev_result_read(char *dst, uint64_t cap, uint64_t *gen,
|
|
uint64_t *len) {
|
|
for (int attempt = 0; attempt < 64; attempt++) {
|
|
uint64_t g1 = __atomic_load_n(&generation, __ATOMIC_ACQUIRE);
|
|
if (g1 & 1) continue; /* a write is in progress */
|
|
size_t n = __atomic_load_n(&result_len, __ATOMIC_RELAXED);
|
|
if (n > RESULT_MAX) n = RESULT_MAX; /* a torn read cannot overrun */
|
|
if ((uint64_t)n > cap) n = (size_t)cap;
|
|
memcpy(dst, result, n);
|
|
/* The copy must be ordered before the second read of the counter, or the
|
|
* check is of a copy the compiler was free to make afterwards. */
|
|
__atomic_thread_fence(__ATOMIC_ACQUIRE);
|
|
if (__atomic_load_n(&generation, __ATOMIC_ACQUIRE) == g1) {
|
|
*gen = g1 / 2;
|
|
*len = (uint64_t)n;
|
|
return 1;
|
|
}
|
|
}
|
|
/* Integer division is the same answer either side of a write in progress:
|
|
* during value k the counter is 2k-1 and k-1 are complete. */
|
|
*gen = __atomic_load_n(&generation, __ATOMIC_ACQUIRE) / 2;
|
|
*len = 0;
|
|
return 0;
|
|
}
|
|
|
|
/* ── The shadow stack ───────────────────────────────────────────────── */
|
|
|
|
/* plan.org's "Dev vs release builds" has had *Frames: shadow stack* in the dev
|
|
* column since the beginning. This is it, and it exists for one reason: the
|
|
* more a break loop can show, the less often a real debugger is needed. A
|
|
* stopped program that cannot say where it is sends someone to lldb.
|
|
*
|
|
* Native unwinding would be the other route and is deliberately not taken:
|
|
* plan.org lowers every non-local exit explicitly rather than through platform
|
|
* unwinding, so there is no .eh_frame walk to borrow, and a frame pointer walk
|
|
* gives addresses that only DWARF can turn back into names. A pushed record
|
|
* carries the name itself, is the same on wasm32 as it is here, and needs no
|
|
* agreement with the optimiser about what a frame looks like.
|
|
*
|
|
* The compiler emits the push and the pop inline (emit.ml, [emit_fn]) rather
|
|
* than calling in here: this is on every call in a dev build, and a call to
|
|
* record a call would double the thing being measured.
|
|
*
|
|
* A plain global, not a _Thread_local. It matches what the handler stack and
|
|
* the restart stack in flan_rt.c already assume — one thread runs Flan; the
|
|
* listener thread runs C and the loader and never enters a Flan body. If the
|
|
* language ever grows threads this becomes thread-local and the compiler's
|
|
* two stores become TLS-relative, which is the only change.
|
|
*
|
|
* Nothing in here walks the chain while the game thread is running. The break
|
|
* loop snapshots it on the stopped thread, exactly as it snapshots the restart
|
|
* list and for exactly the same reason: a chain read by another thread is a
|
|
* chain that can be popped underneath the reader. The accessors below are the
|
|
* snapshot's, and take the frame they were given rather than re-reading the
|
|
* head. */
|
|
|
|
typedef struct {
|
|
const char *name; /* the Flan name, qualified; not NUL-terminated */
|
|
int64_t namelen;
|
|
const char *loc; /* file:line:col, as Loc spells it */
|
|
int64_t loclen;
|
|
int32_t nslots;
|
|
int32_t spare;
|
|
} flan_fninfo;
|
|
|
|
typedef struct flan_frame {
|
|
struct flan_frame *prev;
|
|
const flan_fninfo *info;
|
|
/* One entry per slot, each null until the binding that fills that slot has
|
|
* run — so "not bound yet at the point this frame stopped" is a null and
|
|
* needs no liveness analysis to work out. Null altogether for a function
|
|
* with no named slot, and in a release build there is no frame at all.
|
|
* Read through [flan_dev_frame_slot], which is where the bound is checked. */
|
|
void **slots;
|
|
} flan_frame;
|
|
|
|
/* The compiler names this symbol directly. A redefinition module reaches it
|
|
* the same way it reaches any other host global — through the dynamic symbol
|
|
* table, which [--dev] links with -rdynamic. */
|
|
flan_frame *flan_frame_head;
|
|
|
|
/* [i] counts from the innermost. NULL past the end, which is how a caller
|
|
* learns the depth without a second walk. */
|
|
void *flan_dev_frame_at(int32_t i) {
|
|
flan_frame *f = flan_frame_head;
|
|
while (f != NULL && i > 0) { f = f->prev; i--; }
|
|
return f;
|
|
}
|
|
|
|
int32_t flan_dev_frame_count(void) {
|
|
int32_t n = 0;
|
|
for (flan_frame *f = flan_frame_head; f != NULL; f = f->prev) {
|
|
n++;
|
|
if (n > 100000) break; /* a corrupt chain says so rather than hanging */
|
|
}
|
|
return n;
|
|
}
|
|
|
|
const char *flan_dev_frame_name(const void *frame, int64_t *len) {
|
|
const flan_frame *f = frame;
|
|
if (f == NULL || f->info == NULL) { *len = 0; return NULL; }
|
|
*len = f->info->namelen;
|
|
return f->info->name;
|
|
}
|
|
|
|
const char *flan_dev_frame_loc(const void *frame, int64_t *len) {
|
|
const flan_frame *f = frame;
|
|
if (f == NULL || f->info == NULL) { *len = 0; return NULL; }
|
|
*len = f->info->loclen;
|
|
return f->info->loc;
|
|
}
|
|
|
|
int32_t flan_dev_frame_nslots(const void *frame) {
|
|
const flan_frame *f = frame;
|
|
return (f == NULL || f->info == NULL) ? 0 : f->info->nslots;
|
|
}
|
|
|
|
/* Where slot [i] of this frame lives, or NULL — which means one of three
|
|
* things, all of which are "there is nothing to read here": this build records
|
|
* no slots, the index is not one of them, or the binding that fills it had not
|
|
* run when the frame stopped. A caller renders what it is given and refuses
|
|
* what it is not; nothing here guesses. */
|
|
void *flan_dev_frame_slot(const void *frame, int32_t i) {
|
|
const flan_frame *f = frame;
|
|
if (f == NULL || f->info == NULL || f->slots == NULL) return NULL;
|
|
if (i < 0 || i >= f->info->nslots) return NULL;
|
|
return f->slots[i];
|
|
}
|
|
|