flan/tools/colon-to-dot.py
Joseph Ferano 9a820d86cd Sweep every field label from the colon spelling to the dot
The script is in tools/ rather than thrown away, because two lanes are
writing Flan in the old spelling right now and their files need the same
pass at merge.

It works on forms, not on text: a keyword becomes a dot only where it sits
in a field-label position inside a brace, so an enum member in value
position, a map key inside an EDN string and a type-position {K V} are all
left alone. :keys keeps its colon -- it names no field.
2026-09-12 14:47:54 +07:00

238 lines
8.3 KiB
Python
Executable File

#!/usr/bin/env python3
"""Rewrite struct field labels from the colon spelling to the dot spelling.
`{:x 1.0 :y 2.0}` becomes `{.x 1.0 .y 2.0}`, and the destructuring pair
`{inner :field}` becomes `{inner .field}`. `:keys` is left alone: it names no
field, it is an instruction to the compiler, so the dot keeps exactly one
meaning -- "this names a field".
Only keywords that sit in a *field-label* position inside a brace form are
touched. Enum members, map keys and every keyword inside a string literal or a
comment are left as they are. The lexing rules here mirror lib/reader.ml.
Re-runnable: converting an already-converted file is a no-op, so this can be
run again over files a parallel branch wrote in the old spelling.
tools/colon-to-dot.py <file-or-dir>... # rewrite .flan in place
tools/colon-to-dot.py --check <file-or-dir>... # report, change nothing
tools/colon-to-dot.py --in-strings <file.ml>... # Flan inside "..." literals
tools/colon-to-dot.py --raw-ml <file.ml>... # Flan in a {flan|...|flan} block
A directory is walked for `.flan` files only. An `.ml` file is converted when it
is named on the command line, which is how the Flan source embedded in
`lib/prelude.ml` and in the tests gets swept; it is deliberately not automatic,
because an OCaml record written `{ v: value }` would look like a field label to
the scan. Read the diff when sweeping `.ml`.
"""
import sys, os
DELIM = set('()[]{}";`~ \t\n\r,')
OPENERS = {'(': ')', '[': ']', '{': '}'}
CLOSERS = {')', ']', '}'}
class Atom:
def __init__(self, start, end, text):
self.start, self.end, self.text = start, end, text
self.tok = start # never moves; `start` may slide onto a quote sigil
def label(self):
"""The keyword name if this atom is a `:kw`, else None."""
t = self.text
return t[1:] if len(t) > 1 and t[0] == ':' else None
class Seq:
def __init__(self, open_char, start):
self.open_char, self.start = open_char, start
self.end = start
self.items = []
def label(self):
return None
def lex_forms(src, i, end, stop=None):
"""Read forms from src[i:end] until `stop` (a closing char) or exhaustion.
Returns (items, next_index). A quote/quasiquote/unquote prefix is folded
into the form it applies to, so a quoted value stays one element.
"""
items = []
n = end
pending_prefix = None # start offset of a sigil awaiting its form
def push(node):
nonlocal pending_prefix
if pending_prefix is not None:
node.start = pending_prefix
pending_prefix = None
items.append(node)
while i < n:
c = src[i]
if c in ' \t\n\r,':
i += 1
elif c == ';': # line comment
while i < n and src[i] != '\n':
i += 1
elif c == '"': # string literal
j = i + 1
while j < n and src[j] != '"':
j += 2 if src[j] == '\\' else 1
j = min(j + 1, n)
push(Atom(i, j, src[i:j]))
i = j
elif c == '\\': # character literal
j = i + 1
if j < n:
j += 1 # always one char
while j < n and src[j] not in DELIM:
j += 1
push(Atom(i, j, src[i:j]))
i = j
elif c in "'`~": # quote sugar
if pending_prefix is None:
pending_prefix = i
i += 2 if (c == '~' and i + 1 < n and src[i + 1] == '@') else 1
elif c in OPENERS:
node = Seq(c, i)
node.items, i = lex_forms(src, i + 1, n, OPENERS[c])
node.end = i
push(node)
elif c in CLOSERS:
return items, i + 1
else: # symbol or keyword
j = i
while j < n and src[j] not in DELIM:
j += 1
if j == i:
j = i + 1
push(Atom(i, j, src[i:j]))
i = j
return items, n
def collect(node, out):
"""Walk the form tree, recording the offsets of every colon to rewrite."""
if isinstance(node, Seq):
if node.open_char == '{':
items = node.items
k = 0
while k < len(items):
a = items[k]
b = items[k + 1] if k + 1 < len(items) else None
a_label = a.label()
b_label = b.label() if b is not None else None
if a_label == 'keys':
pass # a directive, not a field
elif a_label is not None:
out.append(a.tok) # {:field value}
elif b_label is not None and b_label != 'keys':
out.append(b.tok) # {pattern :field}
k += 2
for it in node.items:
collect(it, out)
def convert(src):
items, _ = lex_forms(src, 0, len(src))
out = []
for it in items:
collect(it, out)
if not out:
return src, 0
chars = list(src)
for off in out:
assert chars[off] == ':', "expected ':' at offset %d" % off
chars[off] = '.'
return ''.join(chars), len(out)
def convert_in_ocaml_strings(src):
"""Convert Flan source that sits inside ordinary OCaml `"..."` literals.
The tests hold their Flan snippets that way, so the plain scan skips right
over them. Each literal's raw text is scanned on its own, with `\\"` masked
to a same-length filler first so an escaped quote cannot be mistaken for the
start of a Flan string. Masking preserves length, so offsets map back 1:1.
"""
chars = list(src)
total = 0
i, n = 0, len(src)
while i < n:
c = src[i]
if c == '"':
j = i + 1
while j < n and src[j] != '"':
j += 2 if src[j] == '\\' else 1
body = src[i + 1:j]
masked = body.replace('\\"', '\\x')
_, offs = _offsets(masked)
for off in offs:
pos = i + 1 + off
if chars[pos] == ':':
chars[pos] = '.'
total += 1
i = j + 1
elif c == '(' and i + 1 < n and src[i + 1] == '*': # OCaml comment
i += 2
else:
i += 1
return ''.join(chars), total
def _offsets(src):
items, _ = lex_forms(src, 0, len(src))
out = []
for it in items:
collect(it, out)
return src, out
def walk(paths):
for p in paths:
if os.path.isdir(p):
for root, dirs, files in os.walk(p):
dirs[:] = [d for d in dirs
if d not in ('_build', '.git', 'vendor', 'node_modules')]
for f in sorted(files):
if f.endswith('.flan'):
yield os.path.join(root, f)
else:
yield p
def main(argv):
check = '--check' in argv
in_strings = '--in-strings' in argv
allow_raw_ml = '--raw-ml' in argv
paths = [a for a in argv[1:] if not a.startswith('--')] or ['.']
total_files = total_sites = 0
for path in walk(paths):
if path.endswith('.ml') and not in_strings and not allow_raw_ml:
sys.stderr.write(
"%s: an .ml file needs --in-strings (its Flan is inside OCaml "
"string literals), or --raw-ml if the Flan is in a {flan|...|flan} "
"block. A plain scan would read OCaml's `::` as a field label.\n"
% path)
return 2
with open(path, encoding='utf-8') as fh:
src = fh.read()
new, n = convert_in_ocaml_strings(src) if in_strings else convert(src)
if n:
total_files += 1
total_sites += n
print("%s: %d" % (path, n))
if not check:
with open(path, 'w', encoding='utf-8') as fh:
fh.write(new)
verb = "would convert" if check else "converted"
print("%s %d field labels across %d files" % (verb, total_sites, total_files))
return 1 if (check and total_sites) else 0
if __name__ == '__main__':
sys.exit(main(sys.argv))