feat(tooling): port the doc/lint/grammar checks to Ludic (no Python)
Replace the Python doc/lint/vocabulary guards with Ludic equivalents that run through the `x` task runner, so the checks need no Python interpreter: x check-docs every ```ludic doc fence parses (or is marked) x check-impl every implemented feature has a docs/language page x check-vocabulary vocabulary in sync across grammar / lexer / header / parser x lint-asset <file> validate one editor .json / .xml asset New fragments: tools/x/json.ludic (a small JSON reader — objects/arrays/ strings with \uXXXX + surrogates/numbers/literals, used by the vocabulary check's grammar navigation and the asset validator) and tools/x/checks.ludic (the checks + string helpers + a minimal XML well-formedness validator). `x test-tools` now runs the vocabulary + docs-coverage checks and the JSON/XML asset validation through Ludic instead of python3; ci.yml's docs-coverage step calls `x check-impl` / `x check-vocabulary`. Each port was verified against its former Python script for exact verdict parity on the clean tree and on injected drift (a removed keyword, a broken grammar alternation, an undocumented method). Deletes the superseded scripts: tools/check-vocabulary.py, tools/check-docs.py, tools/docgen/check-impl.py, tools/docgen/validate.py. The docgen site generator (gen.py/check.py/palette.py) and the LSP protocol driver (test-lsp.py) remain and are tracked separately. Toolchain unchanged (seed byte-identical); `x test` (56) and `x test-tools` (29) stay green. Part of #31 Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
parent
c040fff8c8
commit
e65e862244
11 changed files with 922 additions and 486 deletions
|
|
@ -75,6 +75,10 @@ jobs:
|
||||||
- name: Docs cover the implementation
|
- name: Docs cover the implementation
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
|
# check-impl / check-vocabulary are written in Ludic and run through
|
||||||
|
# x — no Python. (check-vocabulary also runs in `x test-tools`.) The
|
||||||
|
# site generator (gen.py) is still Python; its port is tracked.
|
||||||
|
bin/x check-impl
|
||||||
|
bin/x check-vocabulary
|
||||||
python3 tools/docgen/gen.py --out build/pages
|
python3 tools/docgen/gen.py --out build/pages
|
||||||
python3 tools/docgen/check.py build/pages
|
python3 tools/docgen/check.py build/pages
|
||||||
python3 tools/docgen/check-impl.py
|
|
||||||
|
|
|
||||||
|
|
@ -1,78 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""check-docs.py — every ```ludic fence in the docs is checked against the
|
|
||||||
compiler, so documentation cannot drift away from the language.
|
|
||||||
|
|
||||||
A fence declares its own intent with an in-fence comment (`#` is a Ludic
|
|
||||||
comment, so the marker is valid code and visible to a reader):
|
|
||||||
|
|
||||||
# doc-check: skip illustrative or pseudo-syntax; not checked
|
|
||||||
# doc-check: expect-error must FAIL to compile (error demonstrations)
|
|
||||||
|
|
||||||
Anything else must PARSE. The gate is `ludicc --fmt`, not a full build: it runs
|
|
||||||
lex + parse, which is what catches syntax drift, but does not resolve
|
|
||||||
identifiers — so an excerpt may reference components and registers that live in
|
|
||||||
the surrounding program it was lifted from, and still be checked.
|
|
||||||
|
|
||||||
A bare fragment is wrapped first: a declaration list goes inside
|
|
||||||
`game DocCheck { ... }`, loose statements inside a Start system.
|
|
||||||
"""
|
|
||||||
import re, subprocess, sys, os, tempfile
|
|
||||||
|
|
||||||
LC = './build/ludicc'
|
|
||||||
DECL = ('program','property','model','handler','enum','function',
|
|
||||||
'extern','const','var','ui','struct','import')
|
|
||||||
|
|
||||||
def classify(body):
|
|
||||||
first = next((l.strip() for l in body.split('\n')
|
|
||||||
if l.strip() and not l.strip().startswith('#')), '')
|
|
||||||
head = first.split('(')[0].split()[0] if first else ''
|
|
||||||
if head == 'program': return 'whole'
|
|
||||||
return 'decls' if head in DECL else 'stmts'
|
|
||||||
|
|
||||||
def wraps(body, kind):
|
|
||||||
"""Candidate framings, best guess first. An excerpt often mixes declarations
|
|
||||||
with loose statements, so both are tried and either parsing counts."""
|
|
||||||
if kind == 'whole': return [body]
|
|
||||||
as_decls = 'program DocCheck {\n' + body + '\n}\n'
|
|
||||||
as_stmts = 'program DocCheck {\n handler DocS phase Start {\n' + body + '\n }\n}\n'
|
|
||||||
return [as_decls, as_stmts] if kind == 'decls' else [as_stmts, as_decls]
|
|
||||||
|
|
||||||
def fences(path):
|
|
||||||
t = open(path).read()
|
|
||||||
for m in re.finditer(r'^```ludic\n(.*?)^```', t, re.S | re.M):
|
|
||||||
yield t[:m.start()].count('\n') + 1, m.group(1)
|
|
||||||
|
|
||||||
def main(argv):
|
|
||||||
report = '--report' in argv
|
|
||||||
docs = [p for p in argv if p.endswith('.md')]
|
|
||||||
ok = bad = skipped = 0
|
|
||||||
failures = []
|
|
||||||
for path in docs:
|
|
||||||
for line, body in fences(path):
|
|
||||||
if '# doc-check: skip' in body:
|
|
||||||
skipped += 1; continue
|
|
||||||
expect_error = '# doc-check: expect-error' in body
|
|
||||||
kind = classify(body)
|
|
||||||
parsed, first_err = False, None
|
|
||||||
for src in wraps(body, kind):
|
|
||||||
with tempfile.NamedTemporaryFile('w', suffix='.ludic', delete=False) as f:
|
|
||||||
f.write(src); tmp = f.name
|
|
||||||
# bytes, not text: a diagnostic may echo a partial multibyte char
|
|
||||||
r = subprocess.run([LC, tmp, '--fmt'], capture_output=True)
|
|
||||||
os.unlink(tmp)
|
|
||||||
if r.returncode == 0: parsed = True; break
|
|
||||||
# keep the best-guess framing's error; later ones are fallbacks
|
|
||||||
if first_err is None: first_err = r.stderr
|
|
||||||
if parsed != (not expect_error):
|
|
||||||
bad += 1
|
|
||||||
want = 'be rejected' if expect_error else 'parse'
|
|
||||||
err = (first_err or b'').decode('utf-8','replace').strip().split('\n')[0]
|
|
||||||
failures.append(f"{path}:{line} expected to {want}: {err[:90]}")
|
|
||||||
else:
|
|
||||||
ok += 1
|
|
||||||
print(f" doc fences: {ok} as documented, {bad} drifted, {skipped} skipped")
|
|
||||||
for f in failures: print(f" {f}")
|
|
||||||
if report: return 0
|
|
||||||
return 1 if bad else 0
|
|
||||||
|
|
||||||
sys.exit(main(sys.argv[1:]))
|
|
||||||
|
|
@ -1,179 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""Guard against the failure mode this whole layout exists to prevent.
|
|
||||||
|
|
||||||
The Ludic vocabulary — keywords, types, phases, builtins, intrinsics — is
|
|
||||||
written down in five places that cannot include each other:
|
|
||||||
|
|
||||||
compiler/**/*.c BUILTINS[] the language itself
|
|
||||||
compiler/**/*.c INTRINSICS[] the language itself
|
|
||||||
tools/ludic-tools/ ludic_syntax.h the toolchain's source of truth
|
|
||||||
tools/editors/shared/ the TextMate grammar (JSON, no includes)
|
|
||||||
tools/editors/jetbrains/ LudicVocabulary (Kotlin, no includes)
|
|
||||||
|
|
||||||
Adding a builtin to the compiler and forgetting the rest is silent: the editor
|
|
||||||
just stops colouring it, and nobody notices for months. So it is checked.
|
|
||||||
"""
|
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import re
|
|
||||||
import sys
|
|
||||||
|
|
||||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
||||||
problems = []
|
|
||||||
|
|
||||||
|
|
||||||
def read(*parts):
|
|
||||||
with open(os.path.join(ROOT, *parts)) as fh:
|
|
||||||
return fh.read()
|
|
||||||
|
|
||||||
|
|
||||||
def c_string_list(text, name):
|
|
||||||
"""Names in `static const char* NAME[] = { "a", "b", 0 };`"""
|
|
||||||
m = re.search(r"\b" + name + r"\s*\[\s*\]\s*=\s*\{(.*?)\}\s*;", text, re.S)
|
|
||||||
if not m:
|
|
||||||
problems.append(f"{name}: not found")
|
|
||||||
return set()
|
|
||||||
return set(re.findall(r'"([^"]+)"', m.group(1)))
|
|
||||||
|
|
||||||
|
|
||||||
def c_table_names(text, name):
|
|
||||||
"""First string of each row in a `{ "name", ... }` table."""
|
|
||||||
m = re.search(r"\b" + name + r"\s*\[\s*\]\s*=\s*\{(.*?)\n\s*\}\s*;", text, re.S)
|
|
||||||
if not m:
|
|
||||||
problems.append(f"{name}: not found")
|
|
||||||
return set()
|
|
||||||
return set(re.findall(r'\{\s*"([^"]+)"', m.group(1)))
|
|
||||||
|
|
||||||
|
|
||||||
def kotlin_set(text, name):
|
|
||||||
"""Names in `val NAME = setOf(...)`, found by matching the parentheses —
|
|
||||||
a non-greedy regex silently runs past a one-line setOf into the next one."""
|
|
||||||
m = re.search(r"\bval\s+" + name + r"\s*=\s*setOf\(", text)
|
|
||||||
if not m:
|
|
||||||
problems.append(f"Kotlin {name}: not found")
|
|
||||||
return set()
|
|
||||||
i = m.end()
|
|
||||||
depth = 1
|
|
||||||
while i < len(text) and depth:
|
|
||||||
if text[i] == "(":
|
|
||||||
depth += 1
|
|
||||||
elif text[i] == ")":
|
|
||||||
depth -= 1
|
|
||||||
i += 1
|
|
||||||
body = re.sub(r"//[^\n]*", "", text[m.end():i - 1])
|
|
||||||
return set(re.findall(r'"([^"]+)"', body))
|
|
||||||
|
|
||||||
|
|
||||||
def grammar_alternation(grammar, path, marker):
|
|
||||||
"""Names in a `\\b(a|b|c)\\b` match inside the TextMate grammar."""
|
|
||||||
node = grammar
|
|
||||||
for key in path:
|
|
||||||
node = node[key]
|
|
||||||
for pat in node:
|
|
||||||
rx = pat.get("match", "")
|
|
||||||
if marker in rx:
|
|
||||||
inner = re.search(r"\\b\((.*?)\)\\b", rx)
|
|
||||||
if inner:
|
|
||||||
return set(inner.group(1).split("|"))
|
|
||||||
problems.append(f"grammar: no rule containing {marker!r}")
|
|
||||||
return set()
|
|
||||||
|
|
||||||
|
|
||||||
def compare(label, reference, other, other_label, ignore=frozenset()):
|
|
||||||
missing = (reference - other) - ignore
|
|
||||||
extra = (other - reference) - ignore
|
|
||||||
if missing:
|
|
||||||
problems.append(f"{other_label} is missing {label}: {', '.join(sorted(missing))}")
|
|
||||||
if extra:
|
|
||||||
problems.append(f"{other_label} has unknown {label}: {', '.join(sorted(extra))}")
|
|
||||||
|
|
||||||
|
|
||||||
syntax_h = read("tools", "ludic-tools", "ludic_syntax.h")
|
|
||||||
def read_table_owner(table):
|
|
||||||
"""Find whichever compiler translation unit declares a table.
|
|
||||||
|
|
||||||
The compiler is split by concern and files move; hardcoding a path here
|
|
||||||
turns a routine refactor into a spurious vocabulary failure.
|
|
||||||
"""
|
|
||||||
import glob
|
|
||||||
for path in sorted(glob.glob(os.path.join(ROOT, "compiler", "**", "*.c"), recursive=True)):
|
|
||||||
text = open(path, encoding="utf-8").read()
|
|
||||||
if re.search(r"\b%s\s*\[\s*\]" % re.escape(table), text):
|
|
||||||
return text
|
|
||||||
return None # the compiler is now written in Ludic (selfhost/); no C table
|
|
||||||
|
|
||||||
ludicc_c = read_table_owner("BUILTINS")
|
|
||||||
native_c = read_table_owner("INTRINSICS")
|
|
||||||
kotlin = read("tools", "editors", "jetbrains", "src", "main", "kotlin", "io", "ludic", "ide", "LudicTokens.kt")
|
|
||||||
grammar = json.loads(read("tools", "editors", "shared", "ludic.tmLanguage.json"))
|
|
||||||
|
|
||||||
# --- the toolchain header is the reference ---------------------------------
|
|
||||||
h_decl = c_string_list(syntax_h, "LUDIC_KW_DECL")
|
|
||||||
h_clause = c_string_list(syntax_h, "LUDIC_KW_CLAUSE")
|
|
||||||
h_stmt = c_string_list(syntax_h, "LUDIC_KW_STMT")
|
|
||||||
h_types = c_string_list(syntax_h, "LUDIC_TYPES")
|
|
||||||
h_phases = c_string_list(syntax_h, "LUDIC_PHASES")
|
|
||||||
h_widgets = c_string_list(syntax_h, "LUDIC_WIDGETS")
|
|
||||||
h_builtins = c_table_names(syntax_h, "LUDIC_BUILTINS")
|
|
||||||
h_intrinsics = c_table_names(syntax_h, "LUDIC_INTRINSICS")
|
|
||||||
|
|
||||||
# --- against the compiler's builtin/intrinsic tables, when a C compiler is
|
|
||||||
# present. The compiler is now written in Ludic (selfhost/), so this
|
|
||||||
# cross-check is skipped; the editor tools remain checked against each other.
|
|
||||||
if ludicc_c is not None:
|
|
||||||
compare("builtins", c_table_names(ludicc_c, "BUILTINS"), h_builtins, "ludic_syntax.h")
|
|
||||||
if native_c is not None:
|
|
||||||
compare("intrinsics", c_table_names(native_c, "INTRINSICS"), h_intrinsics, "ludic_syntax.h")
|
|
||||||
|
|
||||||
# --- against the JetBrains lexer -------------------------------------------
|
|
||||||
compare("declaration keywords", h_decl, kotlin_set(kotlin, "DECL"), "LudicTokens.kt")
|
|
||||||
compare("clause keywords", h_clause, kotlin_set(kotlin, "CLAUSE"), "LudicTokens.kt")
|
|
||||||
compare("statement keywords", h_stmt, kotlin_set(kotlin, "STMT"), "LudicTokens.kt")
|
|
||||||
compare("primitive types", h_types, kotlin_set(kotlin, "PRIMITIVES"), "LudicTokens.kt")
|
|
||||||
compare("phases", h_phases, kotlin_set(kotlin, "PHASES"), "LudicTokens.kt")
|
|
||||||
compare("widgets", h_widgets, kotlin_set(kotlin, "WIDGETS"), "LudicTokens.kt")
|
|
||||||
compare("builtins", h_builtins | h_intrinsics, kotlin_set(kotlin, "BUILTINS"), "LudicTokens.kt")
|
|
||||||
|
|
||||||
# --- against the TextMate grammar ------------------------------------------
|
|
||||||
gpath = ["repository", "builtin", "patterns"]
|
|
||||||
compare("builtins", h_builtins, grammar_alternation(grammar, gpath, "rng_chance"), "ludic.tmLanguage.json")
|
|
||||||
compare("intrinsics", h_intrinsics, grammar_alternation(grammar, gpath, "as_fixed"), "ludic.tmLanguage.json")
|
|
||||||
kpath = ["repository", "keyword", "patterns"]
|
|
||||||
compare("phases", h_phases, grammar_alternation(grammar, kpath, "FixedUpdate"), "ludic.tmLanguage.json")
|
|
||||||
compare("primitive types", h_types, grammar_alternation(grammar, kpath, "fixed"), "ludic.tmLanguage.json")
|
|
||||||
|
|
||||||
# --- against the actual self-hosted parser ---------------------------------
|
|
||||||
# The compiler moved to Ludic, so the old C cross-check (above) is dark. This
|
|
||||||
# replaces it: a declaration or clause keyword the highlighter colours must be
|
|
||||||
# one the parser actually dispatches on, or nobody notices when a keyword is
|
|
||||||
# highlighted but silently unparsed (exactly what happened to `scene`, `edge`,
|
|
||||||
# `needs`, …). RESERVED is the escape hatch for documented, not-yet-implemented
|
|
||||||
# keywords — and it is checked too, so a reserved word that gets implemented
|
|
||||||
# must be promoted out of it.
|
|
||||||
def parser_keywords():
|
|
||||||
kws = set()
|
|
||||||
for name in ("parse.ludic", "parse_game.ludic"):
|
|
||||||
text = open(os.path.join(ROOT, "selfhost", "frontend", name), encoding="utf-8").read()
|
|
||||||
kws |= set(re.findall(r'is_id\("([a-z]+)"\)', text))
|
|
||||||
kws |= set(re.findall(r'streq\(t\.text,\s*"([a-z]+)"\)', text))
|
|
||||||
return kws
|
|
||||||
|
|
||||||
pkw = parser_keywords()
|
|
||||||
h_reserved = c_string_list(syntax_h, "LUDIC_KW_RESERVED")
|
|
||||||
if not pkw:
|
|
||||||
problems.append("could not extract any keywords from selfhost/frontend/parse*.ludic")
|
|
||||||
else:
|
|
||||||
for label, kws in (("declaration keywords", h_decl), ("clause keywords", h_clause)):
|
|
||||||
unparsed = (kws - pkw) - h_reserved
|
|
||||||
if unparsed:
|
|
||||||
problems.append(f"ludic_syntax.h lists {label} the selfhost parser never dispatches on: {', '.join(sorted(unparsed))}")
|
|
||||||
promoted = h_reserved & pkw
|
|
||||||
if promoted:
|
|
||||||
problems.append(f"LUDIC_KW_RESERVED lists keywords the parser now accepts — promote them into the highlighted vocabulary: {', '.join(sorted(promoted))}")
|
|
||||||
|
|
||||||
if problems:
|
|
||||||
print("vocabulary drift:", file=sys.stderr)
|
|
||||||
for p in problems:
|
|
||||||
print(" -", p, file=sys.stderr)
|
|
||||||
sys.exit(1)
|
|
||||||
sys.exit(0)
|
|
||||||
|
|
@ -64,7 +64,7 @@ hovering any token shows a summary card from the real API data.
|
||||||
```bash
|
```bash
|
||||||
python3 tools/docgen/gen.py --out build/pages # generate the whole site
|
python3 tools/docgen/gen.py --out build/pages # generate the whole site
|
||||||
python3 tools/docgen/check.py build/pages # coverage + duplicate-token + link guard
|
python3 tools/docgen/check.py build/pages # coverage + duplicate-token + link guard
|
||||||
python3 tools/docgen/validate.py # compile every ```ludic example with bin/ludicc
|
bin/x check-docs # parse every ```ludic doc fence (Ludic; no Python)
|
||||||
```
|
```
|
||||||
|
|
||||||
`check.py` fails CI if any symbol in `inventory.json` lacks a page, if a token is
|
`check.py` fails CI if any symbol in `inventory.json` lacks a page, if a token is
|
||||||
|
|
|
||||||
|
|
@ -1,163 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""Docs coverage, anchored to the implementation — not a hand-kept list.
|
|
||||||
|
|
||||||
The failure mode this prevents: you add a feature to the language and forget to
|
|
||||||
document it. Nobody notices until someone goes looking for the docs that were
|
|
||||||
never written. So the set of things that MUST have a doc page is read straight
|
|
||||||
from the implementation, and every one is required to have a page (and every
|
|
||||||
namespace-method page is required to correspond to something real):
|
|
||||||
|
|
||||||
* namespace methods selfhost/backend/emit_call.ludic `emit_ns_call` + the
|
|
||||||
`is_<ns>_ns` predicates it delegates to (Math/Text/List)
|
|
||||||
* keywords tools/ludic-tools/ludic_syntax.h LUDIC_KW_* (minus
|
|
||||||
LUDIC_KW_RESERVED, which is not-yet-implemented)
|
|
||||||
* primitive types ludic_syntax.h LUDIC_TYPES
|
|
||||||
* phases ludic_syntax.h LUDIC_PHASES
|
|
||||||
|
|
||||||
Coverage means: the symbol appears in the `tokens:` of some docs/language page.
|
|
||||||
For namespace methods that page is docs/language/<ns>/<ns>-<method>.md with
|
|
||||||
token `<Ns>.<method>`. Run: python3 tools/docgen/check-impl.py
|
|
||||||
"""
|
|
||||||
import os
|
|
||||||
import re
|
|
||||||
import sys
|
|
||||||
|
|
||||||
ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
||||||
LANG = os.path.join(ROOT, "docs", "language")
|
|
||||||
problems = []
|
|
||||||
|
|
||||||
|
|
||||||
def read(*parts):
|
|
||||||
with open(os.path.join(ROOT, *parts), encoding="utf-8") as fh:
|
|
||||||
return fh.read()
|
|
||||||
|
|
||||||
|
|
||||||
def fn_body(text, name):
|
|
||||||
"""The source of `function <name>` up to the next top-level `function ` (or EOF)."""
|
|
||||||
m = re.search(r"^function %s\b" % re.escape(name), text, re.M)
|
|
||||||
if not m:
|
|
||||||
return ""
|
|
||||||
rest = text[m.end():]
|
|
||||||
nxt = re.search(r"^function ", rest, re.M)
|
|
||||||
return text[m.start(): m.end() + (nxt.start() if nxt else len(rest))]
|
|
||||||
|
|
||||||
|
|
||||||
def all_selfhost_source():
|
|
||||||
src = ""
|
|
||||||
d = os.path.join(ROOT, "selfhost")
|
|
||||||
files = []
|
|
||||||
for dirpath, _dirs, names in os.walk(d):
|
|
||||||
for fn in names:
|
|
||||||
if fn.endswith(".ludic"):
|
|
||||||
files.append(os.path.join(dirpath, fn))
|
|
||||||
for path in sorted(files):
|
|
||||||
src += open(path, encoding="utf-8").read() + "\n"
|
|
||||||
return src
|
|
||||||
|
|
||||||
|
|
||||||
def compiler_ns_methods():
|
|
||||||
"""{Namespace: set(method)} the compiler actually dispatches on."""
|
|
||||||
allsrc = all_selfhost_source()
|
|
||||||
body = fn_body(allsrc, "emit_ns_call")
|
|
||||||
if not body:
|
|
||||||
problems.append("emit_ns_call not found in selfhost/ sources")
|
|
||||||
return {}
|
|
||||||
result = {}
|
|
||||||
# each `if (ns == "X")` opens a block that runs to the next such marker
|
|
||||||
chunks = re.split(r'if \(ns == "', body)
|
|
||||||
for chunk in chunks[1:]:
|
|
||||||
mn = re.match(r'(\w+)"', chunk)
|
|
||||||
if not mn:
|
|
||||||
continue
|
|
||||||
ns = mn.group(1)
|
|
||||||
methods = set(re.findall(r'meth == "([A-Za-z_][A-Za-z0-9_]*)"', chunk))
|
|
||||||
if not methods:
|
|
||||||
# delegated form: `if is_<ns>_ns(meth) { return emit_<ns>_ns(...) }`
|
|
||||||
pm = re.search(r"is_(\w+)_ns\(meth\)", chunk)
|
|
||||||
if pm:
|
|
||||||
pred = fn_body(allsrc, "is_%s_ns" % pm.group(1))
|
|
||||||
methods = set(re.findall(r'meth == "([A-Za-z_][A-Za-z0-9_]*)"', pred))
|
|
||||||
if methods:
|
|
||||||
result.setdefault(ns, set()).update(methods)
|
|
||||||
return result
|
|
||||||
|
|
||||||
|
|
||||||
def documented():
|
|
||||||
"""(tokens set, {(ns, member): id} for namespace-method pages)."""
|
|
||||||
tokens = set()
|
|
||||||
nsmethods = {}
|
|
||||||
for cat in sorted(os.listdir(LANG)):
|
|
||||||
cdir = os.path.join(LANG, cat)
|
|
||||||
if not os.path.isdir(cdir):
|
|
||||||
continue
|
|
||||||
for fn in os.listdir(cdir):
|
|
||||||
if not fn.endswith(".md") or fn == "_section.md":
|
|
||||||
continue
|
|
||||||
meta = {}
|
|
||||||
for line in read("docs", "language", cat, fn).split("\n"):
|
|
||||||
m = re.match(r"^(\w+):\s*(.*)$", line)
|
|
||||||
if m:
|
|
||||||
meta[m.group(1)] = m.group(2).strip()
|
|
||||||
if line.strip() == "---" and meta:
|
|
||||||
break
|
|
||||||
tokens.update(meta.get("tokens", "").split())
|
|
||||||
if meta.get("kind") == "namespace-method" and meta.get("ns") and meta.get("member"):
|
|
||||||
nsmethods[(meta["ns"], meta["member"])] = meta.get("id", fn[:-3])
|
|
||||||
return tokens, nsmethods
|
|
||||||
|
|
||||||
|
|
||||||
def vocab_sets():
|
|
||||||
h = read("tools", "ludic-tools", "ludic_syntax.h")
|
|
||||||
|
|
||||||
def clist(name):
|
|
||||||
m = re.search(r"\b" + name + r"\s*\[\s*\]\s*=\s*\{(.*?)\}\s*;", h, re.S)
|
|
||||||
if not m:
|
|
||||||
problems.append("ludic_syntax.h: %s not found" % name)
|
|
||||||
return set()
|
|
||||||
return set(re.findall(r'"([^"]+)"', m.group(1)))
|
|
||||||
|
|
||||||
reserved = clist("LUDIC_KW_RESERVED")
|
|
||||||
keywords = (clist("LUDIC_KW_DECL") | clist("LUDIC_KW_CLAUSE")
|
|
||||||
| clist("LUDIC_KW_STMT")) - reserved
|
|
||||||
return keywords, clist("LUDIC_TYPES"), clist("LUDIC_PHASES")
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
|
||||||
ns_impl = compiler_ns_methods()
|
|
||||||
tokens, ns_doc = documented()
|
|
||||||
|
|
||||||
# 1) every implemented namespace method has a page with the right token
|
|
||||||
for ns, methods in sorted(ns_impl.items()):
|
|
||||||
for meth in sorted(methods):
|
|
||||||
page = os.path.join(LANG, ns.lower(), "%s-%s.md" % (ns.lower(), meth))
|
|
||||||
if (ns, meth) not in ns_doc:
|
|
||||||
problems.append("undocumented %s.%s — add %s (tokens: %s.%s)"
|
|
||||||
% (ns, meth, os.path.relpath(page, ROOT), ns, meth))
|
|
||||||
elif "%s.%s" % (ns, meth) not in tokens:
|
|
||||||
problems.append("%s.%s documented but its page lacks that `tokens:` entry" % (ns, meth))
|
|
||||||
|
|
||||||
# 2) every documented namespace method corresponds to real dispatch
|
|
||||||
for (ns, meth), sid in sorted(ns_doc.items()):
|
|
||||||
if meth not in ns_impl.get(ns, set()):
|
|
||||||
problems.append("stale doc %s: %s.%s is not dispatched by the compiler" % (sid, ns, meth))
|
|
||||||
|
|
||||||
# 3) every implemented keyword / type / phase is documented
|
|
||||||
keywords, types, phases = vocab_sets()
|
|
||||||
for label, names in (("keyword", keywords), ("type", types), ("phase", phases)):
|
|
||||||
for n in sorted(names):
|
|
||||||
if n not in tokens:
|
|
||||||
problems.append("undocumented %s: %s (no docs/language page lists it in `tokens:`)" % (label, n))
|
|
||||||
|
|
||||||
n = sum(len(v) for v in ns_impl.values())
|
|
||||||
if problems:
|
|
||||||
print("docs-vs-implementation drift:", file=sys.stderr)
|
|
||||||
for p in problems:
|
|
||||||
print(" -", p, file=sys.stderr)
|
|
||||||
return 1
|
|
||||||
print("docs cover the implementation: %d namespace methods, %d keywords, %d types, %d phases"
|
|
||||||
% (n, len(keywords), len(types), len(phases)))
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
sys.exit(main())
|
|
||||||
|
|
@ -1,55 +0,0 @@
|
||||||
#!/usr/bin/env python3
|
|
||||||
"""Compile-validate every ```ludic example in the docs against the real ludicc.
|
|
||||||
Full programs (start with program/module) compile directly; fragments are wrapped
|
|
||||||
(as declarations, then as statements) like tools/check-docs.py. Reports failures."""
|
|
||||||
import os, re, subprocess, tempfile, sys
|
|
||||||
REPO=os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))); LC=os.path.join(REPO,"bin","ludicc")
|
|
||||||
FENCE=re.compile(r"```ludic\n(.*?)\n```", re.S)
|
|
||||||
DECL=("program","property","model","handler","enum","function","extern","const","var","ui","event","module","import","scene","machine")
|
|
||||||
|
|
||||||
def first_kw(body):
|
|
||||||
for l in body.split("\n"):
|
|
||||||
s=l.strip()
|
|
||||||
if s and not s.startswith("#"): return s.split("(")[0].split()[0].split("{")[0]
|
|
||||||
return ""
|
|
||||||
|
|
||||||
def candidates(body):
|
|
||||||
kw=first_kw(body)
|
|
||||||
if kw in ("program","module"): return [body]
|
|
||||||
d="program DocCheck {\n"+body+"\n}\n"
|
|
||||||
s="program DocCheck {\n handler DocS phase Start {\n"+body+"\n }\n}\n"
|
|
||||||
return [d,s] if kw in DECL else [s,d]
|
|
||||||
|
|
||||||
def compiles(src, full):
|
|
||||||
with tempfile.NamedTemporaryFile("w",suffix=".ludic",delete=False) as f:
|
|
||||||
f.write(src); tmp=f.name
|
|
||||||
flag=["--emit-llvm","-o",tmp+".ll"] if full else ["--fmt"]
|
|
||||||
r=subprocess.run([LC,tmp]+flag,capture_output=True)
|
|
||||||
os.unlink(tmp)
|
|
||||||
try: os.unlink(tmp+".ll")
|
|
||||||
except OSError: pass
|
|
||||||
return r.returncode==0, (r.stderr or b"").decode("utf-8","replace").strip().split("\n")[0]
|
|
||||||
|
|
||||||
def files():
|
|
||||||
for base in ("docs/language","docs/site/snippets"):
|
|
||||||
for root,_,fs in os.walk(os.path.join(REPO,base)):
|
|
||||||
for fn in fs:
|
|
||||||
if fn.endswith(".md") or fn.endswith(".ludic"): yield os.path.join(root,fn)
|
|
||||||
|
|
||||||
ok=fail=skip=0; fails=[]
|
|
||||||
for p in files():
|
|
||||||
t=open(p).read()
|
|
||||||
blocks=FENCE.findall(t) if p.endswith(".md") else [t]
|
|
||||||
for b in blocks:
|
|
||||||
if 'import "' in b: skip+=1; continue
|
|
||||||
good=False; err=""
|
|
||||||
full = first_kw(b) in ("program","module")
|
|
||||||
for cand in candidates(b):
|
|
||||||
c,e=compiles(cand, full)
|
|
||||||
if c: good=True; break
|
|
||||||
if not err: err=e
|
|
||||||
if good: ok+=1
|
|
||||||
else: fail+=1; fails.append((os.path.relpath(p,REPO),err))
|
|
||||||
print(f"examples: {ok} compile, {fail} fail, {skip} skipped (import)")
|
|
||||||
for p,e in fails: print(f" FAIL {p}: {e[:100]}")
|
|
||||||
sys.exit(1 if fail else 0)
|
|
||||||
|
|
@ -141,7 +141,7 @@ output. The formatter cannot change what a program means.
|
||||||
The vocabulary is written down in five places that cannot include each other —
|
The vocabulary is written down in five places that cannot include each other —
|
||||||
the compiler's two tables, `ludic_syntax.h`, the TextMate grammar (JSON), and the
|
the compiler's two tables, `ludic_syntax.h`, the TextMate grammar (JSON), and the
|
||||||
JetBrains lexer (Kotlin). Adding a builtin and forgetting the rest is silent
|
JetBrains lexer (Kotlin). Adding a builtin and forgetting the rest is silent
|
||||||
failure, so `tools/check-vocabulary.py` compares all five and
|
failure, so `bin/x check-vocabulary` (written in Ludic) compares all five, and
|
||||||
`bin/x test-tools` runs it.
|
`bin/x test-tools` runs it.
|
||||||
|
|
||||||
When you add a keyword or builtin: put it in `ludic_syntax.h`, then run
|
When you add a keyword or builtin: put it in `ludic_syntax.h`, then run
|
||||||
|
|
|
||||||
721
tools/x/checks.ludic
Normal file
721
tools/x/checks.ludic
Normal file
|
|
@ -0,0 +1,721 @@
|
||||||
|
# checks.ludic — the documentation / vocabulary lint suite, in Ludic.
|
||||||
|
#
|
||||||
|
# Ports the Python guards that used to live under tools/ (check-docs.py,
|
||||||
|
# check-vocabulary.py, tools/docgen/check-impl.py) so the doc/lint tooling runs
|
||||||
|
# through `x` with no Python in the loop. Each is a plain `x` subcommand and
|
||||||
|
# reuses the prelude (read_file / shq / capture / the PASS/FAIL harness).
|
||||||
|
#
|
||||||
|
# String work is over NUL-terminated byte buffers (read_file), reached with
|
||||||
|
# plain indexing; `==` on pointers is a byte-string compare (the whole toolchain
|
||||||
|
# leans on this).
|
||||||
|
|
||||||
|
# ---- small string helpers ---------------------------------------------------
|
||||||
|
|
||||||
|
# length of a NUL-terminated buffer
|
||||||
|
function slen(s: pointer) -> int { var n = 0; while s[n] != 0 { n = n + 1 }; return n }
|
||||||
|
|
||||||
|
# a fresh NUL-terminated copy of s[start .. end) (end exclusive)
|
||||||
|
function sslice(s: pointer, start: int, end: int) -> pointer {
|
||||||
|
if end < start { return "" }
|
||||||
|
let n = end - start
|
||||||
|
let b = bytes(n + 1)
|
||||||
|
var i = 0
|
||||||
|
while i < n { b[i] = s[start + i]; i = i + 1 }
|
||||||
|
b[n] = 0
|
||||||
|
return b
|
||||||
|
}
|
||||||
|
|
||||||
|
# index of the first byte of `needle` in `hay` at or after `from`, else -1
|
||||||
|
function s_index(hay: pointer, needle: pointer, from: int) -> int {
|
||||||
|
let hn = slen(hay)
|
||||||
|
let nn = slen(needle)
|
||||||
|
if nn == 0 { return from }
|
||||||
|
var i = from
|
||||||
|
while i + nn <= hn {
|
||||||
|
var j = 0
|
||||||
|
while j < nn and hay[i + j] == needle[j] { j = j + 1 }
|
||||||
|
if j == nn { return i }
|
||||||
|
i = i + 1
|
||||||
|
}
|
||||||
|
return 0 - 1
|
||||||
|
}
|
||||||
|
|
||||||
|
function s_contains(hay: pointer, needle: pointer) -> bool { return s_index(hay, needle, 0) >= 0 }
|
||||||
|
|
||||||
|
# does `hay` contain `needle` exactly at position `at`?
|
||||||
|
function s_starts_at(hay: pointer, at: int, needle: pointer) -> bool {
|
||||||
|
let nn = slen(needle)
|
||||||
|
var i = 0
|
||||||
|
while i < nn { if hay[at + i] != needle[i] { return false }; i = i + 1 }
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
# does `s` (a whole line) begin with `pre`?
|
||||||
|
function s_starts(s: pointer, pre: pointer) -> bool {
|
||||||
|
let pn = slen(pre)
|
||||||
|
var i = 0
|
||||||
|
while i < pn { if s[i] != pre[i] { return false }; i = i + 1 }
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
# is byte c an ASCII space/tab?
|
||||||
|
function is_ws(c: int) -> bool { return c == 32 or c == 9 }
|
||||||
|
|
||||||
|
# the substring from `start` up to the next '\n' (or end)
|
||||||
|
function line_at(s: pointer, start: int) -> pointer {
|
||||||
|
var e = start
|
||||||
|
while s[e] != 0 and s[e] != 10 { e = e + 1 }
|
||||||
|
return sslice(s, start, e)
|
||||||
|
}
|
||||||
|
|
||||||
|
# trim leading/trailing ASCII whitespace (space, tab, cr, nl)
|
||||||
|
function s_trim(s: pointer) -> pointer {
|
||||||
|
let n = slen(s)
|
||||||
|
var a = 0
|
||||||
|
while a < n and (is_ws(s[a]) or s[a] == 10 or s[a] == 13) { a = a + 1 }
|
||||||
|
var b = n
|
||||||
|
while b > a and (is_ws(s[b - 1]) or s[b - 1] == 10 or s[b - 1] == 13) { b = b - 1 }
|
||||||
|
return sslice(s, a, b)
|
||||||
|
}
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# check-docs — every ```ludic fence in the docs must parse (or be marked)
|
||||||
|
# ============================================================================
|
||||||
|
# A fence declares its intent inline (# is a Ludic comment, so the marker is
|
||||||
|
# valid code): `# doc-check: skip` (not checked), `# doc-check: expect-error`
|
||||||
|
# (must FAIL to parse). Everything else must parse under `ludicc --fmt` (lex +
|
||||||
|
# parse, no identifier resolution), after wrapping a bare fragment.
|
||||||
|
|
||||||
|
# the declaration heads that mean "these are top-level decls, wrap in program"
|
||||||
|
function is_decl_head(h: pointer) -> bool {
|
||||||
|
if h == "program" or h == "property" or h == "model" or h == "handler" { return true }
|
||||||
|
if h == "enum" or h == "function" or h == "extern" or h == "const" { return true }
|
||||||
|
if h == "var" or h == "ui" or h == "struct" or h == "import" { return true }
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
# classify a fence body: "whole" (a full program), "decls", or "stmts"
|
||||||
|
function classify_fence(body: pointer) -> pointer {
|
||||||
|
let n = slen(body)
|
||||||
|
var i = 0
|
||||||
|
while i < n {
|
||||||
|
let ln = line_at(body, i)
|
||||||
|
let t = s_trim(ln)
|
||||||
|
if slen(t) > 0 and t[0] != 35 { # non-empty, not a '#' comment
|
||||||
|
# head = first token, split on '(' or whitespace
|
||||||
|
var e = 0
|
||||||
|
let tn = slen(t)
|
||||||
|
while e < tn and t[e] != 40 and not is_ws(t[e]) { e = e + 1 }
|
||||||
|
let head = sslice(t, 0, e)
|
||||||
|
if head == "program" { return "whole" }
|
||||||
|
if is_decl_head(head) { return "decls" }
|
||||||
|
return "stmts"
|
||||||
|
}
|
||||||
|
i = i + slen(ln) + 1
|
||||||
|
}
|
||||||
|
return "stmts"
|
||||||
|
}
|
||||||
|
|
||||||
|
# does the wrapped source at `path` parse? (ludicc --fmt = lex+parse gate)
|
||||||
|
function fence_parses(src: pointer) -> bool {
|
||||||
|
write_file("/tmp/x_docchk.ludic", src)
|
||||||
|
return shq("bin/ludicc /tmp/x_docchk.ludic --fmt > /dev/null 2>&1")
|
||||||
|
}
|
||||||
|
|
||||||
|
# try the candidate framings for a fence; true if any parses
|
||||||
|
function fence_body_parses(body: pointer, kind: pointer) -> bool {
|
||||||
|
if kind == "whole" { return fence_parses(body) }
|
||||||
|
let as_decls = "program DocCheck {\n" + body + "\n}\n"
|
||||||
|
let as_stmts = "program DocCheck {\n handler DocS phase Start {\n" + body + "\n }\n}\n"
|
||||||
|
if kind == "decls" {
|
||||||
|
if fence_parses(as_decls) { return true }
|
||||||
|
return fence_parses(as_stmts)
|
||||||
|
}
|
||||||
|
if fence_parses(as_stmts) { return true }
|
||||||
|
return fence_parses(as_decls)
|
||||||
|
}
|
||||||
|
|
||||||
|
var DC_OK: int = 0
|
||||||
|
var DC_BAD: int = 0
|
||||||
|
var DC_SKIP: int = 0
|
||||||
|
var DC_FAILS: pointer = ""
|
||||||
|
|
||||||
|
# scan one markdown file's ```ludic fences
|
||||||
|
function check_docs_file(path: pointer) -> void {
|
||||||
|
let t = read_file(path)
|
||||||
|
if t == null { return }
|
||||||
|
var i = 0
|
||||||
|
while true {
|
||||||
|
let open = s_index(t, "```ludic\n", i)
|
||||||
|
if open < 0 { break }
|
||||||
|
# require the fence to sit at a line start
|
||||||
|
if open != 0 and t[open - 1] != 10 { i = open + 1; continue }
|
||||||
|
let bstart = open + 9 # past "```ludic\n"
|
||||||
|
let close = s_index(t, "\n```", bstart)
|
||||||
|
if close < 0 { break }
|
||||||
|
let body = sslice(t, bstart, close + 1) # include trailing newline
|
||||||
|
let line = 1 + count_nl(t, open)
|
||||||
|
i = close + 4
|
||||||
|
if s_contains(body, "# doc-check: skip") { DC_SKIP = DC_SKIP + 1; continue }
|
||||||
|
let expect_err = s_contains(body, "# doc-check: expect-error")
|
||||||
|
let kind = classify_fence(body)
|
||||||
|
let parsed = fence_body_parses(body, kind)
|
||||||
|
if parsed == (not expect_err) {
|
||||||
|
DC_OK = DC_OK + 1
|
||||||
|
} else {
|
||||||
|
DC_BAD = DC_BAD + 1
|
||||||
|
if expect_err { DC_FAILS = DC_FAILS + " " + path + ":" + string(line) + " expected to be rejected\n" }
|
||||||
|
else { DC_FAILS = DC_FAILS + " " + path + ":" + string(line) + " expected to parse\n" }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# number of '\n' in s[0 .. upto)
|
||||||
|
function count_nl(s: pointer, upto: int) -> int {
|
||||||
|
var n = 0; var i = 0
|
||||||
|
while i < upto { if s[i] == 10 { n = n + 1 }; i = i + 1 }
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
|
||||||
|
# usage: x check-docs (scans the user-facing docs + docs/language/**)
|
||||||
|
function cmd_check_docs() -> int {
|
||||||
|
ensure_ludicc()
|
||||||
|
DC_OK = 0; DC_BAD = 0; DC_SKIP = 0; DC_FAILS = ""
|
||||||
|
# the corpus: root references + every per-symbol page
|
||||||
|
let list = capture("{ ls LANGUAGE.md COMPILING.md README.md CONTRIBUTING.md 2>/dev/null; find docs -name '*.md' 2>/dev/null; }")
|
||||||
|
let n = slen(list)
|
||||||
|
var i = 0
|
||||||
|
while i < n {
|
||||||
|
let ln = line_at(list, i)
|
||||||
|
i = i + slen(ln) + 1
|
||||||
|
let path = s_trim(ln)
|
||||||
|
if slen(path) > 0 { check_docs_file(path) }
|
||||||
|
}
|
||||||
|
print(` doc fences: {string(DC_OK)} as documented, {string(DC_BAD)} drifted, {string(DC_SKIP)} skipped`)
|
||||||
|
if DC_BAD > 0 { out(DC_FAILS) }
|
||||||
|
if DC_BAD > 0 { return 1 }
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# check-impl — every implemented feature has a docs/language page
|
||||||
|
# ============================================================================
|
||||||
|
# Reads the implementation (emit_ns_call dispatch + the is_<ns>_ns predicates,
|
||||||
|
# and ludic_syntax.h's keyword/type/phase tables) and the per-symbol docs, then
|
||||||
|
# asserts they agree. Replaces the former tools/docgen/check-impl.py.
|
||||||
|
|
||||||
|
# a double-quote byte as a string (the lexer has no \" escape we rely on)
|
||||||
|
function dq() -> pointer { let b = bytes(2); b[0] = 34; b[1] = 0; return b }
|
||||||
|
|
||||||
|
# set-style string list
|
||||||
|
function set_has(l: []pointer, s: pointer) -> bool {
|
||||||
|
var i = 0
|
||||||
|
while i < len(l) { if l[i] == s { return true }; i = i + 1 }
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
function set_add(l: []pointer, s: pointer) -> void { if not set_has(l, s) { push(l, s) } }
|
||||||
|
|
||||||
|
# the body of `function <name>(` up to the next top-level `function ` (or EOF)
|
||||||
|
function fn_body(src: pointer, name: pointer) -> pointer {
|
||||||
|
let needle = "function " + name + "("
|
||||||
|
var p = 0 - 1
|
||||||
|
if s_starts(src, needle) { p = 0 }
|
||||||
|
else {
|
||||||
|
let q = s_index(src, "\n" + needle, 0)
|
||||||
|
if q >= 0 { p = q + 1 }
|
||||||
|
}
|
||||||
|
if p < 0 { return "" }
|
||||||
|
let nxt = s_index(src, "\nfunction ", p + 1)
|
||||||
|
if nxt < 0 { return sslice(src, p, slen(src)) }
|
||||||
|
return sslice(src, p, nxt)
|
||||||
|
}
|
||||||
|
|
||||||
|
# every identifier string following each occurrence of `marker` (read to a quote)
|
||||||
|
function collect_after(src: pointer, marker: pointer) -> []pointer {
|
||||||
|
let out = new []pointer
|
||||||
|
let mn = slen(marker)
|
||||||
|
var i = 0
|
||||||
|
while true {
|
||||||
|
let p = s_index(src, marker, i)
|
||||||
|
if p < 0 { break }
|
||||||
|
var e = p + mn
|
||||||
|
while src[e] != 0 and src[e] != 34 { e = e + 1 }
|
||||||
|
push(out, sslice(src, p + mn, e))
|
||||||
|
i = e + 1
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
# quoted strings inside the `NAME[] = { ... };` C table in `h`
|
||||||
|
function table_set(h: pointer, name: pointer) -> []pointer {
|
||||||
|
let out = new []pointer
|
||||||
|
let p = s_index(h, name + "[]", 0)
|
||||||
|
if p < 0 { return out }
|
||||||
|
let b = s_index(h, "{", p)
|
||||||
|
let e = s_index(h, "};", b)
|
||||||
|
if b < 0 or e < 0 { return out }
|
||||||
|
var i = b
|
||||||
|
while i < e {
|
||||||
|
if h[i] == 34 {
|
||||||
|
var j = i + 1
|
||||||
|
while j < e and h[j] != 34 { j = j + 1 }
|
||||||
|
set_add(out, sslice(h, i + 1, j))
|
||||||
|
i = j + 1
|
||||||
|
} else { i = i + 1 }
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
# concatenate every selfhost/*.ludic into one buffer (via cat, then read)
|
||||||
|
function read_all_selfhost() -> pointer {
|
||||||
|
run("find selfhost -name '*.ludic' | sort | xargs cat > /tmp/x_allsh.txt 2>/dev/null")
|
||||||
|
let s = read_file("/tmp/x_allsh.txt")
|
||||||
|
if s == null { return "" }
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
# {ns}.{method} pairs the compiler actually dispatches on -> into `pairs`
|
||||||
|
function collect_impl_pairs(allsrc: pointer, pairs: []pointer) -> void {
|
||||||
|
let body = fn_body(allsrc, "emit_ns_call")
|
||||||
|
let mns = "if (ns == " + dq()
|
||||||
|
let mmeth = "meth == " + dq()
|
||||||
|
let mnslen = slen(mns)
|
||||||
|
var i = 0
|
||||||
|
while true {
|
||||||
|
let p = s_index(body, mns, i)
|
||||||
|
if p < 0 { break }
|
||||||
|
var ne = p + mnslen
|
||||||
|
while body[ne] != 0 and body[ne] != 34 { ne = ne + 1 }
|
||||||
|
let nsname = sslice(body, p + mnslen, ne)
|
||||||
|
let nxt = s_index(body, mns, ne)
|
||||||
|
var chunkEnd = slen(body)
|
||||||
|
if nxt >= 0 { chunkEnd = nxt }
|
||||||
|
let chunk = sslice(body, p, chunkEnd)
|
||||||
|
var methods = collect_after(chunk, mmeth)
|
||||||
|
if len(methods) == 0 {
|
||||||
|
let dp = s_index(chunk, "is_", 0)
|
||||||
|
if dp >= 0 {
|
||||||
|
var pe = dp
|
||||||
|
while chunk[pe] != 0 and chunk[pe] != 40 { pe = pe + 1 }
|
||||||
|
let predname = sslice(chunk, dp, pe)
|
||||||
|
methods = collect_after(fn_body(allsrc, predname), mmeth)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var k = 0
|
||||||
|
while k < len(methods) { set_add(pairs, nsname + "." + methods[k]); k = k + 1 }
|
||||||
|
if nxt < 0 { break }
|
||||||
|
i = nxt
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# gather the docs facts: token set, and the (ns.member) pairs of ns-method pages
|
||||||
|
function collect_docs(tokens: []pointer, docpairs: []pointer) -> void {
|
||||||
|
let list = capture("find docs/language -name '*.md' ! -name '_section.md' 2>/dev/null")
|
||||||
|
let n = slen(list)
|
||||||
|
var i = 0
|
||||||
|
while i < n {
|
||||||
|
let ln = line_at(list, i)
|
||||||
|
i = i + slen(ln) + 1
|
||||||
|
let path = s_trim(ln)
|
||||||
|
if slen(path) == 0 { continue }
|
||||||
|
let t = read_file(path)
|
||||||
|
if t == null { continue }
|
||||||
|
if not s_starts(t, "---") { continue }
|
||||||
|
let end = s_index(t, "\n---", 3)
|
||||||
|
if end < 0 { continue }
|
||||||
|
let fm = sslice(t, 3, end)
|
||||||
|
# walk front-matter lines
|
||||||
|
var kind = ""; var ns = ""; var member = ""
|
||||||
|
let fn2 = slen(fm)
|
||||||
|
var j = 0
|
||||||
|
while j < fn2 {
|
||||||
|
let fl = line_at(fm, j)
|
||||||
|
j = j + slen(fl) + 1
|
||||||
|
let c = s_index(fl, ":", 0)
|
||||||
|
if c < 0 { continue }
|
||||||
|
let key = s_trim(sslice(fl, 0, c))
|
||||||
|
let val = s_trim(sslice(fl, c + 1, slen(fl)))
|
||||||
|
if key == "tokens" {
|
||||||
|
# split val on spaces
|
||||||
|
var a = 0
|
||||||
|
let vn = slen(val)
|
||||||
|
while a < vn {
|
||||||
|
while a < vn and val[a] == 32 { a = a + 1 }
|
||||||
|
var e = a
|
||||||
|
while e < vn and val[e] != 32 { e = e + 1 }
|
||||||
|
if e > a { set_add(tokens, sslice(val, a, e)) }
|
||||||
|
a = e
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if key == "kind" { kind = val }
|
||||||
|
if key == "ns" { ns = val }
|
||||||
|
if key == "member" { member = val }
|
||||||
|
}
|
||||||
|
if kind == "namespace-method" and slen(ns) > 0 and slen(member) > 0 {
|
||||||
|
set_add(docpairs, ns + "." + member)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
var CI_PROB: pointer = ""
|
||||||
|
var CI_NPROB: int = 0
|
||||||
|
function ci_problem(msg: pointer) -> void { CI_PROB = CI_PROB + " - " + msg + "\n"; CI_NPROB = CI_NPROB + 1 }
|
||||||
|
|
||||||
|
# usage: x check-impl
|
||||||
|
function cmd_check_impl() -> int {
|
||||||
|
CI_PROB = ""; CI_NPROB = 0
|
||||||
|
let allsrc = read_all_selfhost()
|
||||||
|
let impl = new []pointer
|
||||||
|
collect_impl_pairs(allsrc, impl)
|
||||||
|
let tokens = new []pointer
|
||||||
|
let docpairs = new []pointer
|
||||||
|
collect_docs(tokens, docpairs)
|
||||||
|
|
||||||
|
# 1) every implemented ns-method has a page with the right token
|
||||||
|
var i = 0
|
||||||
|
while i < len(impl) {
|
||||||
|
let pair = impl[i]
|
||||||
|
if not set_has(docpairs, pair) { ci_problem("undocumented " + pair + " — add a docs/language page (tokens: " + pair + ")") }
|
||||||
|
else { if not set_has(tokens, pair) { ci_problem(pair + " documented but its page lacks that tokens entry") } }
|
||||||
|
i = i + 1
|
||||||
|
}
|
||||||
|
# 2) every documented ns-method corresponds to real dispatch
|
||||||
|
i = 0
|
||||||
|
while i < len(docpairs) {
|
||||||
|
if not set_has(impl, docpairs[i]) { ci_problem("stale doc: " + docpairs[i] + " is not dispatched by the compiler") }
|
||||||
|
i = i + 1
|
||||||
|
}
|
||||||
|
# 3) every implemented keyword / type / phase is documented
|
||||||
|
let h = read_file("tools/ludic-tools/ludic_syntax.h")
|
||||||
|
let reserved = table_set(h, "LUDIC_KW_RESERVED")
|
||||||
|
let kws = new []pointer
|
||||||
|
add_all_except(kws, table_set(h, "LUDIC_KW_DECL"), reserved)
|
||||||
|
add_all_except(kws, table_set(h, "LUDIC_KW_CLAUSE"), reserved)
|
||||||
|
add_all_except(kws, table_set(h, "LUDIC_KW_STMT"), reserved)
|
||||||
|
let types = table_set(h, "LUDIC_TYPES")
|
||||||
|
let phases = table_set(h, "LUDIC_PHASES")
|
||||||
|
check_all_documented(kws, tokens, "keyword")
|
||||||
|
check_all_documented(types, tokens, "type")
|
||||||
|
check_all_documented(phases, tokens, "phase")
|
||||||
|
|
||||||
|
if CI_NPROB > 0 {
|
||||||
|
err("docs-vs-implementation drift:\n")
|
||||||
|
err(CI_PROB)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
print("docs cover the implementation: " + string(len(impl)) + " namespace methods, " + string(len(kws)) + " keywords, " + string(len(types)) + " types, " + string(len(phases)) + " phases")
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
# add every element of `src` not in `deny` to `dst` (set semantics)
|
||||||
|
function add_all_except(dst: []pointer, src: []pointer, deny: []pointer) -> void {
|
||||||
|
var i = 0
|
||||||
|
while i < len(src) { if not set_has(deny, src[i]) { set_add(dst, src[i]) }; i = i + 1 }
|
||||||
|
}
|
||||||
|
# every name must appear in the token set, else a problem
|
||||||
|
function check_all_documented(names: []pointer, tokens: []pointer, label: pointer) -> void {
|
||||||
|
var i = 0
|
||||||
|
while i < len(names) {
|
||||||
|
if not set_has(tokens, names[i]) { ci_problem("undocumented " + label + ": " + names[i] + " (no docs/language page lists it in tokens)") }
|
||||||
|
i = i + 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# check-vocabulary — the language vocabulary is written down in several places
|
||||||
|
# that cannot include each other (ludic_syntax.h, the JetBrains Kotlin lexer,
|
||||||
|
# the TextMate grammar, the self-host parser); drift between them is silent, so
|
||||||
|
# it is checked. Replaces the former tools/check-vocabulary.py.
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
# a 3-byte needle as a string
|
||||||
|
function bx3(a: int, b: int, c: int) -> pointer { let z = bytes(4); z[0] = a; z[1] = b; z[2] = c; z[3] = 0; return z }
|
||||||
|
|
||||||
|
function all_lower(s: pointer) -> bool {
|
||||||
|
let n = slen(s)
|
||||||
|
if n == 0 { return false }
|
||||||
|
var i = 0
|
||||||
|
while i < n { if s[i] < 97 or s[i] > 122 { return false }; i = i + 1 }
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
# split `s` on '|' into a set
|
||||||
|
function split_pipe(s: pointer) -> []pointer {
|
||||||
|
let out = new []pointer
|
||||||
|
let n = slen(s)
|
||||||
|
var a = 0
|
||||||
|
while a <= n {
|
||||||
|
var e = a
|
||||||
|
while e < n and s[e] != 124 { e = e + 1 }
|
||||||
|
set_add(out, sslice(s, a, e))
|
||||||
|
a = e + 1
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
# first quoted string after each row-opening '{' in the `name[]` C table
|
||||||
|
function c_table_names(h: pointer, name: pointer) -> []pointer {
|
||||||
|
let out = new []pointer
|
||||||
|
let p = s_index(h, name + "[]", 0)
|
||||||
|
if p < 0 { return out }
|
||||||
|
let b = s_index(h, "{", p)
|
||||||
|
let e = s_index(h, "};", b)
|
||||||
|
if b < 0 or e < 0 { return out }
|
||||||
|
var i = b + 1
|
||||||
|
while i < e {
|
||||||
|
if h[i] == 123 { # '{' opens a row
|
||||||
|
var j = i + 1
|
||||||
|
while j < e and (h[j] == 32 or h[j] == 9 or h[j] == 10) { j = j + 1 }
|
||||||
|
if h[j] == 34 {
|
||||||
|
var k = j + 1
|
||||||
|
while k < e and h[k] != 34 { k = k + 1 }
|
||||||
|
set_add(out, sslice(h, j + 1, k))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
i = i + 1
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
# names in a Kotlin `val NAME = setOf( ... )` (parens matched, // comments cut)
|
||||||
|
function kotlin_set(text: pointer, name: pointer) -> []pointer {
|
||||||
|
let out = new []pointer
|
||||||
|
let marker = "val " + name + " = setOf("
|
||||||
|
let p = s_index(text, marker, 0)
|
||||||
|
if p < 0 { return out }
|
||||||
|
var i = p + slen(marker)
|
||||||
|
var depth = 1
|
||||||
|
let region_start = i
|
||||||
|
let tn = slen(text)
|
||||||
|
while i < tn and depth > 0 {
|
||||||
|
if text[i] == 40 { depth = depth + 1 }
|
||||||
|
if text[i] == 41 { depth = depth - 1 }
|
||||||
|
if depth == 0 { break }
|
||||||
|
i = i + 1
|
||||||
|
}
|
||||||
|
# collect quoted strings in [region_start, i), skipping //-comments
|
||||||
|
var k = region_start
|
||||||
|
while k < i {
|
||||||
|
if text[k] == 47 and text[k + 1] == 47 { # '//'
|
||||||
|
while k < i and text[k] != 10 { k = k + 1 }
|
||||||
|
} else { if text[k] == 34 {
|
||||||
|
var e = k + 1
|
||||||
|
while e < i and text[e] != 34 { e = e + 1 }
|
||||||
|
set_add(out, sslice(text, k + 1, e))
|
||||||
|
k = e + 1
|
||||||
|
} else { k = k + 1 } }
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
# the alternation inside repository.<node>.patterns' first match containing marker
|
||||||
|
function grammar_alt(root: JVal, node: pointer, marker: pointer) -> []pointer {
|
||||||
|
let out = new []pointer
|
||||||
|
let pats = j_get(j_get(j_get(root, "repository"), node), "patterns")
|
||||||
|
if pats.t != JV_ARR { return out }
|
||||||
|
var i = 0
|
||||||
|
while i < len(pats.kids) {
|
||||||
|
let m = j_get(pats.kids[i], "match")
|
||||||
|
if m.t == JV_STR and s_contains(m.s, marker) {
|
||||||
|
let op = s_index(m.s, bx3(92, 98, 40), 0) # \b(
|
||||||
|
if op >= 0 {
|
||||||
|
let cl = s_index(m.s, bx3(41, 92, 98), op) # )\b
|
||||||
|
if cl >= 0 { return split_pipe(sslice(m.s, op + 3, cl)) }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
i = i + 1
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
# keywords the self-host parser dispatches on: is_id("x") + streq(t.text, "x")
|
||||||
|
function parser_keywords() -> []pointer {
|
||||||
|
let out = new []pointer
|
||||||
|
add_parser_kw(out, "selfhost/frontend/parse.ludic")
|
||||||
|
add_parser_kw(out, "selfhost/frontend/parse_game.ludic")
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
function add_parser_kw(out: []pointer, path: pointer) -> void {
|
||||||
|
let t = read_file(path)
|
||||||
|
if t == null { return }
|
||||||
|
add_lower(out, collect_after(t, "is_id(" + dq()))
|
||||||
|
add_lower(out, collect_after(t, "streq(t.text, " + dq()))
|
||||||
|
}
|
||||||
|
function add_lower(out: []pointer, src: []pointer) -> void {
|
||||||
|
var i = 0
|
||||||
|
while i < len(src) { if all_lower(src[i]) { set_add(out, src[i]) }; i = i + 1 }
|
||||||
|
}
|
||||||
|
|
||||||
|
var CV_PROB: pointer = ""
|
||||||
|
var CV_N: int = 0
|
||||||
|
function cv_problem(msg: pointer) -> void { CV_PROB = CV_PROB + " - " + msg + "\n"; CV_N = CV_N + 1 }
|
||||||
|
|
||||||
|
# report names in `ref` missing from `other`, and names in `other` not in `ref`
|
||||||
|
function cmp_sets(label: pointer, ref: []pointer, other: []pointer, other_label: pointer) -> void {
|
||||||
|
var i = 0
|
||||||
|
while i < len(ref) { if not set_has(other, ref[i]) { cv_problem(other_label + " is missing " + label + ": " + ref[i]) }; i = i + 1 }
|
||||||
|
i = 0
|
||||||
|
while i < len(other) { if not set_has(ref, other[i]) { cv_problem(other_label + " has unknown " + label + ": " + other[i]) }; i = i + 1 }
|
||||||
|
}
|
||||||
|
|
||||||
|
# union of two sets
|
||||||
|
function set_union(a: []pointer, b: []pointer) -> []pointer {
|
||||||
|
let out = new []pointer
|
||||||
|
var i = 0
|
||||||
|
while i < len(a) { set_add(out, a[i]); i = i + 1 }
|
||||||
|
i = 0
|
||||||
|
while i < len(b) { set_add(out, b[i]); i = i + 1 }
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
# usage: x check-vocabulary
|
||||||
|
function cmd_check_vocab() -> int {
|
||||||
|
CV_PROB = ""; CV_N = 0
|
||||||
|
let h = read_file("tools/ludic-tools/ludic_syntax.h")
|
||||||
|
if h == null { err("check-vocabulary: ludic_syntax.h missing\n"); return 2 }
|
||||||
|
let h_decl = table_set(h, "LUDIC_KW_DECL")
|
||||||
|
let h_clause = table_set(h, "LUDIC_KW_CLAUSE")
|
||||||
|
let h_stmt = table_set(h, "LUDIC_KW_STMT")
|
||||||
|
let h_types = table_set(h, "LUDIC_TYPES")
|
||||||
|
let h_phases = table_set(h, "LUDIC_PHASES")
|
||||||
|
let h_widgets = table_set(h, "LUDIC_WIDGETS")
|
||||||
|
let h_builtins = c_table_names(h, "LUDIC_BUILTINS")
|
||||||
|
let h_intrinsics = c_table_names(h, "LUDIC_INTRINSICS")
|
||||||
|
let h_reserved = table_set(h, "LUDIC_KW_RESERVED")
|
||||||
|
|
||||||
|
# --- against the JetBrains lexer ---
|
||||||
|
let kt = read_file("tools/editors/jetbrains/src/main/kotlin/io/ludic/ide/LudicTokens.kt")
|
||||||
|
if kt != null {
|
||||||
|
cmp_sets("declaration keywords", h_decl, kotlin_set(kt, "DECL"), "LudicTokens.kt")
|
||||||
|
cmp_sets("clause keywords", h_clause, kotlin_set(kt, "CLAUSE"), "LudicTokens.kt")
|
||||||
|
cmp_sets("statement keywords", h_stmt, kotlin_set(kt, "STMT"), "LudicTokens.kt")
|
||||||
|
cmp_sets("primitive types", h_types, kotlin_set(kt, "PRIMITIVES"), "LudicTokens.kt")
|
||||||
|
cmp_sets("phases", h_phases, kotlin_set(kt, "PHASES"), "LudicTokens.kt")
|
||||||
|
cmp_sets("widgets", h_widgets, kotlin_set(kt, "WIDGETS"), "LudicTokens.kt")
|
||||||
|
cmp_sets("builtins", set_union(h_builtins, h_intrinsics), kotlin_set(kt, "BUILTINS"), "LudicTokens.kt")
|
||||||
|
}
|
||||||
|
|
||||||
|
# --- against the TextMate grammar ---
|
||||||
|
let gt = read_file("tools/editors/shared/ludic.tmLanguage.json")
|
||||||
|
if gt != null {
|
||||||
|
let g = json_parse(gt)
|
||||||
|
cmp_sets("builtins", h_builtins, grammar_alt(g, "builtin", "rng_chance"), "ludic.tmLanguage.json")
|
||||||
|
cmp_sets("intrinsics", h_intrinsics, grammar_alt(g, "builtin", "as_fixed"), "ludic.tmLanguage.json")
|
||||||
|
cmp_sets("phases", h_phases, grammar_alt(g, "keyword", "FixedUpdate"), "ludic.tmLanguage.json")
|
||||||
|
cmp_sets("primitive types", h_types, grammar_alt(g, "keyword", "fixed"), "ludic.tmLanguage.json")
|
||||||
|
}
|
||||||
|
|
||||||
|
# --- against the self-host parser ---
|
||||||
|
let pkw = parser_keywords()
|
||||||
|
if len(pkw) == 0 {
|
||||||
|
cv_problem("could not extract any keywords from selfhost/frontend/parse*.ludic")
|
||||||
|
} else {
|
||||||
|
cmp_unparsed("declaration keywords", h_decl, pkw, h_reserved)
|
||||||
|
cmp_unparsed("clause keywords", h_clause, pkw, h_reserved)
|
||||||
|
# reserved words the parser now accepts should be promoted
|
||||||
|
var i = 0
|
||||||
|
while i < len(h_reserved) { if set_has(pkw, h_reserved[i]) { cv_problem("LUDIC_KW_RESERVED lists a keyword the parser now accepts — promote it: " + h_reserved[i]) }; i = i + 1 }
|
||||||
|
}
|
||||||
|
|
||||||
|
if CV_N > 0 {
|
||||||
|
err("vocabulary drift:\n")
|
||||||
|
err(CV_PROB)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
# keywords in `kws` the parser never dispatches on (minus reserved) are a problem
|
||||||
|
function cmp_unparsed(label: pointer, kws: []pointer, pkw: []pointer, reserved: []pointer) -> void {
|
||||||
|
var i = 0
|
||||||
|
while i < len(kws) {
|
||||||
|
if not set_has(pkw, kws[i]) {
|
||||||
|
if not set_has(reserved, kws[i]) { cv_problem("ludic_syntax.h lists a " + label + " the selfhost parser never dispatches on: " + kws[i]) }
|
||||||
|
}
|
||||||
|
i = i + 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# json/xml asset validation — replaces the python3 json.load / xml.dom checks
|
||||||
|
# the editor-toolchain suite used on the shared grammar + plugin assets.
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
# minimal XML well-formedness: tags balance and nest, quotes respected.
|
||||||
|
function xml_valid(text: pointer) -> bool {
|
||||||
|
let n = slen(text)
|
||||||
|
let stack = new []pointer
|
||||||
|
var sp = 0
|
||||||
|
var i = 0
|
||||||
|
while i < n {
|
||||||
|
if text[i] != 60 { i = i + 1; continue } # seek '<'
|
||||||
|
if text[i + 1] == 63 { # <? ... ?>
|
||||||
|
let e = s_index(text, "?>", i)
|
||||||
|
if e < 0 { return false }
|
||||||
|
i = e + 2; continue
|
||||||
|
}
|
||||||
|
if text[i + 1] == 33 { # <! comment / cdata / doctype
|
||||||
|
if text[i + 2] == 45 and text[i + 3] == 45 { # <!-- ... -->
|
||||||
|
let e = s_index(text, "-->", i)
|
||||||
|
if e < 0 { return false }
|
||||||
|
i = e + 3; continue
|
||||||
|
}
|
||||||
|
if s_starts_at(text, i, "<![CDATA[") { # <![CDATA[ ... ]]>
|
||||||
|
let e = s_index(text, "]]>", i)
|
||||||
|
if e < 0 { return false }
|
||||||
|
i = e + 3; continue
|
||||||
|
}
|
||||||
|
let e = s_index(text, ">", i)
|
||||||
|
if e < 0 { return false }
|
||||||
|
i = e + 1; continue
|
||||||
|
}
|
||||||
|
if text[i + 1] == 47 { # closing </name>
|
||||||
|
var j = i + 2
|
||||||
|
while j < n and text[j] != 62 and text[j] != 32 and text[j] != 9 and text[j] != 10 { j = j + 1 }
|
||||||
|
let name = sslice(text, i + 2, j)
|
||||||
|
let e = s_index(text, ">", j)
|
||||||
|
if e < 0 { return false }
|
||||||
|
if sp == 0 { return false }
|
||||||
|
if stack[sp - 1] != name { return false }
|
||||||
|
sp = sp - 1
|
||||||
|
i = e + 1; continue
|
||||||
|
}
|
||||||
|
# opening tag: read the name
|
||||||
|
var j = i + 1
|
||||||
|
while j < n and text[j] != 62 and text[j] != 32 and text[j] != 9 and text[j] != 10 and text[j] != 47 { j = j + 1 }
|
||||||
|
let name = sslice(text, i + 1, j)
|
||||||
|
if slen(name) == 0 { return false }
|
||||||
|
# advance to '>', skipping quoted attribute values
|
||||||
|
var k = j
|
||||||
|
while k < n and text[k] != 62 {
|
||||||
|
if text[k] == 34 { k = k + 1; while k < n and text[k] != 34 { k = k + 1 } }
|
||||||
|
else { if text[k] == 39 { k = k + 1; while k < n and text[k] != 39 { k = k + 1 } } }
|
||||||
|
k = k + 1
|
||||||
|
}
|
||||||
|
if k >= n { return false }
|
||||||
|
if text[k - 1] != 47 { # not self-closing -> push
|
||||||
|
if sp < len(stack) { stack[sp] = name } else { push(stack, name) }
|
||||||
|
sp = sp + 1
|
||||||
|
}
|
||||||
|
i = k + 1
|
||||||
|
}
|
||||||
|
return sp == 0
|
||||||
|
}
|
||||||
|
|
||||||
|
# usage: x lint-asset <file> (validates one .json or .xml editor asset)
|
||||||
|
function cmd_lint_asset() -> int {
|
||||||
|
if arg_count() < 3 { err("usage: x lint-asset <file.json|file.xml>\n"); return 2 }
|
||||||
|
let path = arg(2)
|
||||||
|
let t = read_file(path)
|
||||||
|
if t == null { err("lint-asset: cannot read " + path + "\n"); return 1 }
|
||||||
|
if s_index(path, ".json", 0) >= 0 {
|
||||||
|
if json_valid(t) { return 0 }
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if s_index(path, ".xml", 0) >= 0 {
|
||||||
|
if xml_valid(t) { return 0 }
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
err("lint-asset: unknown asset type " + path + "\n")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
174
tools/x/json.ludic
Normal file
174
tools/x/json.ludic
Normal file
|
|
@ -0,0 +1,174 @@
|
||||||
|
# json.ludic — a small JSON reader for the Ludic tooling (vocabulary check,
|
||||||
|
# docs generator). Enough of JSON to read the editor grammar and the docs
|
||||||
|
# config: objects, arrays, strings (with \uXXXX + surrogates), numbers, the
|
||||||
|
# literals. Values are a JVal tree; object member order is preserved.
|
||||||
|
|
||||||
|
property JVal { t: int = 0, num: int = 0, b: int = 0, s: pointer = null, kids: []JVal, keys: []pointer }
|
||||||
|
|
||||||
|
const JV_NULL: int = 0
|
||||||
|
const JV_BOOL: int = 1
|
||||||
|
const JV_NUM: int = 2
|
||||||
|
const JV_STR: int = 3
|
||||||
|
const JV_ARR: int = 4
|
||||||
|
const JV_OBJ: int = 5
|
||||||
|
|
||||||
|
function jval_new(t: int) -> JVal {
|
||||||
|
let v = new JVal
|
||||||
|
v.t = t
|
||||||
|
v.s = ""
|
||||||
|
v.kids = new []JVal
|
||||||
|
v.keys = new []pointer
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
var jbuf: pointer = ""
|
||||||
|
var jpos: int = 0
|
||||||
|
var jerr: bool = false
|
||||||
|
|
||||||
|
function j_ws() -> void {
|
||||||
|
while jbuf[jpos] == 32 or jbuf[jpos] == 9 or jbuf[jpos] == 10 or jbuf[jpos] == 13 { jpos = jpos + 1 }
|
||||||
|
}
|
||||||
|
|
||||||
|
function j_hex(c: int) -> int {
|
||||||
|
if c >= 48 and c <= 57 { return c - 48 }
|
||||||
|
if c >= 97 and c <= 102 { return c - 97 + 10 }
|
||||||
|
if c >= 65 and c <= 70 { return c - 65 + 10 }
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
function j_hex4() -> int {
|
||||||
|
var v = 0; var i = 0
|
||||||
|
while i < 4 { v = v * 16 + j_hex(jbuf[jpos]); jpos = jpos + 1; i = i + 1 }
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
# append codepoint cp to buffer b at *n (UTF-8); returns new n
|
||||||
|
function utf8_put(b: pointer, n: int, cp: int) -> int {
|
||||||
|
if cp < 128 { b[n] = cp; return n + 1 }
|
||||||
|
if cp < 2048 { b[n] = 192 + (cp >> 6); b[n + 1] = 128 + (cp & 63); return n + 2 }
|
||||||
|
if cp < 65536 { b[n] = 224 + (cp >> 12); b[n + 1] = 128 + ((cp >> 6) & 63); b[n + 2] = 128 + (cp & 63); return n + 3 }
|
||||||
|
b[n] = 240 + (cp >> 18); b[n + 1] = 128 + ((cp >> 12) & 63); b[n + 2] = 128 + ((cp >> 6) & 63); b[n + 3] = 128 + (cp & 63)
|
||||||
|
return n + 4
|
||||||
|
}
|
||||||
|
|
||||||
|
# parse a JSON string starting at the opening quote; returns a fresh buffer
|
||||||
|
function j_string() -> pointer {
|
||||||
|
jpos = jpos + 1 # opening quote
|
||||||
|
# over-allocate: escapes only ever shrink the byte count
|
||||||
|
let cap = slen(jbuf) - jpos + 1
|
||||||
|
let out = bytes(cap + 4)
|
||||||
|
var n = 0
|
||||||
|
while jbuf[jpos] != 0 and jbuf[jpos] != 34 {
|
||||||
|
if jbuf[jpos] == 92 {
|
||||||
|
jpos = jpos + 1
|
||||||
|
let c = jbuf[jpos]; jpos = jpos + 1
|
||||||
|
if c == 110 { out[n] = 10; n = n + 1 }
|
||||||
|
else { if c == 116 { out[n] = 9; n = n + 1 }
|
||||||
|
else { if c == 114 { out[n] = 13; n = n + 1 }
|
||||||
|
else { if c == 98 { out[n] = 8; n = n + 1 }
|
||||||
|
else { if c == 102 { out[n] = 12; n = n + 1 }
|
||||||
|
else { if c == 117 {
|
||||||
|
var cp = j_hex4()
|
||||||
|
if cp >= 55296 and cp < 56320 and jbuf[jpos] == 92 and jbuf[jpos + 1] == 117 {
|
||||||
|
jpos = jpos + 2
|
||||||
|
let lo = j_hex4()
|
||||||
|
cp = 65536 + ((cp - 55296) << 10) + (lo - 56320)
|
||||||
|
}
|
||||||
|
n = utf8_put(out, n, cp)
|
||||||
|
} else { out[n] = c; n = n + 1 } } } } } }
|
||||||
|
} else { out[n] = jbuf[jpos]; n = n + 1; jpos = jpos + 1 }
|
||||||
|
}
|
||||||
|
if jbuf[jpos] == 34 { jpos = jpos + 1 }
|
||||||
|
out[n] = 0
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
function j_value() -> JVal {
|
||||||
|
j_ws()
|
||||||
|
let c = jbuf[jpos]
|
||||||
|
if c == 123 { # '{'
|
||||||
|
jpos = jpos + 1
|
||||||
|
let v = jval_new(JV_OBJ)
|
||||||
|
j_ws()
|
||||||
|
if jbuf[jpos] == 125 { jpos = jpos + 1; return v }
|
||||||
|
while true {
|
||||||
|
j_ws()
|
||||||
|
var key = ""
|
||||||
|
if jbuf[jpos] == 34 { key = j_string() } else { jerr = true }
|
||||||
|
j_ws(); if jbuf[jpos] == 58 { jpos = jpos + 1 } else { jerr = true }
|
||||||
|
let kid = j_value()
|
||||||
|
push(v.keys, key); push(v.kids, kid)
|
||||||
|
j_ws()
|
||||||
|
if jbuf[jpos] == 44 { jpos = jpos + 1 } else { if jbuf[jpos] == 125 { jpos = jpos + 1; return v } else { return v } }
|
||||||
|
}
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
if c == 91 { # '['
|
||||||
|
jpos = jpos + 1
|
||||||
|
let v = jval_new(JV_ARR)
|
||||||
|
j_ws()
|
||||||
|
if jbuf[jpos] == 93 { jpos = jpos + 1; return v }
|
||||||
|
while true {
|
||||||
|
let kid = j_value()
|
||||||
|
push(v.kids, kid)
|
||||||
|
j_ws()
|
||||||
|
if jbuf[jpos] == 44 { jpos = jpos + 1 } else { if jbuf[jpos] == 93 { jpos = jpos + 1; return v } else { return v } }
|
||||||
|
}
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
if c == 34 { # '"'
|
||||||
|
let v = jval_new(JV_STR)
|
||||||
|
v.s = j_string()
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
if c == 116 { # true
|
||||||
|
jpos = jpos + 4; let v = jval_new(JV_BOOL); v.b = 1; return v
|
||||||
|
}
|
||||||
|
if c == 102 { # false
|
||||||
|
jpos = jpos + 5; let v = jval_new(JV_BOOL); v.b = 0; return v
|
||||||
|
}
|
||||||
|
if c == 110 { # null
|
||||||
|
jpos = jpos + 4; return jval_new(JV_NULL)
|
||||||
|
}
|
||||||
|
# number
|
||||||
|
let start = jpos
|
||||||
|
if jbuf[jpos] == 45 { jpos = jpos + 1 }
|
||||||
|
if jbuf[jpos] < 48 or jbuf[jpos] > 57 { jerr = true; if jbuf[jpos] != 0 { jpos = jpos + 1 }; return jval_new(JV_NULL) }
|
||||||
|
while (jbuf[jpos] >= 48 and jbuf[jpos] <= 57) or jbuf[jpos] == 46 or jbuf[jpos] == 101 or jbuf[jpos] == 69 or jbuf[jpos] == 43 or jbuf[jpos] == 45 { jpos = jpos + 1 }
|
||||||
|
let v = jval_new(JV_NUM)
|
||||||
|
v.s = sslice(jbuf, start, jpos)
|
||||||
|
v.num = str_to_int(v.s)
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
# parse whole `text`; returns the root JVal (JV_NULL on empty)
|
||||||
|
function json_parse(text: pointer) -> JVal {
|
||||||
|
jbuf = text; jpos = 0; jerr = false
|
||||||
|
return j_value()
|
||||||
|
}
|
||||||
|
|
||||||
|
# is `text` well-formed JSON that consumes to the end?
|
||||||
|
function json_valid(text: pointer) -> bool {
|
||||||
|
let v = json_parse(text)
|
||||||
|
j_ws()
|
||||||
|
if jerr { return false }
|
||||||
|
if jpos != slen(text) { return false }
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
# integer value of a decimal string (ignores any fractional part)
|
||||||
|
function str_to_int(s: pointer) -> int {
|
||||||
|
var i = 0; var neg = false
|
||||||
|
if s[0] == 45 { neg = true; i = 1 }
|
||||||
|
var v = 0
|
||||||
|
while s[i] >= 48 and s[i] <= 57 { v = v * 10 + (s[i] - 48); i = i + 1 }
|
||||||
|
if neg { return 0 - v }
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
# object member lookup by key (JV_NULL if absent or not an object)
|
||||||
|
function j_get(v: JVal, key: pointer) -> JVal {
|
||||||
|
if v.t != JV_OBJ { return jval_new(JV_NULL) }
|
||||||
|
var i = 0
|
||||||
|
while i < len(v.keys) { if v.keys[i] == key { return v.kids[i] }; i = i + 1 }
|
||||||
|
return jval_new(JV_NULL)
|
||||||
|
}
|
||||||
|
|
@ -17,6 +17,8 @@ program X {
|
||||||
import "tools.ludic"
|
import "tools.ludic"
|
||||||
import "selfhost_test.ludic"
|
import "selfhost_test.ludic"
|
||||||
import "test.ludic"
|
import "test.ludic"
|
||||||
|
import "json.ludic"
|
||||||
|
import "checks.ludic"
|
||||||
import "release.ludic"
|
import "release.ludic"
|
||||||
|
|
||||||
function usage() -> void {
|
function usage() -> void {
|
||||||
|
|
@ -36,6 +38,12 @@ program X {
|
||||||
print(" x test-tools the editor-toolchain suite")
|
print(" x test-tools the editor-toolchain suite")
|
||||||
print(" x golden regenerate selfhost/golden/renders.sha256 (review with git diff)")
|
print(" x golden regenerate selfhost/golden/renders.sha256 (review with git diff)")
|
||||||
print("")
|
print("")
|
||||||
|
print("doc / lint checks (Ludic, no Python):")
|
||||||
|
print(" x check-docs every ```ludic doc fence parses (or is marked skip/expect-error)")
|
||||||
|
print(" x check-impl every implemented feature has a docs/language page")
|
||||||
|
print(" x check-vocabulary the vocabulary is in sync across grammar / lexer / header / parser")
|
||||||
|
print(" x lint-asset <file> validate one editor .json / .xml asset")
|
||||||
|
print("")
|
||||||
print("release:")
|
print("release:")
|
||||||
print(" x version print the toolchain version (ludicc --version)")
|
print(" x version print the toolchain version (ludicc --version)")
|
||||||
print(" x release [major|minor|patch] [--publish]")
|
print(" x release [major|minor|patch] [--publish]")
|
||||||
|
|
@ -68,6 +76,10 @@ program X {
|
||||||
if (cmd == "test") { exit(cmd_test()) }
|
if (cmd == "test") { exit(cmd_test()) }
|
||||||
if (cmd == "selfhost-test") { exit(cmd_selfhost_test()) }
|
if (cmd == "selfhost-test") { exit(cmd_selfhost_test()) }
|
||||||
if (cmd == "test-tools") { exit(cmd_test_tools()) }
|
if (cmd == "test-tools") { exit(cmd_test_tools()) }
|
||||||
|
if (cmd == "check-docs") { exit(cmd_check_docs()) }
|
||||||
|
if (cmd == "check-impl") { exit(cmd_check_impl()) }
|
||||||
|
if (cmd == "check-vocabulary") { exit(cmd_check_vocab()) }
|
||||||
|
if (cmd == "lint-asset") { exit(cmd_lint_asset()) }
|
||||||
if (cmd == "golden") { exit(cmd_golden()) }
|
if (cmd == "golden") { exit(cmd_golden()) }
|
||||||
if (cmd == "bootstrap") { exit(cmd_bootstrap()) }
|
if (cmd == "bootstrap") { exit(cmd_bootstrap()) }
|
||||||
if (cmd == "bootstrap-cfree") { exit(cmd_bootstrap_cfree()) }
|
if (cmd == "bootstrap-cfree") { exit(cmd_bootstrap_cfree()) }
|
||||||
|
|
|
||||||
|
|
@ -168,25 +168,25 @@ function cmd_test_tools() -> int {
|
||||||
|
|
||||||
# the vocabulary lives in one place; drift between it and its copies (the
|
# the vocabulary lives in one place; drift between it and its copies (the
|
||||||
# TextMate grammar, the Kotlin lexer) is the failure mode this layout prevents.
|
# TextMate grammar, the Kotlin lexer) is the failure mode this layout prevents.
|
||||||
if shq("python3 tools/check-vocabulary.py") { ok("vocabulary in sync across grammar/lexer/header") } else { bad("vocabulary drifted") }
|
# The check is written in Ludic (tools/x/checks.ludic) and runs through x — no
|
||||||
|
# Python in the loop.
|
||||||
|
if shq("bin/x check-vocabulary 2>/dev/null") { ok("vocabulary in sync across grammar/lexer/header") } else { bad("vocabulary drifted") }
|
||||||
|
|
||||||
# every feature the compiler actually implements — namespace methods, keywords,
|
# every feature the compiler actually implements — namespace methods, keywords,
|
||||||
# types, phases — must have a docs/language page. This reads the implementation
|
# types, phases — must have a docs/language page. This reads the implementation
|
||||||
# (emit_ns_call + ludic_syntax.h), so shipping a feature without docs fails here.
|
# (emit_ns_call + ludic_syntax.h), so shipping a feature without docs fails here.
|
||||||
if shq("python3 tools/docgen/check-impl.py") { ok("docs cover every implemented feature") } else { bad("docs drifted from the implementation") }
|
# Ported to Ludic; runs through x.
|
||||||
|
if shq("bin/x check-impl > /dev/null 2>&1") { ok("docs cover every implemented feature") } else { bad("docs drifted from the implementation") }
|
||||||
|
|
||||||
return report()
|
return report()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# JSON/XML validity, checked by the Ludic validators in checks.ludic (no Python).
|
||||||
function test_json(path: pointer) -> void {
|
function test_json(path: pointer) -> void {
|
||||||
let bn = capture_line(`basename {path}`)
|
let bn = capture_line(`basename {path}`)
|
||||||
if shq(`python3 -c 'import json,sys;json.load(open(sys.argv[1]))' {path} 2>/dev/null`) {
|
if shq(`bin/x lint-asset {path}`) { ok(`valid JSON: {bn}`) } else { bad(`invalid JSON: {path}`) }
|
||||||
ok(`valid JSON: {bn}`)
|
|
||||||
} else { bad(`invalid JSON: {path}`) }
|
|
||||||
}
|
}
|
||||||
function test_xml(path: pointer) -> void {
|
function test_xml(path: pointer) -> void {
|
||||||
let bn = capture_line(`basename {path}`)
|
let bn = capture_line(`basename {path}`)
|
||||||
if shq(`python3 -c 'import xml.dom.minidom,sys;xml.dom.minidom.parse(sys.argv[1])' {path} 2>/dev/null`) {
|
if shq(`bin/x lint-asset {path}`) { ok(`valid XML: {bn}`) } else { bad(`invalid XML: {path}`) }
|
||||||
ok(`valid XML: {bn}`)
|
|
||||||
} else { bad(`invalid XML: {path}`) }
|
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue