Documentation namespace cleanup (issue #38). Audit outcome: - `date` vs `datetime` are NOT duplicates — `Date` is calendar days since the epoch, `DateTime` is instants (seconds); distinct runtime namespaces. Kept both. - `network` vs `networking` WAS a real duplicate. Every other stdlib area documents only its namespace (`World.*`, `Screen.*`, …), never the bare builtins it lowers to. Networking alone also documented the low-level `net_*`/builtin forms under `networking/`, duplicating the `Network.*` pages under `network/`. Removed `networking/`; `network/` (the `Network` namespace, which the compiler and LSP both expose) is canonical. Folded the `@Sync`/`@Owned` framing into `network/_section.md` so no context is lost. - Dropped the `networking` key from docgen inventory.json. Guard (AC3): `tools/docgen/check.py` now fails if any `ns:` is documented from more than one directory, or if two sections share an id or (case-folded) title — so a duplicate-namespace split cannot silently reappear. `gen.py` + `check.py` pass (34 sections, 365 symbols). Closes #38 Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
148 lines
6.4 KiB
Python
148 lines
6.4 KiB
Python
#!/usr/bin/env python3
|
|
"""check.py — coverage + integrity guard for the generated docs site.
|
|
|
|
Fails (exit 1) if:
|
|
* the pages contract is broken (index.html / api.html / .nojekyll missing);
|
|
* any symbol in tools/docgen/inventory.json lacks a source file AND a page;
|
|
* any per-symbol source file still carries only its one-line seed (i.e. was
|
|
scaffolded but never written up) — so "every symbol is really documented";
|
|
* a highlighter link target (page) does not exist in the output.
|
|
|
|
Usage: python3 tools/docgen/check.py [site-dir] (default build/pages)
|
|
"""
|
|
import json, os, re, sys
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
LANG = os.path.join(ROOT, "docs", "language")
|
|
DOCGEN = os.path.join(ROOT, "tools", "docgen")
|
|
|
|
def parse_front(path):
|
|
t = open(path, encoding="utf-8").read()
|
|
meta = {}
|
|
if t.startswith("---"):
|
|
end = t.find("\n---", 3)
|
|
if end != -1:
|
|
for line in t[3:end].strip("\n").split("\n"):
|
|
if line.strip() and ":" in line:
|
|
k, v = line.split(":", 1); meta[k.strip()] = v.strip()
|
|
body = t[end+4:].strip()
|
|
else:
|
|
body = t
|
|
else:
|
|
body = t
|
|
return meta, body
|
|
|
|
def main(site):
|
|
problems, warnings = [], []
|
|
|
|
# 1) pages contract
|
|
for f in ("index.html", "api.html", ".nojekyll"):
|
|
if not os.path.exists(os.path.join(site, f)):
|
|
problems.append("missing required file: " + f)
|
|
|
|
# 2) coverage against the authoritative inventory
|
|
inv = json.load(open(os.path.join(DOCGEN, "inventory.json")))
|
|
# map id -> source path
|
|
id2src = {}
|
|
thin = []
|
|
for cat in sorted(os.listdir(LANG)):
|
|
cdir = os.path.join(LANG, cat)
|
|
if not os.path.isdir(cdir):
|
|
continue
|
|
for fn in os.listdir(cdir):
|
|
if not fn.endswith(".md") or fn == "_section.md":
|
|
continue
|
|
meta, body = parse_front(os.path.join(cdir, fn))
|
|
sid = meta.get("id", fn[:-3]); id2src[sid] = os.path.join(cdir, fn)
|
|
# a "thin" file = body (minus a Parameters block and its tip line) too short
|
|
tipline = meta.get("tip", "")
|
|
btext = re.sub(r"(?is)parameters:.*", "", body).strip()
|
|
btext = btext.replace(tipline, "").strip()
|
|
if len(btext) < 40:
|
|
thin.append(sid)
|
|
for cat, ids in inv.items():
|
|
for sid in ids:
|
|
if sid not in id2src:
|
|
problems.append("no source file for inventory symbol: %s (%s)" % (sid, cat))
|
|
elif site and not os.path.exists(os.path.join(site, sid + ".html")):
|
|
problems.append("no generated page for symbol: %s.html" % sid)
|
|
|
|
# 2b) duplicate token → symbol conflicts (same token documented on two pages)
|
|
tok2ids = {}
|
|
for cat in sorted(os.listdir(LANG)):
|
|
cdir = os.path.join(LANG, cat)
|
|
if not os.path.isdir(cdir):
|
|
continue
|
|
for fn in os.listdir(cdir):
|
|
if not fn.endswith(".md") or fn == "_section.md":
|
|
continue
|
|
meta, _ = parse_front(os.path.join(cdir, fn))
|
|
for tok in meta.get("tokens", "").split():
|
|
tok2ids.setdefault((meta.get("kind",""), tok), set()).add(meta.get("id",""))
|
|
for (kind, tok), ids in sorted(tok2ids.items()):
|
|
if len(ids) > 1:
|
|
problems.append("token %r (%s) documented on multiple pages: %s" % (tok, kind, ", ".join(sorted(ids))))
|
|
|
|
# 2c) one docs directory per namespace / section. Two dirs covering the same
|
|
# runtime namespace (e.g. a `network`/`networking` or `date`/`datetime`
|
|
# split) invite drift — a symbol documented in one and not the other — so
|
|
# fail if any `ns:` is declared from more than one directory, or if two
|
|
# sections share a section id or (case-folded) title.
|
|
ns2dirs, id2dirs, title2dirs = {}, {}, {}
|
|
for cat in sorted(os.listdir(LANG)):
|
|
cdir = os.path.join(LANG, cat)
|
|
if not os.path.isdir(cdir):
|
|
continue
|
|
sp = os.path.join(cdir, "_section.md")
|
|
if os.path.exists(sp):
|
|
smeta, _ = parse_front(sp)
|
|
id2dirs.setdefault(smeta.get("id", cat), set()).add(cat)
|
|
title2dirs.setdefault(smeta.get("title", "").strip().lower(), set()).add(cat)
|
|
for fn in os.listdir(cdir):
|
|
if not fn.endswith(".md") or fn == "_section.md":
|
|
continue
|
|
meta, _ = parse_front(os.path.join(cdir, fn))
|
|
ns = meta.get("ns", "")
|
|
if ns:
|
|
ns2dirs.setdefault(ns, set()).add(cat)
|
|
for ns, dirs in sorted(ns2dirs.items()):
|
|
if len(dirs) > 1:
|
|
problems.append("namespace %r documented from multiple dirs: %s" % (ns, ", ".join(sorted(dirs))))
|
|
for sid, dirs in sorted(id2dirs.items()):
|
|
if len(dirs) > 1:
|
|
problems.append("section id %r declared by multiple dirs: %s" % (sid, ", ".join(sorted(dirs))))
|
|
for title, dirs in sorted(title2dirs.items()):
|
|
if title and len(dirs) > 1:
|
|
problems.append("section title %r shared by multiple dirs: %s" % (title, ", ".join(sorted(dirs))))
|
|
|
|
# 3) thin (un-expanded) symbols — warn, not fail (lets infra land before prose)
|
|
if thin:
|
|
warnings.append("%d symbols still have only seed text: %s%s"
|
|
% (len(thin), ", ".join(sorted(thin)[:12]), " …" if len(thin) > 12 else ""))
|
|
|
|
# 4) highlighter targets exist
|
|
sj = os.path.join(site, "symbols.json")
|
|
if os.path.exists(sj):
|
|
h = json.load(open(sj))["highlight"]
|
|
targets = set()
|
|
for grp in ("keywords", "types", "phases", "annotations"):
|
|
targets |= set(h.get(grp, {}).values())
|
|
for grp in ("builtins", "nsmethods"):
|
|
targets |= set(v["id"] for v in h.get(grp, {}).values())
|
|
targets |= set(h.get("namespaces", {}).values())
|
|
for tgt in sorted(targets):
|
|
if not os.path.exists(os.path.join(site, tgt + ".html")):
|
|
problems.append("highlighter links to %s.html but it was not generated" % tgt)
|
|
|
|
for w in warnings:
|
|
print(" warning:", w)
|
|
if problems:
|
|
print("docs check FAILED:")
|
|
for p in problems:
|
|
print(" -", p)
|
|
return 1
|
|
print("docs check OK: %s (%d symbols documented)" % (site, len(id2src)))
|
|
return 0
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main(sys.argv[1] if len(sys.argv) > 1 else os.path.join(ROOT, "build", "pages")))
|