ludic/tools/ludic-tools/ludic_fmt.h
Orkuncakilkaya 985f9ad8f2 Baseline: Ludic compiler + toolchain, Phase 1 syntax fixes complete
Self-hosted compiler (selfhost/*.ludic), runtime, examples, editor tooling,
and docs. Phase 1 of the syntax-redesign cohesion pass has landed:
edge-system fix, signature-query, when-alias, and the documentation truth-pass.
Suite green (14/14), C-free bootstrap fixpoint holds.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-08-27 15:15:35 +03:00

339 lines
16 KiB
C

/* ============================================================================
* ludic_fmt.h — the canonical Ludic formatter.
*
* This is deliberately NOT `ludicc --fmt`. The compiler's printer walks the AST
* after import splicing, so it drops every comment and inlines every imported
* file into whichever file you pointed it at — fine for inspecting what the
* compiler saw, catastrophic as an editor's "format on save".
*
* This formatter works on the token stream instead:
* - comments and blank lines survive, because they are tokens;
* - nothing is ever dropped or reordered, because every token is re-emitted
* in order — the only freedom taken is the whitespace between them;
* - lines are re-indented and respaced but never joined or split, so the
* author keeps control of line structure and a format-on-save never
* rewrites a file out from under someone mid-edit.
* ==========================================================================*/
#ifndef LUDIC_FMT_H
#define LUDIC_FMT_H
#include "ludic_syntax.h"
typedef struct { char* b; size_t n, cap; } FSB;
static void fsb_ensure(FSB* s, size_t add){
if (s->n + add + 1 > s->cap){ s->cap = (s->n + add + 1) * 2; s->b = realloc(s->b, s->cap); }
}
static void fsb_add(FSB* s, const char* z, size_t l){ fsb_ensure(s, l); memcpy(s->b + s->n, z, l); s->n += l; s->b[s->n] = 0; }
static void fsb_puts(FSB* s, const char* z){ fsb_add(s, z, strlen(z)); }
static void fsb_putc(FSB* s, char c){ fsb_add(s, &c, 1); }
static void fsb_indent(FSB* s, int n){ for (int i = 0; i < n; i++) fsb_putc(s, ' '); }
/* Can this token be one side of a member access? */
static int lud_name_like(int kind){
return kind == LT_ID || kind == LT_KW || kind == LT_TYPE || kind == LT_PHASE || kind == LT_BOOL;
}
/* A '.' hugs an operand, a closing bracket, and another dot — the last so that
* a run of dots stays a run of dots instead of being spaced into pieces. */
static int lud_dot_tight(const LLex* L, const LTok* t){
if (lud_name_like(t->kind)) return 1;
if (t->kind != LT_OP) return 0;
int n = ltok_len(t);
char c0 = L->src[t->start];
if (n == 1 && (c0 == ')' || c0 == ']' || c0 == '.')) return 1;
if (n == 2 && c0 == '.' && L->src[t->start + 1] == '.') return 1;
return 0;
}
/* Is the token at index i a unary '-' / '!' rather than a binary operator?
* Unary iff nothing that can end an operand precedes it. */
static int fmt_is_unary(const LLex* L, int i){
const LTok* t = &L->v[i];
if (t->kind != LT_OP) return 0;
int n = ltok_len(t);
if (!(n == 1 && (L->src[t->start] == '-' || L->src[t->start] == '!'))) return 0;
int p = ltok_prev_sig(L, i);
if (p < 0) return 1;
const LTok* pt = &L->v[p];
switch (pt->kind){
case LT_ID: case LT_INT: case LT_FLOAT: case LT_STR: case LT_CHAR:
case LT_BOOL: case LT_TYPE: case LT_PHASE:
return 0;
case LT_OP: {
char c = L->src[pt->start];
/* a closing bracket ends an operand; every other operator does not */
return !(ltok_len(pt) == 1 && (c == ')' || c == ']' || c == '}'));
}
default: return 1; /* keyword, annotation, comment: operand starts here */
}
}
/* Whitespace between the previous emitted token (prev) and the current one. */
static int fmt_space_before(const LLex* L, int prev, int cur){
if (prev < 0) return 0;
const LTok* p = &L->v[prev];
const LTok* c = &L->v[cur];
const char* src = L->src;
int plen = ltok_len(p), clen = ltok_len(c);
char p0 = src[p->start], c0 = src[c->start];
int p1 = (plen == 1), c1 = (clen == 1);
/* An error token is bytes we did not understand. Whatever spacing it had
* is the only spacing we can justify, so it is preserved verbatim. */
if (p->kind == LT_ERR || c->kind == LT_ERR) return c->start - p->end;
/* nothing hugs a closer, a separator or a member dot from the left */
if (c1 && (c0 == ')' || c0 == ']' || c0 == ',' || c0 == ':' || c0 == ';')) return 0;
/* `.` binds tight only when it really is member access — `a.b`. A lone dot
* next to a brace is something else (or a typo) and gets normal spacing. */
if (c1 && c0 == '.' && c->kind == LT_OP && lud_dot_tight(L, p)) return 0;
if (p1 && p0 == '.' && p->kind == LT_OP && lud_dot_tight(L, c)) return 0;
/* nothing hugs an opener from the right */
if (p1 && (p0 == '(' || p0 == '[') && p->kind == LT_OP) return 0;
/* a unary sign binds to its operand */
if (p->kind == LT_OP && fmt_is_unary(L, prev)) return 0;
/* @anno(args) */
if (p->kind == LT_ANNO && c1 && c0 == '(') return 0;
if (c1 && c0 == '(' && c->kind == LT_OP){
/* `fn move(` and `clear(` hug; `if (`, `return (`, `x * (` do not */
switch (p->kind){
case LT_ID: case LT_TYPE: case LT_PHASE: return 0;
case LT_OP: return !(p1 && (p0 == ')' || p0 == ']'));
default: return 1;
}
}
return 1;
}
/* Clause keywords hang under the declaration they qualify: a system's
* `phase`/`query` lines and a function's contract lines are indented one level
* past the `system`/`fn` they belong to, with the body brace back at the
* declaration's own level. Every example in the tree is written that way. */
static int fmt_is_clause_word(const LLex* L, int i){
static const char* CLAUSES[] = { "phase","query","reads","writes","needs","uses",
"requires","ensures","invariant","effects", 0 };
if (i < 0 || i >= L->n || L->v[i].kind != LT_KW) return 0;
char b[32]; ltok_text(L, i, b, sizeof(b));
return lud_in(CLAUSES, b);
}
/* One pass over the token stream, re-emitting it with canonical whitespace.
*
* Two carve-outs keep the result idiomatic rather than merely uniform:
* - a run of two or more spaces is preserved verbatim, so hand-aligned
* columns (`const R_DIR: int = 0`) and deliberately set-off trailing
* comments survive a format-on-save;
* - `id=Root` inside a `ui` block and `{Enemy}` inside a query stay tight,
* because those are the spellings the language documents and uses.
*/
static char* ludic_format(const char* src, int indent_width){
LLex L; lud_lex(&L, src);
FSB o = {0};
/* One entry per open brace, holding the indent level to return to. Every
* brace opened on the same line shares that line's level, so a line like
* `if a { if b { if c {` steps in by ONE level, not three — and the line
* that closes them all lands back where it started. */
int stack[512]; int hang[512]; int sp = 0;
int cur = 0;
int open = 0; /* unclosed ( or [ : the author owns the alignment */
int brack = 0; /* unclosed [ only: query-term context */
int ui_depth = -1; /* brace depth just outside the innermost `ui` block */
int pending_blank = 0;
int wrote_any = 0;
int prev_line_had_comment = 0;
int i = 0;
while (i < L.n && L.v[i].kind != LT_EOF){
int a = i;
while (i < L.n && L.v[i].kind != LT_NL && L.v[i].kind != LT_EOF) i++;
int b = i; /* [a,b) are this line's tokens */
if (i < L.n && L.v[i].kind == LT_NL) i++;
if (a == b){ /* a blank line */
if (wrote_any) pending_blank = 1;
continue;
}
/* ---- where does this line start? --------------------------------- */
int line_level = cur;
{ /* leading closers belong to the level of the line that opened them */
int tsp = sp, t = a;
while (t < b && L.v[t].kind == LT_OP && ltok_len(&L.v[t]) == 1 && src[L.v[t].start] == '}'){
if (tsp > 0) line_level = stack[--tsp];
t++;
}
}
/* Original column of this line, for the cases where the author's
* alignment is the only sensible answer. */
int orig_ind = 0;
for (int k = L.linestart[L.v[a].line]; k < L.v[a].start; k++) orig_ind += (src[k] == '\t') ? 4 : 1;
/* A comment on its own line, indented past its block and following a
* line that itself ended in a comment, is the continuation of that
* comment — a column the author chose, not stray indentation. */
int comment_run = (L.v[a].kind == LT_COMMENT && b == a + 1 && prev_line_had_comment);
int ind;
if (open > 0 || (sp > 0 && hang[sp - 1])){
/* Inside an unclosed call, query, or a brace that was opened with
* content trailing it, the author is aligning to a column the
* formatter cannot see. Leave those lines exactly as written. */
int ls = L.linestart[L.v[a].line];
ind = 0;
for (int k = ls; k < L.v[a].start; k++) ind += (src[k] == '\t') ? 4 : 1;
} else {
ind = line_level * indent_width + (fmt_is_clause_word(&L, a) ? indent_width : 0);
if (comment_run && orig_ind > ind) ind = orig_ind;
}
if (pending_blank && wrote_any) fsb_putc(&o, '\n');
pending_blank = 0;
fsb_indent(&o, ind);
/* ---- the tokens --------------------------------------------------- */
int prev = -1;
int line_brack = brack, line_ui_open = (ui_depth >= 0 && sp > ui_depth);
for (int t = a; t < b; t++){
const LTok* tk = &L.v[t];
int gap = (prev >= 0) ? tk->start - L.v[prev].end : 0;
int want;
if (tk->kind == LT_COMMENT){
want = (prev >= 0) ? 2 : 0; /* set a trailing comment off */
} else {
want = fmt_space_before(&L, prev, t);
/* a query's {Tag} filter is one word, not a record literal */
if (line_brack > 0){
if (tk->kind == LT_OP && ltok_len(tk) == 1 && src[tk->start] == '}') want = 0;
if (prev >= 0 && L.v[prev].kind == LT_OP && ltok_len(&L.v[prev]) == 1 && src[L.v[prev].start] == '{') want = 0;
}
/* widget props are written k=v */
if (line_ui_open){
int eq_here = tk->kind == LT_OP && ltok_len(tk) == 1 && src[tk->start] == '=';
int eq_prev = prev >= 0 && L.v[prev].kind == LT_OP &&
ltok_len(&L.v[prev]) == 1 && src[L.v[prev].start] == '=';
if (eq_here || eq_prev) want = 0;
}
}
/* hand alignment wins over the canonical single space */
if (gap >= 2 && want >= 1){ if (gap > 60) gap = 60; want = gap; }
if (want < 0) want = 0;
for (int k = 0; k < want; k++) fsb_putc(&o, ' ');
fsb_add(&o, src + tk->start, ltok_len(tk));
prev = t;
if (tk->kind == LT_OP && ltok_len(tk) == 1){
char c = src[tk->start];
if (c == '[') line_brack++;
else if (c == ']'){ line_brack--; if (line_brack < 0) line_brack = 0; }
}
}
fsb_putc(&o, '\n');
wrote_any = 1;
prev_line_had_comment = (prev >= 0 && L.v[prev].kind == LT_COMMENT);
/* ---- carry the nesting into the next line ------------------------- */
if (ltok_is(&L, a, "ui") && L.v[a].kind == LT_KW && ui_depth < 0) ui_depth = sp;
{
int level = cur;
for (int t = a; t < b; t++){
const LTok* tk = &L.v[t];
if (tk->kind != LT_OP || ltok_len(tk) != 1) continue;
char c = src[tk->start];
if (c == '{'){
if (sp < 512){
int nxt = ltok_next_sig(&L, t);
stack[sp] = line_level;
hang[sp] = (nxt >= 0 && nxt < b); /* content follows on this line */
sp++;
}
level = line_level + 1;
}
else if (c == '}'){ if (sp > 0) level = stack[--sp]; }
else if (c == '(' || c == '['){ open++; if (c == '[') brack++; }
else if (c == ')' || c == ']'){ open--; if (open < 0) open = 0; if (c == ']'){ brack--; if (brack < 0) brack = 0; } }
}
cur = level;
}
if (ui_depth >= 0 && sp <= ui_depth) ui_depth = -1;
}
lud_lex_free(&L);
if (!o.b) { o.b = malloc(1); o.b[0] = 0; }
return o.b;
}
/* ---------- markdown ------------------------------------------------------
* Fenced Ludic in prose is still Ludic. This rewrites the body of every
* ```ludic fence in a Markdown document and leaves the prose untouched, so
* LANGUAGE.md and README.md can be kept honest by the same formatter as the
* source tree. Fence indentation (a fence inside a list item) is preserved. */
static int md_fence_at(const char* s, int i, int* fence_len, int* info_at, char* marker){
int j = i, n = 0;
while (s[j] == ' ') j++; /* leading indent */
char m = s[j];
if (m != '`' && m != '~') return 0;
while (s[j] == m){ j++; n++; }
if (n < 3) return 0;
*fence_len = n; *info_at = j; *marker = m;
return 1;
}
static int md_info_is_ludic(const char* s, int at){
while (s[at] == ' ' || s[at] == '\t') at++;
if (strncasecmp(s + at, "ludic", 5)) return 0;
char after = s[at + 5];
return after == 0 || after == '\n' || after == ' ' || after == '\t' || after == '\r';
}
static char* ludic_format_markdown(const char* src, int indent_width){
FSB o = {0};
int i = 0;
while (src[i]){
int ls = i; /* line start */
int le = ls; while (src[le] && src[le] != '\n') le++;
int flen, info, indent = 0; char marker;
while (src[ls + indent] == ' ') indent++;
if (md_fence_at(src, ls, &flen, &info, &marker) && md_info_is_ludic(src, info)){
/* copy the opening fence line verbatim */
fsb_add(&o, src + ls, le - ls + (src[le] ? 1 : 0));
i = src[le] ? le + 1 : le;
/* gather the body up to the closing fence */
int bs = i;
int be = bs;
for (;;){
if (!src[be]) break;
int ps = be, pe = ps; while (src[pe] && src[pe] != '\n') pe++;
int clen, cinfo; char cmark;
if (md_fence_at(src, ps, &clen, &cinfo, &cmark) && cmark == marker && clen >= flen){
int only = 1;
for (int k = cinfo; k < pe; k++) if (src[k] != ' ' && src[k] != '\r'){ only = 0; break; }
if (only) break;
}
be = src[pe] ? pe + 1 : pe;
}
/* de-indent the body, format it, re-indent it */
FSB body = {0};
for (int p = bs; p < be; ){
int q = p; while (src[q] && src[q] != '\n') q++;
int skip = 0; while (skip < indent && p + skip < q && src[p + skip] == ' ') skip++;
fsb_add(&body, src + p + skip, q - (p + skip));
fsb_putc(&body, '\n');
p = src[q] ? q + 1 : q;
}
char* f = ludic_format(body.b ? body.b : "", indent_width);
for (char* p = f; *p; ){
char* q = strchr(p, '\n'); if (!q) q = p + strlen(p);
if (q > p) fsb_indent(&o, indent);
fsb_add(&o, p, q - p);
fsb_putc(&o, '\n');
p = *q ? q + 1 : q;
}
free(f); free(body.b);
i = be;
continue;
}
fsb_add(&o, src + ls, le - ls + (src[le] ? 1 : 0));
i = src[le] ? le + 1 : le;
}
if (!o.b){ o.b = malloc(1); o.b[0] = 0; }
return o.b;
}
#endif /* LUDIC_FMT_H */