ludic/tools/ludic-tools/fmt.ludic

713 lines
30 KiB
Text

# fmt.ludic — the canonical Ludic formatter, written in Ludic (replaces the C
# ludic_fmt_main.c + ludic_fmt.h). Works on the token stream, so comments and
# blank lines survive and nothing is ever dropped or reordered — only the
# whitespace between tokens is normalized. Lines are re-indented and respaced but
# never joined or split. Mirrors tools/ludic-tools/ludic_fmt.h exactly.
#
# ludic-fmt a.ludic print the formatted text
# ludic-fmt -w a.ludic rewrite in place
# ludic-fmt --check a.ludic exit 1 if unformatted
# ludic-fmt a.md format the ```ludic fences in a document
# cat a.ludic | ludic-fmt - filter mode (stdin -> stdout)
program LudicFmt {
import "fmt_lint.ludic"
import "fmt_lint_rules.ludic"
import "fmt_lint_run.ludic"
# ---- token kinds (mirror ludic_syntax.h) ----
const LT_EOF: int = 0
const LT_NL: int = 1
const LT_COMMENT: int = 2
const LT_ID: int = 3
const LT_KW: int = 4
const LT_TYPE: int = 5
const LT_PHASE: int = 6
const LT_BOOL: int = 7
const LT_INT: int = 8
const LT_FLOAT: int = 9
const LT_STR: int = 10
const LT_CHAR: int = 11
const LT_ANNO: int = 12
const LT_OP: int = 13
const LT_ERR: int = 14
# ---- tiny stdio + string helpers (self-contained) ----
function read_file(path: string) -> pointer {
let f = file_open(path, "rb")
if (f == null) { return null }
file_seek(f, 0, 2)
let n = file_tell(f)
file_seek(f, 0, 0)
let buf = bytes(n + 1)
file_read(f, buf, n)
buf[n] = 0
file_close(f)
return buf
}
function cstr_len(s: pointer) -> int { var n = 0; while s[n] != 0 { n += 1 }; return n }
function char_is_digit(c: int) -> bool { return c >= '0' and c <= '9' }
function char_is_alpha(c: int) -> bool {
if c >= 'A' and c <= 'Z' { return true }
if c >= 'a' and c <= 'z' { return true }
return c == '_'
}
function char_is_alnum(c: int) -> bool { return char_is_alpha(c) or char_is_digit(c) }
function char_is_hex(c: int) -> bool { return char_is_digit(c) or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F') }
function itoa(v: int) -> pointer {
if v == 0 { let z = bytes(2); z[0] = '0'; z[1] = 0; return z }
var neg = false; var x = v
if x < 0 { neg = true; x = -x }
let tmp = bytes(16); var n = 0
while x > 0 { tmp[n] = 48 + x % 10; x /= 10; n += 1 }
var total = n
if neg { total += 1 }
let out = bytes(total + 1); var k = 0
if neg { out[0] = '-'; k = 1 }
var i = 0
while i < n { out[k + i] = tmp[n - 1 - i]; i += 1 }
out[total] = 0
return out
}
# ---- a growable byte buffer ----
property Buf { data: pointer = null, len: int = 0, cap: int = 0 }
function buf_new() -> Buf { let b = new Buf; b.cap = 256; b.data = bytes(b.cap); b.len = 0; return b }
function buf_ensure(b: Buf, extra: int) -> void {
if b.len + extra + 1 <= b.cap { return }
while b.len + extra + 1 > b.cap { b.cap *= 2 }
b.data = resize(b.data, b.cap)
}
function buf_putc(b: Buf, c: int) -> void { buf_ensure(b, 1); b.data[b.len] = c; b.len += 1 }
function buf_puts(b: Buf, s: pointer) -> void { var i = 0; while s[i] != 0 { buf_putc(b, s[i]); i += 1 } }
function buf_indent(b: Buf, n: int) -> void { var i = 0; while i < n { buf_putc(b, ' '); i += 1 } }
function buf_str(b: Buf) -> pointer { b.data[b.len] = 0; return b.data }
# append src[a..b) raw
function buf_addrange(b: Buf, s: pointer, a: int, e: int) -> void { var k = a; while k < e { buf_putc(b, s[k]); k += 1 } }
# ---- the token stream (parallel slices) ----
var src: pointer = null
var tk_kind: []int
var tk_start: []int
var tk_end: []int
var tk_line: []int
var linestart: []int
var tk_gen: []int # 1 on a `<`, `>` or `>>` that brackets type arguments
function ntok() -> int { return len(tk_kind) }
function tok_len(i: int) -> int { return tk_end[i] - tk_start[i] }
function tok_text(i: int) -> pointer { return src[tk_start[i]..tk_end[i]] }
function push_tok(k: int, st: int, en: int, ln: int) -> void {
push(tk_kind, k); push(tk_start, st); push(tk_end, en); push(tk_line, ln)
}
# ---- vocabulary classifiers ----
function is_type_word(w: pointer) -> bool {
return (w == "int") or (w == "long") or (w == "fixed") or (w == "countdown") or (w == "bool") or (w == "entity") or (w == "string") or (w == "pointer") or (w == "byte") or (w == "words") or (w == "fixeds") or (w == "pointers") or (w == "void")
}
function is_phase_word(w: pointer) -> bool {
return (w == "Start") or (w == "Input") or (w == "FixedUpdate") or (w == "Update") or (w == "LateUpdate") or (w == "Render")
}
function is_keyword_word(w: pointer) -> bool {
if (w == "program") or (w == "import") or (w == "property") or (w == "model") or (w == "enum") or (w == "ui") { return true }
if (w == "const") or (w == "var") or (w == "function") or (w == "extern") or (w == "handler") or (w == "entry") or (w == "event") or (w == "scene") { return true }
if (w == "phase") or (w == "query") or (w == "on") or (w == "cancellable") or (w == "public") or (w == "layer") or (w == "start") { return true }
if (w == "let") or (w == "return") or (w == "if") or (w == "else") or (w == "while") or (w == "for") or (w == "in") or (w == "spawn") or (w == "despawn") { return true }
if (w == "enable") or (w == "disable") or (w == "match") or (w == "machine") or (w == "state") or (w == "become") or (w == "where") or (w == "prefab") { return true }
if (w == "and") or (w == "or") or (w == "not") or (w == "break") or (w == "continue") or (w == "new") or (w == "emit") or (w == "cancel") { return true }
return false
}
function is_clause_word(w: pointer) -> bool {
return (w == "phase") or (w == "query") or (w == "reads") or (w == "writes") or (w == "needs") or (w == "uses") or (w == "requires") or (w == "ensures") or (w == "invariant") or (w == "effects")
}
# ---- the lexer: keeps comments, newlines, byte spans; never exits on bad input ----
function is_op2(c0: int, c1: int) -> bool {
if c0 == '-' and c1 == '>' { return true } # ->
if c1 == '=' and (c0 == '+' or c0 == '-' or c0 == '*' or c0 == '/' or c0 == '=' or c0 == '!' or c0 == '<' or c0 == '>') { return true } # += -= *= /= == != <= >=
if c0 == '<' and c1 == '<' { return true } # <<
if c0 == '>' and c1 == '>' { return true } # >>
if c0 == '.' and c1 == '.' { return true } # ..
if c0 == '=' and c1 == '>' { return true } # =>
return false
}
function is_op1(c: int) -> bool {
if c == '+' or c == '-' or c == '*' or c == '/' or c == '%' or c == '<' or c == '>' or c == '=' { return true } # + - * / % < > =
if c == '(' or c == ')' or c == '{' or c == '}' or c == '[' or c == ']' { return true } # ( ) { } [ ]
if c == ',' or c == ':' or c == '.' or c == '!' or c == '@' or c == ';' { return true } # , : . ! @ ;
return c == '&' or c == '|' or c == '^' or c == '~' # & | ^ ~
}
# the backtick that closes the template opening at s[i0] (or the end): a `{...}` hole is code, so
# a string, a char or another template inside it is skipped whole - as the compiler reads it
function fmt_tmpl_end(s: pointer, i0: int) -> int {
var i = i0 + 1
while s[i] != 0 and s[i] != '`' {
if s[i] == '\\' and s[i + 1] != 0 { i += 2; continue }
if s[i] == '{' and s[i + 1] == '{' { i += 2; continue }
if s[i] == '{' { i = fmt_hole_end(s, i + 1) }
if s[i] != 0 { i += 1 }
}
return i
}
function fmt_hole_end(s: pointer, i0: int) -> int {
var i = i0
var depth = 1
while s[i] != 0 {
let d = s[i]
if d == '"' or d == '\'' {
i += 1
while s[i] != 0 and s[i] != d { if s[i] == '\\' and s[i + 1] != 0 { i += 1 }; i += 1 }
}
else if d == '`' { i = fmt_tmpl_end(s, i) }
else if d == '{' { depth += 1 }
else if d == '}' {
depth -= 1
if depth == 0 { return i }
}
if s[i] != 0 { i += 1 }
}
return i
}
function lex(s: pointer) -> void {
src = s
tk_kind = new []int; tk_start = new []int; tk_end = new []int; tk_line = new []int
linestart = new []int
push(linestart, 0)
var i = 0; var line = 0
while s[i] != 0 {
let c = s[i]
if c == '\n' { push_tok(LT_NL, i, i + 1, line); i += 1; line += 1; push(linestart, i); continue }
if c == ' ' or c == '\t' or c == '\r' { i += 1; continue }
if c == '#' { # '#' comment to end of line
let st = i; while s[i] != 0 and s[i] != '\n' { i += 1 }; push_tok(LT_COMMENT, st, i, line); continue
}
if c == '"' { # "string"
let st = i; i += 1
while s[i] != 0 and s[i] != '"' and s[i] != '\n' { if s[i] == '\\' and s[i + 1] != 0 { i += 2 } else { i += 1 } }
if s[i] == '"' { i += 1 }
push_tok(LT_STR, st, i, line); continue
}
if c == '`' { # `interpolated`, holes and all
let st = i
i = fmt_tmpl_end(s, i)
if s[i] == '`' { i += 1 }
push_tok(LT_STR, st, i, line); continue
}
if c == '\'' { # 'c'
let st = i; i += 1
if s[i] == '\\' and s[i + 1] != 0 { i += 2 } else { if s[i] != 0 and s[i] != '\n' { i += 1 } }
if s[i] == '\'' { i += 1 }
push_tok(LT_CHAR, st, i, line); continue
}
if char_is_digit(c) {
let st = i
if c == '0' and (s[i + 1] == 'x' or s[i + 1] == 'X') { # 0x hex
i += 2; while char_is_hex(s[i]) { i += 1 }; push_tok(LT_INT, st, i, line); continue
}
while char_is_digit(s[i]) { i += 1 }
if s[i] == '.' and char_is_digit(s[i + 1]) {
i += 1; while char_is_digit(s[i]) { i += 1 }; push_tok(LT_FLOAT, st, i, line); continue
}
push_tok(LT_INT, st, i, line); continue
}
if c == '@' and (char_is_alpha(s[i + 1]) or s[i + 1] == '_') { # @name annotation
let st = i; i += 1; while char_is_alnum(s[i]) { i += 1 }; push_tok(LT_ANNO, st, i, line); continue
}
if char_is_alpha(c) {
let st = i; while char_is_alnum(s[i]) { i += 1 }
let w = s[st..i]
if s[i] == '"' and ((w == "k") or (w == "kn")) { # k"pause.resume": a key literal, one token
i += 1
while s[i] != 0 and s[i] != '"' and s[i] != '\n' { if s[i] == '\\' and s[i + 1] != 0 { i += 2 } else { i += 1 } }
if s[i] == '"' { i += 1 }
push_tok(LT_STR, st, i, line); continue
}
var k = LT_ID
if (w == "true") or (w == "false") or (w == "null") { k = LT_BOOL }
else { if is_type_word(w) { k = LT_TYPE }
else { if is_phase_word(w) { k = LT_PHASE }
else { if is_keyword_word(w) { k = LT_KW } } } }
push_tok(k, st, i, line); continue
}
if is_op2(c, s[i + 1]) { push_tok(LT_OP, i, i + 2, line); i += 2; continue }
if is_op1(c) { push_tok(LT_OP, i, i + 1, line); i += 1; continue }
# anything else: one UTF-8 character's worth as an LT_ERR token
var ln = 1
if c >= 240 { ln = 4 } else { if c >= 224 { ln = 3 } else { if c >= 128 { ln = 2 } } }
var kk = 1
while kk < ln { if s[i + kk] == 0 or (s[i + kk] & 192) != 128 { ln = kk }; kk += 1 }
push_tok(LT_ERR, i, i + ln, line); i += ln
}
push_tok(LT_EOF, i, i, line)
mark_generics()
}
# ---- generics: Pool<Thing>, first<T>(, Map<string, []int> ----
function op_is(i: int, t: pointer) -> bool { return tk_kind[i] == LT_OP and (tok_text(i) == t) }
# a token that may stand inside a type argument list
function type_ish(i: int) -> bool {
let k = tk_kind[i]
if k == LT_ID or k == LT_TYPE { return true }
if k == LT_KW { return (tok_text(i) == "fn") }
if k != LT_OP { return false }
return op_is(i, ",") or op_is(i, "[") or op_is(i, "]") or op_is(i, "<") or op_is(i, ">") or op_is(i, ">>") or op_is(i, "(") or op_is(i, ")") or op_is(i, "->")
}
# a `<` written against a name, closed by a matching `>` on the same line with only a type
# between, opens type arguments; anything else is a comparison or a shift
function mark_generics() -> void {
tk_gen = new []int
var i = 0
while i < ntok() { push(tk_gen, 0); i += 1 }
i = 1
while i < ntok() {
let p = i - 1
if op_is(i, "<") and tk_gen[i] == 0 and (tk_kind[p] == LT_ID or tk_kind[p] == LT_TYPE) and tk_end[p] == tk_start[i] {
var depth = 1
var j = i + 1
var ok = true
while j < ntok() and depth > 0 and ok {
if not type_ish(j) or tk_line[j] != tk_line[i] { ok = false }
else {
if op_is(j, "<") { depth += 1 }
if op_is(j, ">") { depth -= 1 }
if op_is(j, ">>") { depth -= 2 }
if depth > 0 { j += 1 }
}
}
if ok and depth == 0 {
var m = i
while m <= j {
if op_is(m, "<") or op_is(m, ">") or op_is(m, ">>") { tk_gen[m] = 1 }
m += 1
}
}
}
i += 1
}
}
# ---- token-stream helpers ----
function next_sig(i: int) -> int {
var j = i + 1
while j < ntok() { let k = tk_kind[j]; if k != LT_NL and k != LT_COMMENT { return j }; j += 1 }
return -1
}
function prev_sig(i: int) -> int {
var j = i - 1
while j >= 0 { let k = tk_kind[j]; if k != LT_NL and k != LT_COMMENT { return j }; j -= 1 }
return -1
}
function name_like(k: int) -> bool { return k == LT_ID or k == LT_KW or k == LT_TYPE or k == LT_PHASE or k == LT_BOOL }
# is token i written right after a '.', i.e. a member/method name?
function is_member(i: int) -> bool {
let p = prev_sig(i)
return p >= 0 and tk_kind[p] == LT_OP and tok_len(p) == 1 and src[tk_start[p]] == '.'
}
# a '.' hugs an operand, a closing bracket, and another dot
function dot_tight(t: int) -> bool {
if name_like(tk_kind[t]) { return true }
if tk_kind[t] != LT_OP { return false }
let n = tok_len(t); let c0 = src[tk_start[t]]
if n == 1 and (c0 == ')' or c0 == ']' or c0 == '.') { return true }
if n == 2 and c0 == '.' and src[tk_start[t] + 1] == 46 { return true }
return false
}
# is the token at index i a unary '-' / '!' / '~' rather than a binary operator?
function is_unary(i: int) -> bool {
if tk_kind[i] != LT_OP { return false }
let n = tok_len(i)
if not (n == 1 and (src[tk_start[i]] == 45 or src[tk_start[i]] == 33 or src[tk_start[i]] == 126)) { return false }
let p = prev_sig(i)
if p < 0 { return true }
let pk = tk_kind[p]
if pk == LT_ID or pk == LT_INT or pk == LT_FLOAT or pk == LT_STR or pk == LT_CHAR or pk == LT_BOOL or pk == LT_TYPE or pk == LT_PHASE { return false }
if pk == LT_OP {
let c = src[tk_start[p]]
return not (tok_len(p) == 1 and (c == ')' or c == ']' or c == '}')) # a closing bracket ends an operand
}
return true # keyword/annotation/comment: operand starts here
}
# whitespace between the previous emitted token (prev) and the current one (cur)
function space_before(prev: int, cur: int) -> int {
if prev < 0 { return 0 }
let pk = tk_kind[prev]; let ck = tk_kind[cur]
let plen = tok_len(prev); let clen = tok_len(cur)
let p0 = src[tk_start[prev]]; let c0 = src[tk_start[cur]]
let p1 = (plen == 1); let c1 = (clen == 1)
if pk == LT_ERR or ck == LT_ERR { return tk_start[cur] - tk_end[prev] } # preserve an error token's spacing
# type arguments hug: Pool<Thing>, first<T>(xs), Pool<Pool<int>>
if tk_gen[cur] == 1 { return 0 }
if tk_gen[prev] == 1 and op_is(prev, "<") { return 0 }
if tk_gen[prev] == 1 and c1 and c0 == '(' { return 0 }
if c1 and (c0 == ')' or c0 == ']' or c0 == ',' or c0 == ':' or c0 == ';') { return 0 } # ) ] , : ;
if c1 and c0 == '.' and ck == LT_OP and dot_tight(prev) { return 0 }
if p1 and p0 == '.' and pk == LT_OP and dot_tight(cur) { return 0 }
if p1 and (p0 == '(' or p0 == '[') and pk == LT_OP { return 0 } # nothing hugs an opener from the right
# an index or slice hugs its operand: a[i], f()[0], a[i][j]; `query [` and `= [` keep their space
if c1 and c0 == '[' and ck == LT_OP {
if pk == LT_ID or pk == LT_STR { return 0 }
if pk == LT_OP and p1 and (p0 == ')' or p0 == ']') { return 0 }
}
# the element type hugs an empty pair of brackets: `[]int`, `[]Node`
if p1 and p0 == ']' and pk == LT_OP and (ck == LT_ID or ck == LT_TYPE) {
let pp = prev_sig(prev)
if pp >= 0 and tk_kind[pp] == LT_OP and tok_len(pp) == 1 and src[tk_start[pp]] == 91 { return 0 }
}
if pk == LT_OP and is_unary(prev) { return 0 }
if pk == LT_ANNO and c1 and c0 == '(' { return 0 } # @anno(
if c1 and c0 == '(' and ck == LT_OP {
if pk == LT_ID or pk == LT_TYPE or pk == LT_PHASE { return 0 } # function move( / clear(
if pk == LT_KW and (tok_text(prev) == "emit") { return 0 } # emit(...) is a call, `emit E(...)` an event
if pk == LT_KW and is_member(prev) { return 0 } # Prefab.spawn( / Date.new( — a member name is never a keyword
if pk == LT_OP { return 1 } # `= (`, `+ (`, `) (` keep a space
return 1
}
return 1
}
# ---- the formatting pass ----
function format(indent_width: int) -> Buf {
let o = buf_new()
let stack = words(600)
let hang = words(600)
var sp = 0
var cur = 0
var open = 0
var brack = 0
var ui_depth = -1
var pending_blank = 0
var wrote_any = 0
var prev_line_had_comment = 0
let N = ntok()
var i = 0
while i < N and tk_kind[i] != LT_EOF {
let a = i
while i < N and tk_kind[i] != LT_NL and tk_kind[i] != LT_EOF { i += 1 }
let b = i
if i < N and tk_kind[i] == LT_NL { i += 1 }
if a == b { # a blank line
if wrote_any == 1 { pending_blank = 1 }
continue
}
# where does this line start?
var line_level = cur
var tsp = sp
var t = a
while t < b and tk_kind[t] == LT_OP and tok_len(t) == 1 and src[tk_start[t]] == 125 { # leading '}'
if tsp > 0 { tsp -= 1; line_level = stack[tsp] }
t += 1
}
# original column of this line
var orig_ind = 0
var kk = linestart[tk_line[a]]
while kk < tk_start[a] { if src[kk] == '\t' { orig_ind += 4 } else { orig_ind += 1 }; kk += 1 }
# a comment continuing the previous comment line keeps the author's column
var comment_run = 0
if tk_kind[a] == LT_COMMENT and b == a + 1 and prev_line_had_comment == 1 { comment_run = 1 }
var ind = 0
var hangprev = 0
if sp > 0 { hangprev = hang[sp - 1] }
if open > 0 or (sp > 0 and hangprev == 1) { # author owns alignment inside open call/brace
let ls = linestart[tk_line[a]]
ind = 0
var k2 = ls
while k2 < tk_start[a] { if src[k2] == '\t' { ind += 4 } else { ind += 1 }; k2 += 1 }
} else {
var extra = 0
if is_clause_word(tok_text(a)) and tk_kind[a] == LT_KW { extra = indent_width }
ind = line_level * indent_width + extra
if comment_run == 1 and orig_ind > ind { ind = orig_ind }
}
if pending_blank == 1 and wrote_any == 1 { buf_putc(o, '\n') }
pending_blank = 0
buf_indent(o, ind)
# emit the tokens
var prev = -1
var line_brack = brack
var in_tag = 0 # inside a query's {Tag} filter
var line_ui_open = 0
if ui_depth >= 0 and sp > ui_depth { line_ui_open = 1 }
t = a
while t < b {
var want = 0
if tk_kind[t] == LT_COMMENT { if prev >= 0 { want = 2 } else { want = 0 } }
else {
want = space_before(prev, t)
# a query's {Tag} filter is one word: a `{` straight after `[` or `,` opens one
# (a record literal `new R { … }` inside a list keeps its spaces)
if line_brack > 0 and tk_kind[t] == LT_OP and tok_len(t) == 1 {
let c = src[tk_start[t]]
if c == '{' and prev >= 0 and tk_kind[prev] == LT_OP and tok_len(prev) == 1 and (src[tk_start[prev]] == '[' or src[tk_start[prev]] == ',') { in_tag = 1 }
if c == '}' and in_tag == 1 { want = 0; in_tag = 0 }
}
if in_tag == 1 and prev >= 0 and tk_kind[prev] == LT_OP and tok_len(prev) == 1 and src[tk_start[prev]] == '{' { want = 0 }
# a slice range hugs its bounds, `s[a..b]`; a `for i in a .. b` range keeps its spaces
if line_brack > 0 {
if tk_kind[t] == LT_OP and tok_len(t) == 2 and src[tk_start[t]] == 46 and src[tk_start[t] + 1] == 46 { want = 0 }
if prev >= 0 and tk_kind[prev] == LT_OP and tok_len(prev) == 2 and src[tk_start[prev]] == 46 and src[tk_start[prev] + 1] == 46 { want = 0 }
}
if line_ui_open == 1 { # widget props are k=v
var eq_here = 0
if tk_kind[t] == LT_OP and tok_len(t) == 1 and src[tk_start[t]] == 61 { eq_here = 1 }
var eq_prev = 0
if prev >= 0 and tk_kind[prev] == LT_OP and tok_len(prev) == 1 and src[tk_start[prev]] == 61 { eq_prev = 1 }
if eq_here == 1 or eq_prev == 1 { want = 0 }
}
}
var gap = 0
if prev >= 0 { gap = tk_start[t] - tk_end[prev] }
if gap >= 2 and want >= 1 { if gap > 60 { gap = 60 }; want = gap } # hand alignment wins
if want < 0 { want = 0 }
var w2 = 0
while w2 < want { buf_putc(o, ' '); w2 += 1 }
buf_addrange(o, src, tk_start[t], tk_end[t])
prev = t
if tk_kind[t] == LT_OP and tok_len(t) == 1 {
let c = src[tk_start[t]]
if c == '[' { line_brack += 1 }
else { if c == ']' { line_brack -= 1; if line_brack < 0 { line_brack = 0 } } }
}
t += 1
}
buf_putc(o, '\n')
wrote_any = 1
prev_line_had_comment = 0
if prev >= 0 and tk_kind[prev] == LT_COMMENT { prev_line_had_comment = 1 }
# carry the nesting into the next line
if tk_kind[a] == LT_KW and (tok_text(a) == "ui") and ui_depth < 0 { ui_depth = sp }
var level = cur
t = a
while t < b {
if tk_kind[t] == LT_OP and tok_len(t) == 1 {
let c = src[tk_start[t]]
if c == '{' { # '{'
if sp < 512 {
let nxt = next_sig(t)
stack[sp] = line_level
var h = 0
if nxt >= 0 and nxt < b { h = 1 }
hang[sp] = h
sp += 1
}
level = line_level + 1
}
else { if c == '}' { if sp > 0 { sp -= 1; level = stack[sp] } }
else { if c == '(' or c == '[' { open += 1; if c == '[' { brack += 1 } }
else { if c == ')' or c == ']' { open -= 1; if open < 0 { open = 0 }; if c == ']' { brack -= 1; if brack < 0 { brack = 0 } } } } } }
}
t += 1
}
cur = level
if ui_depth >= 0 and sp <= ui_depth { ui_depth = -1 }
}
return o
}
# ---- markdown: format the body of every ```ludic fence, leave prose alone ----
var g_flen: int = 0
var g_info: int = 0
var g_marker: int = 0
# detect a fence at line offset i; sets g_flen/g_info/g_marker; returns bool
function md_fence_at(s: pointer, i: int) -> bool {
var j = i; while s[j] == ' ' { j += 1 }
let m = s[j]
if m != '`' and m != '~' { return false } # ` or ~
var n = 0; while s[j] == m { j += 1; n += 1 }
if n < 3 { return false }
g_flen = n; g_info = j; g_marker = m
return true
}
function md_info_is_ludic(s: pointer, at: int) -> bool {
var a = at; while s[a] == ' ' or s[a] == '\t' { a += 1 }
# case-insensitive "ludic"
if not (((s[a] == 'l' or s[a] == 'L')) and ((s[a + 1] == 'u' or s[a + 1] == 'U')) and ((s[a + 2] == 'd' or s[a + 2] == 'D')) and ((s[a + 3] == 'i' or s[a + 3] == 'I')) and ((s[a + 4] == 'c' or s[a + 4] == 'C'))) { return false }
let af = s[a + 5]
return af == 0 or af == '\n' or af == ' ' or af == '\t' or af == '\r'
}
function line_end(s: pointer, i: int) -> int { var e = i; while s[e] != 0 and s[e] != '\n' { e += 1 }; return e }
function format_markdown(s: pointer, indent_width: int) -> Buf {
let o = buf_new()
var i = 0
while s[i] != 0 {
let ls = i
let le = line_end(s, ls)
var indent = 0; while s[ls + indent] == ' ' { indent += 1 }
if md_fence_at(s, ls) and md_info_is_ludic(s, g_info) {
# copy the opening fence line verbatim (with its newline)
var e0 = le; if s[le] != 0 { e0 = le + 1 }
buf_addrange(o, s, ls, e0)
i = e0
# gather the body up to the closing fence
let bs = i
var be = bs
while true {
if s[be] == 0 { break }
let ps = be; let pe = line_end(s, ps)
if md_fence_at(s, ps) and g_marker == g_marker and g_flen >= g_flen {
# re-run detection for THIS line (md_fence_at set globals for ps)
let cmark = g_marker; let clen = g_flen; let cinfo = g_info
if md_fence_at(s, ps) {
if g_marker == cmark and g_flen >= clen {
var only = 1
var k = g_info
while k < pe { if s[k] != ' ' and s[k] != '\r' { only = 0; break }; k += 1 }
if only == 1 { be = ps; break }
}
}
}
if s[pe] != 0 { be = pe + 1 } else { be = pe }
}
# de-indent the body, format it, re-indent it
let body = buf_new()
var p = bs
while p < be {
var q = line_end(s, p)
var skip = 0
while skip < indent and (p + skip) < q and s[p + skip] == ' ' { skip += 1 }
buf_addrange(body, s, p + skip, q)
buf_putc(body, '\n')
if s[q] != 0 { p = q + 1 } else { p = q }
}
# feed the body through the ludic formatter
lex(buf_str(body))
let f = format(indent_width)
let ftext = buf_str(f)
var fp = 0
while ftext[fp] != 0 {
var fq = fp; while ftext[fq] != 0 and ftext[fq] != '\n' { fq += 1 }
if fq > fp { buf_indent(o, indent) }
buf_addrange(o, ftext, fp, fq)
buf_putc(o, '\n')
if ftext[fq] != 0 { fp = fq + 1 } else { fp = fq }
}
i = be
continue
}
var e1 = le; if s[le] != 0 { e1 = le + 1 }
buf_addrange(o, s, ls, e1)
i = e1
}
return o
}
# ---- ends-with helpers for extension detection ----
function ends_with(s: pointer, suf: pointer) -> bool {
let n = cstr_len(s); let m = cstr_len(suf)
if m > n { return false }
return (s[n - m..n] == suf)
}
# ---- stdin slurp (for `-`) ----
function slurp_stdin() -> pointer {
let b = buf_new()
var c = read_char()
while c >= 0 { buf_putc(b, c); c = read_char() }
return buf_str(b)
}
# format one source string according to its kind (markdown vs ludic)
function format_source(text: pointer, is_md: bool, indent_width: int) -> pointer {
if is_md { let m = format_markdown(text, indent_width); return buf_str(m) }
lex(text)
let f = format(indent_width)
return buf_str(f)
}
function streq(a: pointer, b: pointer) -> bool { return (a == b) }
entry {
var write = false
var check = false
var indent = 2
var quiet = false
var changed = false
var failed = false
var lint_only = false
var lint_init = false
let files = new []pointer
var ai = 1
while ai < arg_count() {
let a = arg(ai)
if (a == "-w") or (a == "--write") { write = true }
else { if (a == "--check") or (a == "-l") { check = true }
else { if (a == "-q") or (a == "--quiet") { quiet = true }
else { if (a == "--indent") { ai += 1; if ai < arg_count() { indent = 0; let d = arg(ai); var di = 0; while d[di] != 0 { indent = indent * 10 + (d[di] - 48); di += 1 } } }
else { if (a == "--lint") { lint_only = true }
else { if (a == "--init-baseline") { lint_only = true; lint_init = true }
else { if (a == "-h") or (a == "--help") { print("ludic-fmt — format Ludic source; --check also checks package.ludic's lint rules, --lint checks the project"); return }
else { push(files, a) } } } } } } }
ai += 1
}
if indent < 1 or indent > 8 { indent = 2 }
# --lint: the project's rules over its paths, against the baseline (L10)
if lint_only {
if not lint_config() { print("ludic-fmt --lint: package.ludic states no `lint` rules"); exit(2) }
let all = new []string
var li = 0
while li < len(lt_paths) {
lint_walk(lt_paths[li], all)
li += 1
}
var lj = 0
while lj < len(all) {
let t = read_file(all[lj])
if t != null { lint_file(all[lj], string(t)) }
lj += 1
}
if not lint_judge(lint_init) { exit(1) }
return
}
let linting = check and lint_config()
# stdin filter mode
if len(files) == 0 or (len(files) == 1 and (files[0] == "-")) {
let text = slurp_stdin()
let out = format_source(text, false, indent)
file_write(file_stdout(), out, cstr_len(out))
return
}
var fi = 0
while fi < len(files) {
let path = files[fi]
let text = read_file(path)
if (text == null) {
let m = `ludic-fmt: cannot open {path}\n`
file_write(file_stderr(), m, cstr_len(m)); failed = true
} else {
let is_md = ends_with(path, ".md") or ends_with(path, ".markdown")
if linting and not is_md { lint_file(path, string(text)) }
let out = format_source(text, is_md, indent)
let same = (out == text)
if check {
if not same { changed = true; if not quiet { print(path) } }
} else { if write {
if not same {
let f = file_open(path, "wb")
if (f == null) { let m = `ludic-fmt: cannot write {path}\n`; file_write(file_stderr(), m, cstr_len(m)); failed = true }
else { file_write(f, out, cstr_len(out)); file_close(f); if not quiet { let m = `formatted {path}\n`; file_write(file_stderr(), m, cstr_len(m)) } }
changed = true
}
} else {
file_write(file_stdout(), out, cstr_len(out))
} }
}
fi += 1
}
if failed { exit(2) }
# the files named are part of a project: judge them, but never rewrite its baseline
if linting {
lt_partial = true
if not lint_judge(false) { exit(1) }
}
if check and changed { exit(1) }
}
}