Some checks are pending
docs / build-and-deploy (push) Waiting to run
Add `long`, a 64-bit signed integer primitive (i64), threaded through codegen: llty; int<->long coercion (coerce_code/to_long) at let/assign/ return/call-args; i64 arithmetic + comparison promotion in emit_bin; unary negate/~; print via %lld and str()/interpolation via @fn_long_str. Editor vocabulary (syntax header, TextMate grammar, formatter, LSP) synced; check-vocabulary green. Numeric literals stay i32 — build large values by widening (documented on the type page). Complete the Hash.* namespace (issue #17) with both 32- and 64-bit algorithms: Hash.of/fnv1a/crc32/mix/combine (32-bit) and Hash.of64/fnv1a_64/mix64 (64-bit, returning long). Deterministic and C-free; CRC-32 (poly 0xEDB88320) and FNV vectors verified against reference implementations. Tests: selfhost/tests/{hash,long}.ludic. Docs: docs/language/hash/*, type-long.md. Seed reseeded; C-free bootstrap fixpoint holds; 23 selfhost + 45 regression + 29 tooling checks green. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
563 lines
24 KiB
Text
563 lines
24 KiB
Text
# fmt.ludic — the canonical Ludic formatter, written in Ludic (replaces the C
|
|
# ludic_fmt_main.c + ludic_fmt.h). Works on the token stream, so comments and
|
|
# blank lines survive and nothing is ever dropped or reordered — only the
|
|
# whitespace between tokens is normalized. Lines are re-indented and respaced but
|
|
# never joined or split. Mirrors tools/ludic-tools/ludic_fmt.h exactly.
|
|
#
|
|
# ludic-fmt a.ludic print the formatted text
|
|
# ludic-fmt -w a.ludic rewrite in place
|
|
# ludic-fmt --check a.ludic exit 1 if unformatted
|
|
# ludic-fmt a.md format the ```ludic fences in a document
|
|
# cat a.ludic | ludic-fmt - filter mode (stdin -> stdout)
|
|
program LudicFmt {
|
|
# ---- token kinds (mirror ludic_syntax.h) ----
|
|
const LT_EOF: int = 0
|
|
const LT_NL: int = 1
|
|
const LT_COMMENT: int = 2
|
|
const LT_ID: int = 3
|
|
const LT_KW: int = 4
|
|
const LT_TYPE: int = 5
|
|
const LT_PHASE: int = 6
|
|
const LT_BOOL: int = 7
|
|
const LT_INT: int = 8
|
|
const LT_FLOAT: int = 9
|
|
const LT_STR: int = 10
|
|
const LT_CHAR: int = 11
|
|
const LT_ANNO: int = 12
|
|
const LT_OP: int = 13
|
|
const LT_ERR: int = 14
|
|
|
|
# ---- tiny stdio + string helpers (self-contained) ----
|
|
fn read_file(path: str) -> ptr {
|
|
let f = file_open(path, "rb")
|
|
if (f == null) { return null }
|
|
file_seek(f, 0, 2)
|
|
let n = file_tell(f)
|
|
file_seek(f, 0, 0)
|
|
let buf = bytes(n + 1)
|
|
file_read(f, buf, n)
|
|
buf[n] = 0
|
|
file_close(f)
|
|
return buf
|
|
}
|
|
fn cstr_len(s: ptr) -> int { var n = 0; while s[n] != 0 { n = n + 1 }; return n }
|
|
fn char_is_digit(c: int) -> bool { return c >= 48 and c <= 57 }
|
|
fn char_is_alpha(c: int) -> bool {
|
|
if c >= 65 and c <= 90 { return true }
|
|
if c >= 97 and c <= 122 { return true }
|
|
return c == 95
|
|
}
|
|
fn char_is_alnum(c: int) -> bool { return char_is_alpha(c) or char_is_digit(c) }
|
|
fn char_is_hex(c: int) -> bool { return char_is_digit(c) or (c >= 97 and c <= 102) or (c >= 65 and c <= 70) }
|
|
fn itoa(v: int) -> ptr {
|
|
if v == 0 { let z = bytes(2); z[0] = 48; z[1] = 0; return z }
|
|
var neg = false; var x = v
|
|
if x < 0 { neg = true; x = 0 - x }
|
|
let tmp = bytes(16); var n = 0
|
|
while x > 0 { tmp[n] = 48 + x % 10; x = x / 10; n = n + 1 }
|
|
var total = n
|
|
if neg { total = total + 1 }
|
|
let out = bytes(total + 1); var k = 0
|
|
if neg { out[0] = 45; k = 1 }
|
|
var i = 0
|
|
while i < n { out[k + i] = tmp[n - 1 - i]; i = i + 1 }
|
|
out[total] = 0
|
|
return out
|
|
}
|
|
|
|
# ---- a growable byte buffer ----
|
|
property Buf { data: ptr = null, len: int = 0, cap: int = 0 }
|
|
fn buf_new() -> Buf { let b = new Buf; b.cap = 256; b.data = bytes(b.cap); b.len = 0; return b }
|
|
fn buf_ensure(b: Buf, extra: int) -> void {
|
|
if b.len + extra + 1 <= b.cap { return }
|
|
while b.len + extra + 1 > b.cap { b.cap = b.cap * 2 }
|
|
b.data = resize(b.data, b.cap)
|
|
}
|
|
fn buf_putc(b: Buf, c: int) -> void { buf_ensure(b, 1); b.data[b.len] = c; b.len = b.len + 1 }
|
|
fn buf_puts(b: Buf, s: ptr) -> void { var i = 0; while s[i] != 0 { buf_putc(b, s[i]); i = i + 1 } }
|
|
fn buf_indent(b: Buf, n: int) -> void { var i = 0; while i < n { buf_putc(b, 32); i = i + 1 } }
|
|
fn buf_str(b: Buf) -> ptr { b.data[b.len] = 0; return b.data }
|
|
# append src[a..b) raw
|
|
fn buf_addrange(b: Buf, s: ptr, a: int, e: int) -> void { var k = a; while k < e { buf_putc(b, s[k]); k = k + 1 } }
|
|
|
|
# ---- the token stream (parallel slices) ----
|
|
var src: ptr = null
|
|
var tk_kind: []int
|
|
var tk_start: []int
|
|
var tk_end: []int
|
|
var tk_line: []int
|
|
var linestart: []int
|
|
|
|
fn ntok() -> int { return len(tk_kind) }
|
|
fn tok_len(i: int) -> int { return tk_end[i] - tk_start[i] }
|
|
fn tok_text(i: int) -> ptr { return src[tk_start[i]..tk_end[i]] }
|
|
fn push_tok(k: int, st: int, en: int, ln: int) -> void {
|
|
push(tk_kind, k); push(tk_start, st); push(tk_end, en); push(tk_line, ln)
|
|
}
|
|
|
|
# ---- vocabulary classifiers ----
|
|
fn is_type_word(w: ptr) -> bool {
|
|
return (w == "int") or (w == "long") or (w == "fixed") or (w == "bool") or (w == "entity") or (w == "str") or (w == "ptr") or (w == "byte") or (w == "words") or (w == "fixeds") or (w == "ptrs") or (w == "void")
|
|
}
|
|
fn is_phase_word(w: ptr) -> bool {
|
|
return (w == "Start") or (w == "Input") or (w == "FixedUpdate") or (w == "Update") or (w == "LateUpdate") or (w == "Render")
|
|
}
|
|
fn is_keyword_word(w: ptr) -> bool {
|
|
if (w == "program") or (w == "import") or (w == "property") or (w == "model") or (w == "enum") or (w == "ui") { return true }
|
|
if (w == "const") or (w == "var") or (w == "fn") or (w == "extern") or (w == "handler") or (w == "entry") or (w == "event") or (w == "scene") { return true }
|
|
if (w == "phase") or (w == "query") or (w == "on") or (w == "cancellable") or (w == "public") or (w == "layer") or (w == "start") { return true }
|
|
if (w == "let") or (w == "return") or (w == "if") or (w == "else") or (w == "while") or (w == "for") or (w == "in") or (w == "spawn") or (w == "despawn") { return true }
|
|
if (w == "enable") or (w == "disable") or (w == "match") or (w == "machine") or (w == "state") or (w == "become") or (w == "where") { return true }
|
|
if (w == "and") or (w == "or") or (w == "not") or (w == "break") or (w == "continue") or (w == "new") or (w == "emit") or (w == "cancel") { return true }
|
|
return false
|
|
}
|
|
fn is_clause_word(w: ptr) -> bool {
|
|
return (w == "phase") or (w == "query") or (w == "reads") or (w == "writes") or (w == "needs") or (w == "uses") or (w == "requires") or (w == "ensures") or (w == "invariant") or (w == "effects")
|
|
}
|
|
|
|
# ---- the lexer: keeps comments, newlines, byte spans; never exits on bad input ----
|
|
fn is_op2(c0: int, c1: int) -> bool {
|
|
if c0 == 45 and c1 == 62 { return true } # ->
|
|
if c1 == 61 and (c0 == 43 or c0 == 45 or c0 == 42 or c0 == 47 or c0 == 61 or c0 == 33 or c0 == 60 or c0 == 62) { return true } # += -= *= /= == != <= >=
|
|
if c0 == 38 and c1 == 38 { return true } # &&
|
|
if c0 == 124 and c1 == 124 { return true } # ||
|
|
if c0 == 46 and c1 == 46 { return true } # ..
|
|
if c0 == 61 and c1 == 62 { return true } # =>
|
|
return false
|
|
}
|
|
fn is_op1(c: int) -> bool {
|
|
return c == 43 or c == 45 or c == 42 or c == 47 or c == 37 or c == 60 or c == 62 or c == 61 or c == 40 or c == 41 or c == 123 or c == 125 or c == 91 or c == 93 or c == 44 or c == 58 or c == 46 or c == 33 or c == 64 or c == 59
|
|
}
|
|
|
|
fn lex(s: ptr) -> void {
|
|
src = s
|
|
tk_kind = new []int; tk_start = new []int; tk_end = new []int; tk_line = new []int
|
|
linestart = new []int
|
|
push(linestart, 0)
|
|
var i = 0; var line = 0
|
|
while s[i] != 0 {
|
|
let c = s[i]
|
|
if c == 10 { push_tok(LT_NL, i, i + 1, line); i = i + 1; line = line + 1; push(linestart, i); continue }
|
|
if c == 32 or c == 9 or c == 13 { i = i + 1; continue }
|
|
if c == 35 { # '#' comment to end of line
|
|
let st = i; while s[i] != 0 and s[i] != 10 { i = i + 1 }; push_tok(LT_COMMENT, st, i, line); continue
|
|
}
|
|
if c == 34 { # "string"
|
|
let st = i; i = i + 1
|
|
while s[i] != 0 and s[i] != 34 and s[i] != 10 { if s[i] == 92 and s[i + 1] != 0 { i = i + 2 } else { i = i + 1 } }
|
|
if s[i] == 34 { i = i + 1 }
|
|
push_tok(LT_STR, st, i, line); continue
|
|
}
|
|
if c == 96 { # `interpolated`
|
|
let st = i; i = i + 1
|
|
while s[i] != 0 and s[i] != 96 { if s[i] == 92 and s[i + 1] != 0 { i = i + 2 } else { i = i + 1 } }
|
|
if s[i] == 96 { i = i + 1 }
|
|
push_tok(LT_STR, st, i, line); continue
|
|
}
|
|
if c == 39 { # 'c'
|
|
let st = i; i = i + 1
|
|
if s[i] == 92 and s[i + 1] != 0 { i = i + 2 } else { if s[i] != 0 and s[i] != 10 { i = i + 1 } }
|
|
if s[i] == 39 { i = i + 1 }
|
|
push_tok(LT_CHAR, st, i, line); continue
|
|
}
|
|
if char_is_digit(c) {
|
|
let st = i
|
|
if c == 48 and (s[i + 1] == 120 or s[i + 1] == 88) { # 0x hex
|
|
i = i + 2; while char_is_hex(s[i]) { i = i + 1 }; push_tok(LT_INT, st, i, line); continue
|
|
}
|
|
while char_is_digit(s[i]) { i = i + 1 }
|
|
if s[i] == 46 and char_is_digit(s[i + 1]) {
|
|
i = i + 1; while char_is_digit(s[i]) { i = i + 1 }; push_tok(LT_FLOAT, st, i, line); continue
|
|
}
|
|
push_tok(LT_INT, st, i, line); continue
|
|
}
|
|
if c == 64 and (char_is_alpha(s[i + 1]) or s[i + 1] == 95) { # @name annotation
|
|
let st = i; i = i + 1; while char_is_alnum(s[i]) { i = i + 1 }; push_tok(LT_ANNO, st, i, line); continue
|
|
}
|
|
if char_is_alpha(c) {
|
|
let st = i; while char_is_alnum(s[i]) { i = i + 1 }
|
|
let w = s[st..i]
|
|
var k = LT_ID
|
|
if (w == "true") or (w == "false") or (w == "null") { k = LT_BOOL }
|
|
else { if is_type_word(w) { k = LT_TYPE }
|
|
else { if is_phase_word(w) { k = LT_PHASE }
|
|
else { if is_keyword_word(w) { k = LT_KW } } } }
|
|
push_tok(k, st, i, line); continue
|
|
}
|
|
if is_op2(c, s[i + 1]) { push_tok(LT_OP, i, i + 2, line); i = i + 2; continue }
|
|
if is_op1(c) { push_tok(LT_OP, i, i + 1, line); i = i + 1; continue }
|
|
# anything else: one UTF-8 character's worth as an LT_ERR token
|
|
var ln = 1
|
|
if c >= 240 { ln = 4 } else { if c >= 224 { ln = 3 } else { if c >= 128 { ln = 2 } } }
|
|
var kk = 1
|
|
while kk < ln { if s[i + kk] == 0 or (s[i + kk] & 192) != 128 { ln = kk }; kk = kk + 1 }
|
|
push_tok(LT_ERR, i, i + ln, line); i = i + ln
|
|
}
|
|
push_tok(LT_EOF, i, i, line)
|
|
}
|
|
|
|
# ---- token-stream helpers ----
|
|
fn next_sig(i: int) -> int {
|
|
var j = i + 1
|
|
while j < ntok() { let k = tk_kind[j]; if k != LT_NL and k != LT_COMMENT { return j }; j = j + 1 }
|
|
return 0 - 1
|
|
}
|
|
fn prev_sig(i: int) -> int {
|
|
var j = i - 1
|
|
while j >= 0 { let k = tk_kind[j]; if k != LT_NL and k != LT_COMMENT { return j }; j = j - 1 }
|
|
return 0 - 1
|
|
}
|
|
fn name_like(k: int) -> bool { return k == LT_ID or k == LT_KW or k == LT_TYPE or k == LT_PHASE or k == LT_BOOL }
|
|
# a '.' hugs an operand, a closing bracket, and another dot
|
|
fn dot_tight(t: int) -> bool {
|
|
if name_like(tk_kind[t]) { return true }
|
|
if tk_kind[t] != LT_OP { return false }
|
|
let n = tok_len(t); let c0 = src[tk_start[t]]
|
|
if n == 1 and (c0 == 41 or c0 == 93 or c0 == 46) { return true }
|
|
if n == 2 and c0 == 46 and src[tk_start[t] + 1] == 46 { return true }
|
|
return false
|
|
}
|
|
# is the token at index i a unary '-'/'!' rather than a binary operator?
|
|
fn is_unary(i: int) -> bool {
|
|
if tk_kind[i] != LT_OP { return false }
|
|
let n = tok_len(i)
|
|
if not (n == 1 and (src[tk_start[i]] == 45 or src[tk_start[i]] == 33)) { return false }
|
|
let p = prev_sig(i)
|
|
if p < 0 { return true }
|
|
let pk = tk_kind[p]
|
|
if pk == LT_ID or pk == LT_INT or pk == LT_FLOAT or pk == LT_STR or pk == LT_CHAR or pk == LT_BOOL or pk == LT_TYPE or pk == LT_PHASE { return false }
|
|
if pk == LT_OP {
|
|
let c = src[tk_start[p]]
|
|
return not (tok_len(p) == 1 and (c == 41 or c == 93 or c == 125)) # a closing bracket ends an operand
|
|
}
|
|
return true # keyword/annotation/comment: operand starts here
|
|
}
|
|
|
|
# whitespace between the previous emitted token (prev) and the current one (cur)
|
|
fn space_before(prev: int, cur: int) -> int {
|
|
if prev < 0 { return 0 }
|
|
let pk = tk_kind[prev]; let ck = tk_kind[cur]
|
|
let plen = tok_len(prev); let clen = tok_len(cur)
|
|
let p0 = src[tk_start[prev]]; let c0 = src[tk_start[cur]]
|
|
let p1 = (plen == 1); let c1 = (clen == 1)
|
|
if pk == LT_ERR or ck == LT_ERR { return tk_start[cur] - tk_end[prev] } # preserve an error token's spacing
|
|
if c1 and (c0 == 41 or c0 == 93 or c0 == 44 or c0 == 58 or c0 == 59) { return 0 } # ) ] , : ;
|
|
if c1 and c0 == 46 and ck == LT_OP and dot_tight(prev) { return 0 }
|
|
if p1 and p0 == 46 and pk == LT_OP and dot_tight(cur) { return 0 }
|
|
if p1 and (p0 == 40 or p0 == 91) and pk == LT_OP { return 0 } # nothing hugs an opener from the right
|
|
if pk == LT_OP and is_unary(prev) { return 0 }
|
|
if pk == LT_ANNO and c1 and c0 == 40 { return 0 } # @anno(
|
|
if c1 and c0 == 40 and ck == LT_OP {
|
|
if pk == LT_ID or pk == LT_TYPE or pk == LT_PHASE { return 0 } # fn move( / clear(
|
|
if pk == LT_OP { if p1 and (p0 == 41 or p0 == 93) { return 1 }; return 0 }
|
|
return 1
|
|
}
|
|
return 1
|
|
}
|
|
|
|
# ---- the formatting pass ----
|
|
fn format(indent_width: int) -> Buf {
|
|
let o = buf_new()
|
|
let stack = words(600)
|
|
let hang = words(600)
|
|
var sp = 0
|
|
var cur = 0
|
|
var open = 0
|
|
var brack = 0
|
|
var ui_depth = 0 - 1
|
|
var pending_blank = 0
|
|
var wrote_any = 0
|
|
var prev_line_had_comment = 0
|
|
let N = ntok()
|
|
var i = 0
|
|
while i < N and tk_kind[i] != LT_EOF {
|
|
let a = i
|
|
while i < N and tk_kind[i] != LT_NL and tk_kind[i] != LT_EOF { i = i + 1 }
|
|
let b = i
|
|
if i < N and tk_kind[i] == LT_NL { i = i + 1 }
|
|
|
|
if a == b { # a blank line
|
|
if wrote_any == 1 { pending_blank = 1 }
|
|
continue
|
|
}
|
|
|
|
# where does this line start?
|
|
var line_level = cur
|
|
var tsp = sp
|
|
var t = a
|
|
while t < b and tk_kind[t] == LT_OP and tok_len(t) == 1 and src[tk_start[t]] == 125 { # leading '}'
|
|
if tsp > 0 { tsp = tsp - 1; line_level = stack[tsp] }
|
|
t = t + 1
|
|
}
|
|
# original column of this line
|
|
var orig_ind = 0
|
|
var kk = linestart[tk_line[a]]
|
|
while kk < tk_start[a] { if src[kk] == 9 { orig_ind = orig_ind + 4 } else { orig_ind = orig_ind + 1 }; kk = kk + 1 }
|
|
# a comment continuing the previous comment line keeps the author's column
|
|
var comment_run = 0
|
|
if tk_kind[a] == LT_COMMENT and b == a + 1 and prev_line_had_comment == 1 { comment_run = 1 }
|
|
|
|
var ind = 0
|
|
var hangprev = 0
|
|
if sp > 0 { hangprev = hang[sp - 1] }
|
|
if open > 0 or (sp > 0 and hangprev == 1) { # author owns alignment inside open call/brace
|
|
let ls = linestart[tk_line[a]]
|
|
ind = 0
|
|
var k2 = ls
|
|
while k2 < tk_start[a] { if src[k2] == 9 { ind = ind + 4 } else { ind = ind + 1 }; k2 = k2 + 1 }
|
|
} else {
|
|
var extra = 0
|
|
if is_clause_word(tok_text(a)) and tk_kind[a] == LT_KW { extra = indent_width }
|
|
ind = line_level * indent_width + extra
|
|
if comment_run == 1 and orig_ind > ind { ind = orig_ind }
|
|
}
|
|
|
|
if pending_blank == 1 and wrote_any == 1 { buf_putc(o, 10) }
|
|
pending_blank = 0
|
|
buf_indent(o, ind)
|
|
|
|
# emit the tokens
|
|
var prev = 0 - 1
|
|
var line_brack = brack
|
|
var line_ui_open = 0
|
|
if ui_depth >= 0 and sp > ui_depth { line_ui_open = 1 }
|
|
t = a
|
|
while t < b {
|
|
var want = 0
|
|
if tk_kind[t] == LT_COMMENT { if prev >= 0 { want = 2 } else { want = 0 } }
|
|
else {
|
|
want = space_before(prev, t)
|
|
if line_brack > 0 { # query {Tag} filter is one word
|
|
if tk_kind[t] == LT_OP and tok_len(t) == 1 and src[tk_start[t]] == 125 { want = 0 }
|
|
if prev >= 0 and tk_kind[prev] == LT_OP and tok_len(prev) == 1 and src[tk_start[prev]] == 123 { want = 0 }
|
|
}
|
|
if line_ui_open == 1 { # widget props are k=v
|
|
var eq_here = 0
|
|
if tk_kind[t] == LT_OP and tok_len(t) == 1 and src[tk_start[t]] == 61 { eq_here = 1 }
|
|
var eq_prev = 0
|
|
if prev >= 0 and tk_kind[prev] == LT_OP and tok_len(prev) == 1 and src[tk_start[prev]] == 61 { eq_prev = 1 }
|
|
if eq_here == 1 or eq_prev == 1 { want = 0 }
|
|
}
|
|
}
|
|
var gap = 0
|
|
if prev >= 0 { gap = tk_start[t] - tk_end[prev] }
|
|
if gap >= 2 and want >= 1 { if gap > 60 { gap = 60 }; want = gap } # hand alignment wins
|
|
if want < 0 { want = 0 }
|
|
var w2 = 0
|
|
while w2 < want { buf_putc(o, 32); w2 = w2 + 1 }
|
|
buf_addrange(o, src, tk_start[t], tk_end[t])
|
|
prev = t
|
|
if tk_kind[t] == LT_OP and tok_len(t) == 1 {
|
|
let c = src[tk_start[t]]
|
|
if c == 91 { line_brack = line_brack + 1 }
|
|
else { if c == 93 { line_brack = line_brack - 1; if line_brack < 0 { line_brack = 0 } } }
|
|
}
|
|
t = t + 1
|
|
}
|
|
buf_putc(o, 10)
|
|
wrote_any = 1
|
|
prev_line_had_comment = 0
|
|
if prev >= 0 and tk_kind[prev] == LT_COMMENT { prev_line_had_comment = 1 }
|
|
|
|
# carry the nesting into the next line
|
|
if tk_kind[a] == LT_KW and (tok_text(a) == "ui") and ui_depth < 0 { ui_depth = sp }
|
|
var level = cur
|
|
t = a
|
|
while t < b {
|
|
if tk_kind[t] == LT_OP and tok_len(t) == 1 {
|
|
let c = src[tk_start[t]]
|
|
if c == 123 { # '{'
|
|
if sp < 512 {
|
|
let nxt = next_sig(t)
|
|
stack[sp] = line_level
|
|
var h = 0
|
|
if nxt >= 0 and nxt < b { h = 1 }
|
|
hang[sp] = h
|
|
sp = sp + 1
|
|
}
|
|
level = line_level + 1
|
|
}
|
|
else { if c == 125 { if sp > 0 { sp = sp - 1; level = stack[sp] } }
|
|
else { if c == 40 or c == 91 { open = open + 1; if c == 91 { brack = brack + 1 } }
|
|
else { if c == 41 or c == 93 { open = open - 1; if open < 0 { open = 0 }; if c == 93 { brack = brack - 1; if brack < 0 { brack = 0 } } } } } }
|
|
}
|
|
t = t + 1
|
|
}
|
|
cur = level
|
|
if ui_depth >= 0 and sp <= ui_depth { ui_depth = 0 - 1 }
|
|
}
|
|
return o
|
|
}
|
|
|
|
# ---- markdown: format the body of every ```ludic fence, leave prose alone ----
|
|
var g_flen: int = 0
|
|
var g_info: int = 0
|
|
var g_marker: int = 0
|
|
# detect a fence at line offset i; sets g_flen/g_info/g_marker; returns bool
|
|
fn md_fence_at(s: ptr, i: int) -> bool {
|
|
var j = i; while s[j] == 32 { j = j + 1 }
|
|
let m = s[j]
|
|
if m != 96 and m != 126 { return false } # ` or ~
|
|
var n = 0; while s[j] == m { j = j + 1; n = n + 1 }
|
|
if n < 3 { return false }
|
|
g_flen = n; g_info = j; g_marker = m
|
|
return true
|
|
}
|
|
fn md_info_is_ludic(s: ptr, at: int) -> bool {
|
|
var a = at; while s[a] == 32 or s[a] == 9 { a = a + 1 }
|
|
# case-insensitive "ludic"
|
|
if not (((s[a] == 108 or s[a] == 76)) and ((s[a + 1] == 117 or s[a + 1] == 85)) and ((s[a + 2] == 100 or s[a + 2] == 68)) and ((s[a + 3] == 105 or s[a + 3] == 73)) and ((s[a + 4] == 99 or s[a + 4] == 67))) { return false }
|
|
let af = s[a + 5]
|
|
return af == 0 or af == 10 or af == 32 or af == 9 or af == 13
|
|
}
|
|
fn line_end(s: ptr, i: int) -> int { var e = i; while s[e] != 0 and s[e] != 10 { e = e + 1 }; return e }
|
|
|
|
fn format_markdown(s: ptr, indent_width: int) -> Buf {
|
|
let o = buf_new()
|
|
var i = 0
|
|
while s[i] != 0 {
|
|
let ls = i
|
|
let le = line_end(s, ls)
|
|
var indent = 0; while s[ls + indent] == 32 { indent = indent + 1 }
|
|
if md_fence_at(s, ls) and md_info_is_ludic(s, g_info) {
|
|
# copy the opening fence line verbatim (with its newline)
|
|
var e0 = le; if s[le] != 0 { e0 = le + 1 }
|
|
buf_addrange(o, s, ls, e0)
|
|
i = e0
|
|
# gather the body up to the closing fence
|
|
let bs = i
|
|
var be = bs
|
|
while true {
|
|
if s[be] == 0 { break }
|
|
let ps = be; let pe = line_end(s, ps)
|
|
if md_fence_at(s, ps) and g_marker == g_marker and g_flen >= g_flen {
|
|
# re-run detection for THIS line (md_fence_at set globals for ps)
|
|
let cmark = g_marker; let clen = g_flen; let cinfo = g_info
|
|
if md_fence_at(s, ps) {
|
|
if g_marker == cmark and g_flen >= clen {
|
|
var only = 1
|
|
var k = g_info
|
|
while k < pe { if s[k] != 32 and s[k] != 13 { only = 0; break }; k = k + 1 }
|
|
if only == 1 { be = ps; break }
|
|
}
|
|
}
|
|
}
|
|
if s[pe] != 0 { be = pe + 1 } else { be = pe }
|
|
}
|
|
# de-indent the body, format it, re-indent it
|
|
let body = buf_new()
|
|
var p = bs
|
|
while p < be {
|
|
var q = line_end(s, p)
|
|
var skip = 0
|
|
while skip < indent and (p + skip) < q and s[p + skip] == 32 { skip = skip + 1 }
|
|
buf_addrange(body, s, p + skip, q)
|
|
buf_putc(body, 10)
|
|
if s[q] != 0 { p = q + 1 } else { p = q }
|
|
}
|
|
# feed the body through the ludic formatter
|
|
lex(buf_str(body))
|
|
let f = format(indent_width)
|
|
let ftext = buf_str(f)
|
|
var fp = 0
|
|
while ftext[fp] != 0 {
|
|
var fq = fp; while ftext[fq] != 0 and ftext[fq] != 10 { fq = fq + 1 }
|
|
if fq > fp { buf_indent(o, indent) }
|
|
buf_addrange(o, ftext, fp, fq)
|
|
buf_putc(o, 10)
|
|
if ftext[fq] != 0 { fp = fq + 1 } else { fp = fq }
|
|
}
|
|
i = be
|
|
continue
|
|
}
|
|
var e1 = le; if s[le] != 0 { e1 = le + 1 }
|
|
buf_addrange(o, s, ls, e1)
|
|
i = e1
|
|
}
|
|
return o
|
|
}
|
|
|
|
# ---- ends-with helpers for extension detection ----
|
|
fn ends_with(s: ptr, suf: ptr) -> bool {
|
|
let n = cstr_len(s); let m = cstr_len(suf)
|
|
if m > n { return false }
|
|
return (s[n - m..n] == suf)
|
|
}
|
|
|
|
# ---- stdin slurp (for `-`) ----
|
|
fn slurp_stdin() -> ptr {
|
|
let b = buf_new()
|
|
var c = read_char()
|
|
while c >= 0 { buf_putc(b, c); c = read_char() }
|
|
return buf_str(b)
|
|
}
|
|
|
|
# format one source string according to its kind (markdown vs ludic)
|
|
fn format_source(text: ptr, is_md: bool, indent_width: int) -> ptr {
|
|
if is_md { let m = format_markdown(text, indent_width); return buf_str(m) }
|
|
lex(text)
|
|
let f = format(indent_width)
|
|
return buf_str(f)
|
|
}
|
|
|
|
fn streq(a: ptr, b: ptr) -> bool { return (a == b) }
|
|
|
|
entry {
|
|
var write = false
|
|
var check = false
|
|
var indent = 2
|
|
var quiet = false
|
|
var changed = false
|
|
var failed = false
|
|
let files = new []ptr
|
|
var ai = 1
|
|
while ai < arg_count() {
|
|
let a = arg(ai)
|
|
if (a == "-w") or (a == "--write") { write = true }
|
|
else { if (a == "--check") or (a == "-l") { check = true }
|
|
else { if (a == "-q") or (a == "--quiet") { quiet = true }
|
|
else { if (a == "--indent") { ai = ai + 1; if ai < arg_count() { indent = 0; let d = arg(ai); var di = 0; while d[di] != 0 { indent = indent * 10 + (d[di] - 48); di = di + 1 } } }
|
|
else { if (a == "-h") or (a == "--help") { print("ludic-fmt — format Ludic source"); return }
|
|
else { push(files, a) } } } } }
|
|
ai = ai + 1
|
|
}
|
|
if indent < 1 or indent > 8 { indent = 2 }
|
|
|
|
# stdin filter mode
|
|
if len(files) == 0 or (len(files) == 1 and (files[0] == "-")) {
|
|
let text = slurp_stdin()
|
|
let out = format_source(text, false, indent)
|
|
file_write(file_stdout(), out, cstr_len(out))
|
|
return
|
|
}
|
|
|
|
var fi = 0
|
|
while fi < len(files) {
|
|
let path = files[fi]
|
|
let text = read_file(path)
|
|
if (text == null) {
|
|
let m = `ludic-fmt: cannot open {path}\n`
|
|
file_write(file_stderr(), m, cstr_len(m)); failed = true
|
|
} else {
|
|
let is_md = ends_with(path, ".md") or ends_with(path, ".markdown")
|
|
let out = format_source(text, is_md, indent)
|
|
let same = (out == text)
|
|
if check {
|
|
if not same { changed = true; if not quiet { print(path) } }
|
|
} else { if write {
|
|
if not same {
|
|
let f = file_open(path, "wb")
|
|
if (f == null) { let m = `ludic-fmt: cannot write {path}\n`; file_write(file_stderr(), m, cstr_len(m)); failed = true }
|
|
else { file_write(f, out, cstr_len(out)); file_close(f); if not quiet { let m = `formatted {path}\n`; file_write(file_stderr(), m, cstr_len(m)) } }
|
|
changed = true
|
|
}
|
|
} else {
|
|
file_write(file_stdout(), out, cstr_len(out))
|
|
} }
|
|
}
|
|
fi = fi + 1
|
|
}
|
|
if failed { exit(2) }
|
|
if check and changed { exit(1) }
|
|
}
|
|
}
|