--emit-schema FILE writes the compiler's resolved view once the program type-checks: every
record (fields, types, defaults as written, docs, places, attributes), every registry with its
entries in their final order after the open-registry merge (key, constant, index, file:line:col
of the entry and of each field value, and which file contributed which keys), every const, and
the zero-argument functions a fn value can name. Deterministic, schema_version 1; the runtime is
left out. `ludic schema [file] [-o FILE]` wraps it.
--check --diagnostics=json prints every error as one JSON array on stdout: the checker's and the
module rules' all, a parse or lowering error as the last. Tokens and nodes now carry a column.
Fields take several @attributes; @Ref(Registry), @OneOf(PREFIX_), @Range(lo, hi), @Unit("..."),
@Asset("..."), @Color on a field and @AppendOnly / @ByKey on a registry change nothing but go into
the schema, and @Ref naming no registry is an error (every one reported). Fixtures:
examples/lang/attributes.ludic, examples/rejected/ref_unknown.ludic, cases in ludic-dev test.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
272 lines
10 KiB
Text
272 lines
10 KiB
Text
# lex.ludic — source text -> a token slice. Tokens carry their kind, their
|
|
# text (identifiers, strings, operators), an integer value (numbers, char
|
|
# literals) and a line for diagnostics.
|
|
|
|
const TK_ID: int = 0
|
|
const TK_INT: int = 1
|
|
const TK_STR: int = 2
|
|
const TK_OP: int = 3
|
|
const TK_NL: int = 4
|
|
const TK_EOF: int = 5
|
|
const TK_FLOAT: int = 6
|
|
const TK_INTERP: int = 7 # `text {expr} text` — raw content, split by the parser
|
|
|
|
property Tok { kind: int = 0, text: pointer = null, ival: int = 0, line: int = 0, pos: int = -1, end: int = -1, col: int = 0, off: int = -1, oend: int = -1 }
|
|
# col: the token's column (1-based) on its line; off / oend: where it starts and ends in the text
|
|
# lexed, whatever the base (a resource file's own offsets, for the schema's spans - never a rewrite)
|
|
|
|
# where the tokens are in their file (0.S's migration rewrites source by position): a file's own
|
|
# text starts at 0, an interpolation hole at its place in the file, and generated code has none (-1)
|
|
var g_lex_base: int = -1
|
|
var lx_start: int = 0
|
|
var lx_line0: int = 0 # where the line being lexed starts in the text
|
|
|
|
var toks: []Tok
|
|
|
|
function tok_push(kind: int, text: pointer, ival: int, line: int) -> void {
|
|
let t = new Tok
|
|
t.kind = kind; t.text = text; t.ival = ival; t.line = line
|
|
if g_lex_base >= 0 { t.pos = g_lex_base + lx_start }
|
|
t.off = lx_start
|
|
t.col = lx_start - lx_line0 + 1
|
|
push(toks, t)
|
|
}
|
|
|
|
# does src match the 2-char operator op at position i?
|
|
function two_at(src: pointer, i: int, a: int, b: int) -> bool {
|
|
return src[i] == a and src[i + 1] == b
|
|
}
|
|
|
|
function is_op1(c: int) -> bool {
|
|
# + - * / % < > = ( ) { } [ ] , : . ! @
|
|
if c == '+' or c == '-' or c == '*' or c == '/' or c == '%' { return true }
|
|
if c == '<' or c == '>' or c == '=' { return true }
|
|
if c == '(' or c == ')' or c == '{' or c == '}' { return true }
|
|
if c == '[' or c == ']' or c == ',' or c == ':' or c == '.' { return true }
|
|
if c == '!' or c == '@' { return true }
|
|
if c == '&' or c == '|' or c == '^' or c == '~' { return true } # & | ^ ~
|
|
return false
|
|
}
|
|
|
|
# The two bytes that delimit escapes, spelled numerically on purpose: this file
|
|
# is what teaches the compiler to read `'\''` and `'\\'`, and the seed that
|
|
# bootstraps it must lex it without already knowing those spellings.
|
|
const CH_SQUOTE: int = 39 # '
|
|
const CH_BACKSLASH: int = 92 # \
|
|
|
|
# the byte an escape sequence `\e` stands for, shared by "strings" and 'chars':
|
|
# \n \r \t \0 are the named ones; anything else (\\ \' \" \`) is itself.
|
|
function unescape(e: int) -> int {
|
|
if e == 'n' { return 10 }
|
|
if e == 'r' { return 13 }
|
|
if e == 't' { return 9 }
|
|
if e == '0' { return 0 }
|
|
return e
|
|
}
|
|
|
|
# a character the language has no use for is an error, never silently dropped:
|
|
# `x = 5 $ 3` must not compile as `x = 5 3`.
|
|
function lex_error(line: int, c: int) -> void {
|
|
let ch = bytes(2); ch[0] = c; ch[1] = 0
|
|
lex_fail(line, `unexpected character '{ch}' (byte {c})`)
|
|
}
|
|
|
|
# the closing backtick of the template literal that opens at s[i]: a hole's `{...}` is code, so a
|
|
# string, a char or another template inside it is skipped whole - `a {f(`b {c}`)} d` is one literal
|
|
function tmpl_end(s: pointer, i0: int, n: int) -> int {
|
|
var i = i0 + 1
|
|
while i < n and s[i] != '`' {
|
|
if s[i] == CH_BACKSLASH { i += 2; continue }
|
|
if s[i] == '{' and i + 1 < n and s[i + 1] == '{' { i += 2; continue }
|
|
if s[i] == '{' {
|
|
i = hole_end(s, i + 1, n)
|
|
}
|
|
i += 1
|
|
}
|
|
return i
|
|
}
|
|
# the `}` that closes a hole whose code starts at s[i]; n when there is none
|
|
function hole_end(s: pointer, i0: int, n: int) -> int {
|
|
var i = i0
|
|
var depth = 1
|
|
while i < n {
|
|
let d = s[i]
|
|
if d == '"' or d == CH_SQUOTE {
|
|
i += 1
|
|
while i < n and s[i] != d { if s[i] == CH_BACKSLASH { i += 1 }; i += 1 }
|
|
}
|
|
else if d == '`' { i = tmpl_end(s, i, n) }
|
|
else if d == '{' { depth += 1 }
|
|
else if d == '}' {
|
|
depth -= 1
|
|
if depth == 0 { return i }
|
|
}
|
|
i += 1
|
|
}
|
|
return n
|
|
}
|
|
|
|
# report a lexical error in the standard `file:line: error: msg` shape and stop
|
|
function lex_fail(line: int, msg: pointer) -> void {
|
|
if g_diag_json {
|
|
diag_add(g_parse_file, line, lx_start - lx_line0 + 1, "error", msg)
|
|
diag_exit(1)
|
|
}
|
|
let m = `{g_parse_file}:{line}: error: {msg}\n`
|
|
file_write(file_stderr(), m, len(m))
|
|
exit(1)
|
|
}
|
|
|
|
function lex(src: pointer) -> void {
|
|
let saved = g_lex_base
|
|
g_lex_base = 0
|
|
lex_at(src, 1)
|
|
g_lex_base = saved
|
|
}
|
|
|
|
# lex `src` with its first line numbered `first_line` (an interpolation hole is
|
|
# re-lexed on its own, and keeps the line of the string it sits in)
|
|
function lex_at(src: pointer, first_line: int) -> void {
|
|
toks = new []Tok
|
|
lx_line0 = 0
|
|
var i = 0
|
|
var line = first_line
|
|
let n = len(src)
|
|
var lx_n = 0
|
|
while i < n {
|
|
# the token the last turn pushed ends where this one starts looking
|
|
if len(toks) > lx_n {
|
|
if g_lex_base >= 0 { toks[len(toks) - 1].end = g_lex_base + i }
|
|
toks[len(toks) - 1].oend = i
|
|
lx_n = len(toks)
|
|
}
|
|
lx_start = i
|
|
let c = src[i]
|
|
if c == '\n' { tok_push(TK_NL, null, 0, line); line += 1; i += 1; lx_line0 = i; continue }
|
|
if c == ' ' or c == '\t' or c == '\r' { i += 1; continue }
|
|
if c == '#' { # '#' comment to end of line
|
|
while i < n and src[i] != '\n' { i += 1 }
|
|
continue
|
|
}
|
|
if c == '"' { # "string"
|
|
i += 1
|
|
let start = i
|
|
let out = bytes(n)
|
|
var j = 0
|
|
while i < n and src[i] != '"' {
|
|
if src[i] == CH_BACKSLASH { # backslash escape
|
|
out[j] = unescape(src[i + 1]); j += 1; i += 2
|
|
} else { out[j] = src[i]; j += 1; i += 1 }
|
|
}
|
|
i += 1
|
|
out[j] = 0
|
|
tok_push(TK_STR, out, 0, line)
|
|
continue
|
|
}
|
|
if c == '`' { # `interpolated string` — captured raw
|
|
let e = tmpl_end(src, i, n)
|
|
let out = bytes(n)
|
|
var j = 0
|
|
i += 1
|
|
while i < e {
|
|
out[j] = src[i]
|
|
j += 1
|
|
i += 1
|
|
}
|
|
i = e + 1
|
|
out[j] = 0
|
|
tok_push(TK_INTERP, out, 0, line)
|
|
continue
|
|
}
|
|
if c == CH_SQUOTE { # 'c' char literal -> int
|
|
i += 1
|
|
var v = 0
|
|
if src[i] == CH_BACKSLASH { v = unescape(src[i + 1]); i += 2 }
|
|
else { v = src[i]; i += 1 }
|
|
if src[i] != CH_SQUOTE { lex_fail(line, "unterminated character literal (expected closing ')") }
|
|
i += 1
|
|
tok_push(TK_INT, null, v, line)
|
|
continue
|
|
}
|
|
if char_is_digit(c) {
|
|
if c == '0' and src[i + 1] == 'x' { # 0x hex
|
|
var v = 0
|
|
var lv: long = 0
|
|
var digits = 0
|
|
i += 2
|
|
while i < n {
|
|
let h = src[i]
|
|
var d = 0
|
|
if char_is_digit(h) { d = h - 48 }
|
|
else { if h >= 'a' and h <= 'f' { d = h - 87 }
|
|
else { if h >= 'A' and h <= 'F' { d = h - 55 } else { break } } }
|
|
v = v * 16 + d
|
|
lv = lv * long(16) + long(d)
|
|
if digits > 0 or d > 0 { digits += 1 }
|
|
i += 1
|
|
}
|
|
# up to 8 hex digits is a 32-bit pattern (0xFFFFFFFF is -1, as it always was); more is a long
|
|
var big: pointer = null
|
|
if digits > 8 { big = string(lv) }
|
|
tok_push(TK_INT, big, v, line)
|
|
continue
|
|
}
|
|
var v = 0
|
|
var lv: long = 0
|
|
let nstart = i
|
|
while i < n and char_is_digit(src[i]) {
|
|
v = v * 10 + (src[i] - 48)
|
|
lv = lv * long(10) + long(src[i] - 48)
|
|
i += 1
|
|
}
|
|
# a fractional part makes it a Q16.16 fixed literal (its text is kept: in a
|
|
# float context the literal is exactly that decimal instead)
|
|
if i < n and src[i] == '.' and char_is_digit(src[i + 1]) {
|
|
i += 1
|
|
# Accumulate only the first 4 fractional digits: `fnum << 16` must stay
|
|
# in i32 (5+ digits overflow), and Q16.16 resolves ~4-5 decimals anyway.
|
|
# Extra digits are still consumed so they don't become a stray token.
|
|
var fnum = 0; var fden = 1
|
|
while i < n and char_is_digit(src[i]) {
|
|
if fden < 10000 { fnum = fnum * 10 + (src[i] - 48); fden *= 10 }
|
|
i += 1
|
|
}
|
|
let bits = (v << 16) + ((fnum << 16) + (fden >> 1)) / fden
|
|
tok_push(TK_FLOAT, src[nstart..i], bits, line)
|
|
continue
|
|
}
|
|
# a decimal past 2^31 - 1 does not fit an int: it is a long, and keeps its value as text
|
|
var big: pointer = null
|
|
if lv > long(2147483647) { big = string(lv) }
|
|
tok_push(TK_INT, big, v, line)
|
|
continue
|
|
}
|
|
if char_is_alpha(c) {
|
|
let start = i
|
|
while i < n and char_is_alnum(src[i]) { i += 1 }
|
|
tok_push(TK_ID, src[start..i], 0, line)
|
|
continue
|
|
}
|
|
if c == ';' { tok_push(TK_NL, null, 0, line); i += 1; continue } # ';'
|
|
# two-character operators
|
|
if two_at(src, i, 45, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # ->
|
|
if two_at(src, i, 61, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # =>
|
|
if two_at(src, i, 61, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # ==
|
|
if two_at(src, i, 33, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # !=
|
|
if two_at(src, i, 60, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # <=
|
|
if two_at(src, i, 62, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # >=
|
|
if two_at(src, i, 43, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # +=
|
|
if two_at(src, i, 45, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # -=
|
|
if two_at(src, i, 42, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # *=
|
|
if two_at(src, i, 47, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # /=
|
|
if two_at(src, i, 46, 46) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # ..
|
|
if two_at(src, i, 60, 60) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # <<
|
|
if two_at(src, i, 62, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # >>
|
|
if is_op1(c) { tok_push(TK_OP, src[i..i + 1], 0, line); i += 1; continue }
|
|
lex_error(line, c)
|
|
}
|
|
if len(toks) > lx_n and g_lex_base >= 0 { toks[len(toks) - 1].end = g_lex_base + i }
|
|
if len(toks) > lx_n { toks[len(toks) - 1].oend = i }
|
|
lx_start = i
|
|
tok_push(TK_EOF, null, 0, line)
|
|
}
|