ludic/selfhost/frontend/lex.ludic
Orkuncakilkaya 23726afa90
All checks were successful
bootstrap / cfree-fixpoint (push) Successful in 12s
ci / build-and-test (push) Successful in 50s
commit-lint / conventional-commits (push) Successful in 3s
docs / build-and-deploy (push) Successful in 2s
refactor(selfhost): reorganise into concern-based subdirectories
Split the flat 38-file selfhost/ into concern-based subdirectories:

  frontend/        lex, parse, parse_game, ast
  support/         str, buf, io
  backend/         core IR + expression/statement lowering
  backend/game/    ECS/scene/event/world lowering
  backend/stdlib/  the namespaced Math.*/Text.*/Crypto.*/… intrinsics

and split the three oversized emitters at responsibility boundaries so
no file mixes concerns:

  emit_game.ludic  -> + emit_world.ludic         (reflection world table,
                                                  tick helpers, @main synthesis)
  emit_expr.ludic  -> + emit_call.ludic          (namespaced builtins, call
                                                  lowering, expr dispatch)
  emit_text.ludic  -> + emit_text_prelude.ludic  (emitted string-builder runtime)

FRAGS in tools/x/selfhost.ludic is updated to the new paths with the link
order preserved, and the Python doc/vocabulary tooling is updated to walk
the new layout. Because the build is a plain in-order concatenation and
every split lands on a blank-line boundary, the regenerated seed is
byte-identical: `x reseed` leaves selfhost/ludicc.seed.ll unchanged,
`x bootstrap-cfree` still reaches its fixed point, and both `x test` (56)
and `x selfhost-test` (29, incl. golden renders) stay green.

Closes #29

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-08-31 00:26:02 +03:00

158 lines
6.1 KiB
Text

# lex.ludic — source text -> a token slice. Mirrors compiler/front/lex.c.
# Tokens carry their kind, their text (identifiers, strings, operators),
# an integer value (numbers, char literals) and a line for diagnostics.
const TK_ID: int = 0
const TK_INT: int = 1
const TK_STR: int = 2
const TK_OP: int = 3
const TK_NL: int = 4
const TK_EOF: int = 5
const TK_FLOAT: int = 6
const TK_INTERP: int = 7 # `text {expr} text` — raw content, split by the parser
property Tok { kind: int = 0, text: pointer = null, ival: int = 0, line: int = 0 }
var toks: []Tok
function tok_push(kind: int, text: pointer, ival: int, line: int) -> void {
let t = new Tok
t.kind = kind; t.text = text; t.ival = ival; t.line = line
push(toks, t)
}
# does src match the 2-char operator op at position i?
function two_at(src: pointer, i: int, a: int, b: int) -> bool {
return src[i] == a and src[i + 1] == b
}
function is_op1(c: int) -> bool {
# + - * / % < > = ( ) { } [ ] , : . ! @
if c == 43 or c == 45 or c == 42 or c == 47 or c == 37 { return true }
if c == 60 or c == 62 or c == 61 { return true }
if c == 40 or c == 41 or c == 123 or c == 125 { return true }
if c == 91 or c == 93 or c == 44 or c == 58 or c == 46 { return true }
if c == 33 or c == 64 { return true }
if c == 38 or c == 124 or c == 94 or c == 126 { return true } # & | ^ ~
return false
}
function lex(src: pointer) -> void {
toks = new []Tok
var i = 0
var line = 1
let n = len(src)
while i < n {
let c = src[i]
if c == 10 { tok_push(TK_NL, null, 0, line); line = line + 1; i = i + 1; continue }
if c == 32 or c == 9 or c == 13 { i = i + 1; continue }
if c == 35 { # '#' comment to end of line
while i < n and src[i] != 10 { i = i + 1 }
continue
}
if c == 34 { # "string"
i = i + 1
let start = i
let out = bytes(n)
var j = 0
while i < n and src[i] != 34 {
if src[i] == 92 { # backslash escape
let e = src[i + 1]
var r = e
if e == 110 { r = 10 }
if e == 116 { r = 9 }
if e == 48 { r = 0 }
out[j] = r; j = j + 1; i = i + 2
} else { out[j] = src[i]; j = j + 1; i = i + 1 }
}
i = i + 1
out[j] = 0
tok_push(TK_STR, out, 0, line)
continue
}
if c == 96 { # `interpolated string` — captured raw
i = i + 1
let out = bytes(n)
var j = 0
while i < n and src[i] != 96 { out[j] = src[i]; j = j + 1; i = i + 1 }
i = i + 1
out[j] = 0
tok_push(TK_INTERP, out, 0, line)
continue
}
if c == 39 { # 'c' char literal -> int
i = i + 1
var v = 0
if src[i] == 92 {
let e = src[i + 1]
if e == 110 { v = 10 }
if e == 116 { v = 9 }
if e == 48 { v = 0 }
i = i + 2
} else { v = src[i]; i = i + 1 }
if src[i] == 39 { i = i + 1 }
tok_push(TK_INT, null, v, line)
continue
}
if char_is_digit(c) {
if c == 48 and src[i + 1] == 120 { # 0x hex
var v = 0
i = i + 2
while i < n {
let h = src[i]
var d = 0
if char_is_digit(h) { d = h - 48 }
else { if h >= 97 and h <= 102 { d = h - 87 }
else { if h >= 65 and h <= 70 { d = h - 55 } else { break } } }
v = v * 16 + d
i = i + 1
}
tok_push(TK_INT, null, v, line)
continue
}
var v = 0
while i < n and char_is_digit(src[i]) { v = v * 10 + (src[i] - 48); i = i + 1 }
# a fractional part makes it a Q16.16 fixed literal
if i < n and src[i] == 46 and char_is_digit(src[i + 1]) {
i = i + 1
# Accumulate only the first 4 fractional digits: `fnum << 16` must stay
# in i32 (5+ digits overflow), and Q16.16 resolves ~4-5 decimals anyway.
# Extra digits are still consumed so they don't become a stray token.
var fnum = 0; var fden = 1
while i < n and char_is_digit(src[i]) {
if fden < 10000 { fnum = fnum * 10 + (src[i] - 48); fden = fden * 10 }
i = i + 1
}
let bits = (v << 16) + ((fnum << 16) + (fden >> 1)) / fden
tok_push(TK_FLOAT, null, bits, line)
continue
}
tok_push(TK_INT, null, v, line)
continue
}
if char_is_alpha(c) {
let start = i
while i < n and char_is_alnum(src[i]) { i = i + 1 }
tok_push(TK_ID, src[start..i], 0, line)
continue
}
if c == 59 { tok_push(TK_NL, null, 0, line); i = i + 1; continue } # ';'
# two-character operators
if two_at(src, i, 45, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # ->
if two_at(src, i, 61, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # =>
if two_at(src, i, 61, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # ==
if two_at(src, i, 33, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # !=
if two_at(src, i, 60, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # <=
if two_at(src, i, 62, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # >=
if two_at(src, i, 43, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # +=
if two_at(src, i, 45, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # -=
if two_at(src, i, 42, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # *=
if two_at(src, i, 47, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # /=
if two_at(src, i, 46, 46) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # ..
if two_at(src, i, 60, 60) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # <<
if two_at(src, i, 62, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # >>
if is_op1(c) { tok_push(TK_OP, src[i..i + 1], 0, line); i = i + 1; continue }
i = i + 1 # skip anything unrecognised
}
tok_push(TK_EOF, null, 0, line)
}