ludic/runtime/native/regex_vm.ludic
Orkuncakilkaya 57b66bdf47 feat(lang): L4 type checker between parse and emit
selfhost/check/ walks every function, the entry, tests, globals' initializers and
@On listeners with real scopes, and refuses mixed number kinds, text and numbers,
two record types, mismatched slices and fn types, wrong argument counts, wrong
returns and wrong push elements - every mix-up at once, each at its line.
LUDIC_CHECK_REPORT=1 lists them by category. pointer stays untyped (L7's).

What it found is fixed: render3d's HDR scan calling the float-bits extern f_lt
with floats; ludic.shooter's right-stick aim overflowing past half a push;
prof.ludic storing longs in []int; extern arguments now coerced to their
parameters. Text-returning runtime functions say string; Assets.ready says bool.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-24 01:34:49 +03:00

210 lines
7.1 KiB
Text

# ============================================================================
# regex_vm.ludic — the Pike VM (Thompson NFA execution) and the public Regex.*
# API for the engine compiled in regex.ludic. Splits the "run + query" concern
# out of regex.ludic's "parse + compile". Spliced together (regex.ludic imports
# this file), so the two share the Prog/RNode model and OP_* opcodes.
# ============================================================================
# ---- Pike VM ----------------------------------------------------------------
property TList { pc: words, caps: words, n: int }
var rx_seen: words = null
var rx_gen: int = 0
var rx_nc: int = 0
var rx_code: words = null
function rx_add(list: TList, pc: int, caps: words, sp: int, s: pointer, slen: int) -> void {
if rx_seen[pc] == rx_gen { return }
rx_seen[pc] = rx_gen
let op = rx_code[3 * pc]
if op == OP_JMP { rx_add(list, rx_code[3 * pc + 1], caps, sp, s, slen); return }
if op == OP_SPLIT {
rx_add(list, rx_code[3 * pc + 1], caps, sp, s, slen)
rx_add(list, rx_code[3 * pc + 2], caps, sp, s, slen)
return
}
if op == OP_SAVE {
let slot = rx_code[3 * pc + 1]
let old = caps[slot]
caps[slot] = sp
rx_add(list, pc + 1, caps, sp, s, slen)
caps[slot] = old
return
}
if op == OP_BOL {
if sp == 0 { rx_add(list, pc + 1, caps, sp, s, slen) }
return
}
if op == OP_EOL {
if sp == slen { rx_add(list, pc + 1, caps, sp, s, slen) }
else if sp == slen - 1 and (s[sp] & 255) == 10 { rx_add(list, pc + 1, caps, sp, s, slen) }
return
}
# a leaf that consumes (CHAR/ANY/ANYNL/CLASS) or MATCH: record it
let t = list.n
list.pc[t] = pc
var i = 0
while i < rx_nc { list.caps[t * rx_nc + i] = caps[i]; i += 1 }
list.n += 1
}
# run prog over s (length slen) from startpos; returns caps words or null
function rx_run(prog: Prog, s: pointer, slen: int, startpos: int) -> words {
rx_code = prog.code.d
rx_nc = 2 * (prog.ngroups + 1)
let ncode = prog.code.n / 3
rx_seen = words(ncode)
var i = 0
while i < ncode { rx_seen[i] = 0; i += 1 }
rx_gen = 0
var clist = new TList
clist.pc = words(ncode); clist.caps = words(ncode * rx_nc); clist.n = 0
var nlist = new TList
nlist.pc = words(ncode); nlist.caps = words(ncode * rx_nc); nlist.n = 0
let wcaps = words(rx_nc)
var matched: words = null
rx_gen += 1
i = 0
while i < rx_nc { wcaps[i] = -1; i += 1 }
rx_add(clist, 0, wcaps, startpos, s, slen)
var sp = startpos
while true {
if clist.n == 0 { break }
var c = -1
if sp < slen { c = s[sp] & 255 }
nlist.n = 0
rx_gen += 1
var ti = 0
var stop = false
while ti < clist.n and not stop {
let pc = clist.pc[ti]
var k = 0
while k < rx_nc { wcaps[k] = clist.caps[ti * rx_nc + k]; k += 1 }
let op = rx_code[3 * pc]
if op == OP_CHAR {
if c >= 0 and c == rx_code[3 * pc + 1] { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
} else if op == OP_ANY {
if c >= 0 and c != '\n' { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
} else if op == OP_ANYNL {
if c >= 0 { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
} else if op == OP_CLASS {
if c >= 0 and rx_class_has(prog, rx_code[3 * pc + 1], c) { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
} else if op == OP_MATCH {
if matched == null { matched = words(rx_nc) }
k = 0
while k < rx_nc { matched[k] = wcaps[k]; k += 1 }
stop = true
}
ti += 1
}
let tmp = clist; clist = nlist; nlist = tmp
if sp >= slen { break }
sp += 1
}
return matched
}
# ---- public API -------------------------------------------------------------
property Match { str: pointer = null, ng: int = 0, caps: words = null }
function regex_matches(str: pointer, pattern: pointer) -> bool {
let p = regex_compile(pattern)
if p == null { return false }
return rx_run(p, str, rx_slen(str), 0) != null
}
function regex_test(str: pointer, re: Prog) -> bool {
if re == null { return false }
return rx_run(re, str, rx_slen(str), 0) != null
}
function regex_exec(str: pointer, re: Prog) -> Match {
if re == null { return null }
let caps = rx_run(re, str, rx_slen(str), 0)
if caps == null { return null }
let m = new Match
m.str = str; m.ng = re.ngroups; m.caps = caps
return m
}
function regex_find(str: pointer, pattern: pointer) -> Match {
let p = regex_compile(pattern)
if p == null { return null }
return regex_exec(str, p)
}
function regex_next(str: pointer, re: Prog, from: int) -> Match {
if re == null { return null }
let n = rx_slen(str)
if from > n { return null }
let caps = rx_run(re, str, n, from)
if caps == null { return null }
let m = new Match
m.str = str; m.ng = re.ngroups; m.caps = caps
return m
}
function regex_ok(m: Match) -> bool { return m != null }
function regex_group_count(m: Match) -> int { if m == null { return 0 }; return m.ng }
function regex_start(m: Match, n: int) -> int {
if m == null or n < 0 or n > m.ng { return -1 }
return m.caps[2 * n]
}
function regex_end(m: Match, n: int) -> int {
if m == null or n < 0 or n > m.ng { return -1 }
return m.caps[2 * n + 1]
}
function regex_group(m: Match, n: int) -> string {
if m == null or n < 0 or n > m.ng { return "" }
let a = m.caps[2 * n]
let b = m.caps[2 * n + 1]
if a < 0 or b < 0 { return "" }
let out = bytes(b - a + 1)
var i = 0
while i < b - a { out[i] = m.str[a + i]; i += 1 }
out[b - a] = 0
return out
}
function regex_valid(pattern: pointer) -> bool { return regex_compile(pattern) != null }
# regex_replace(str, pattern, repl): replace all non-overlapping matches.
# repl expands \0..\9 (groups; \0 = whole match) and \\ (a literal backslash).
function rx_append(out: IVec, s: pointer, a: int, b: int) -> void {
var i = a
while i < b { iv_push(out, s[i] & 255); i += 1 }
}
function regex_replace(str: pointer, pattern: pointer, repl: pointer) -> string {
let p = regex_compile(pattern)
if p == null { return str }
let n = rx_slen(str)
let rn = rx_slen(repl)
let out = iv_new() # bytes accumulated as ints
var pos = 0
var prev = 0
while pos <= n {
let caps = rx_run(p, str, n, pos)
if caps == null { break }
let ms = caps[0]
let me = caps[1]
rx_append(out, str, prev, ms) # text before the match
# expand replacement
var i = 0
while i < rn {
if repl[i] == '\\' and i + 1 < rn {
let d = repl[i + 1] & 255
if d >= '0' and d <= '9' {
let g = d - 48
if g <= p.ngroups { let a = caps[2 * g]; let b = caps[2 * g + 1]; if a >= 0 and b >= 0 { rx_append(out, str, a, b) } }
i += 2
} else if d == '\\' { iv_push(out, 92); i += 2 }
else { iv_push(out, d); i += 2 }
} else {
iv_push(out, repl[i] & 255); i += 1
}
}
prev = me
if me > pos { pos = me } else { if me < n { iv_push(out, str[me] & 255) }; prev = me + 1; pos = me + 1 }
}
rx_append(out, str, prev, n) # trailing text
let res = bytes(out.n + 1)
var i = 0
while i < out.n { res[i] = out.d[i]; i += 1 }
res[out.n] = 0
return res
}