ludic/runtime/native/regex_vm.ludic
Orkuncakilkaya b798e3024e
All checks were successful
bootstrap / cfree-fixpoint (push) Successful in 26s
ci / build-and-test (push) Successful in 1m6s
commit-lint / conventional-commits (push) Successful in 4s
docs / build-and-deploy (push) Successful in 18s
feat(stdlib): add Regex.* — a linear-time regular-expression engine (#18)
A regular-expression library with PCRE/PECL-compatible syntax, implemented as a
Thompson NFA / Pike VM so a bad pattern from a modder can NEVER cause
catastrophic backtracking — matching is O(n·m), never exponential. `(a+)+$` on
40 non-matching chars, `(a*)*b`, `(.*a){20}b` all run in microseconds; a 50 KB
input scans in ~7 ms.

The engine (runtime/native/regex.ludic + regex_vm.ludic, ~700 lines of Ludic, no
C) parses a pattern to a small bytecode program — an unanchored lazy `.*?` prefix
makes a plain search match anywhere — and the VM runs every alive thread in
lockstep per input byte, deduped by program counter and carrying capture slots
(save/restore, leftmost-greedy priority). Supported: literals, `.`, classes
`[...]` (ranges, negation, `\d \w \s` and their negations), anchors `^ $`,
alternation `|`, capturing and `(?:…)` groups, and `* + ? {n} {n,} {n,m}` in
greedy or lazy form, plus the common escapes; numbered capture groups. Errors are
values — an invalid pattern compiles to null, never a crash. Backreferences and
look-around are out of scope for a linear engine, and on the degenerate case of a
nullable subpattern under an unbounded quantifier positions may differ from a
backtracking engine (the price of the linear-time guarantee) — documented.

Surface (Regex.*, aliased in emit_call.ludic to the regex_* functions):
compile / valid / matches / test / find / exec / next / replace / group /
group_count / start / end / ok.

The runtime is spliced on demand: the parser sets a flag when it sees `Regex.`
and maybe_splice_runtime imports the engine — so it costs nothing in a program
that doesn't use it and works in a plain tool (not just an ECS game).

Verified against Python's `re` as an oracle: a 20k-case grammar fuzzer agrees
100% on realistic patterns (0 / 15000 with capture groups) and 99.8% on group-0
spans across the full pathological grammar, the residual being the documented
nullable-quantifier case. examples/library/regex.ludic asserts the behaviour
(wired into `x test`, now 60 passed); docs: a Regex section + 13 per-symbol
pages, inventory + coverage green. Seed reseeded; the C-free bootstrap fixpoint
holds.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-08-31 12:23:19 +03:00

211 lines
7.2 KiB
Text

# ============================================================================
# regex_vm.ludic — the Pike VM (Thompson NFA execution) and the public Regex.*
# API for the engine compiled in regex.ludic. Splits the "run + query" concern
# out of regex.ludic's "parse + compile". Spliced together (regex.ludic imports
# this file), so the two share the Prog/RNode model and OP_* opcodes.
# ============================================================================
# ---- Pike VM ----------------------------------------------------------------
property TList { pc: words, caps: words, n: int }
var rx_seen: words = null
var rx_gen: int = 0
var rx_nc: int = 0
var rx_code: words = null
function rx_add(list: TList, pc: int, caps: words, sp: int, s: pointer, slen: int) -> void {
if rx_seen[pc] == rx_gen { return }
rx_seen[pc] = rx_gen
let op = rx_code[3 * pc]
if op == OP_JMP { rx_add(list, rx_code[3 * pc + 1], caps, sp, s, slen); return }
if op == OP_SPLIT {
rx_add(list, rx_code[3 * pc + 1], caps, sp, s, slen)
rx_add(list, rx_code[3 * pc + 2], caps, sp, s, slen)
return
}
if op == OP_SAVE {
let slot = rx_code[3 * pc + 1]
let old = caps[slot]
caps[slot] = sp
rx_add(list, pc + 1, caps, sp, s, slen)
caps[slot] = old
return
}
if op == OP_BOL {
if sp == 0 { rx_add(list, pc + 1, caps, sp, s, slen) }
return
}
if op == OP_EOL {
if sp == slen { rx_add(list, pc + 1, caps, sp, s, slen) }
else if sp == slen - 1 and (s[sp] & 255) == 10 { rx_add(list, pc + 1, caps, sp, s, slen) }
return
}
# a leaf that consumes (CHAR/ANY/ANYNL/CLASS) or MATCH: record it
let t = list.n
list.pc[t] = pc
var i = 0
while i < rx_nc { list.caps[t * rx_nc + i] = caps[i]; i = i + 1 }
list.n = list.n + 1
}
# run prog over s (length slen) from startpos; returns caps words or null
function rx_run(prog: Prog, s: pointer, slen: int, startpos: int) -> words {
rx_code = prog.code.d
rx_nc = 2 * (prog.ngroups + 1)
let ncode = prog.code.n / 3
rx_seen = words(ncode)
var i = 0
while i < ncode { rx_seen[i] = 0; i = i + 1 }
rx_gen = 0
var clist = new TList
clist.pc = words(ncode); clist.caps = words(ncode * rx_nc); clist.n = 0
var nlist = new TList
nlist.pc = words(ncode); nlist.caps = words(ncode * rx_nc); nlist.n = 0
let wcaps = words(rx_nc)
var matched: words = null
rx_gen = rx_gen + 1
i = 0
while i < rx_nc { wcaps[i] = 0 - 1; i = i + 1 }
rx_add(clist, 0, wcaps, startpos, s, slen)
var sp = startpos
while true {
if clist.n == 0 { break }
var c = 0 - 1
if sp < slen { c = s[sp] & 255 }
nlist.n = 0
rx_gen = rx_gen + 1
var ti = 0
var stop = false
while ti < clist.n and not stop {
let pc = clist.pc[ti]
var k = 0
while k < rx_nc { wcaps[k] = clist.caps[ti * rx_nc + k]; k = k + 1 }
let op = rx_code[3 * pc]
if op == OP_CHAR {
if c >= 0 and c == rx_code[3 * pc + 1] { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
} else if op == OP_ANY {
if c >= 0 and c != 10 { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
} else if op == OP_ANYNL {
if c >= 0 { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
} else if op == OP_CLASS {
if c >= 0 and rx_class_has(prog, rx_code[3 * pc + 1], c) { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
} else if op == OP_MATCH {
if matched == null { matched = words(rx_nc) }
k = 0
while k < rx_nc { matched[k] = wcaps[k]; k = k + 1 }
stop = true
}
ti = ti + 1
}
let tmp = clist; clist = nlist; nlist = tmp
if sp >= slen { break }
sp = sp + 1
}
return matched
}
# ---- public API -------------------------------------------------------------
property Match { str: pointer = null, ng: int = 0, caps: words = null }
function regex_matches(str: pointer, pattern: pointer) -> bool {
let p = regex_compile(pattern)
if p == null { return false }
return rx_run(p, str, rx_slen(str), 0) != null
}
function regex_test(str: pointer, re: Prog) -> bool {
if re == null { return false }
return rx_run(re, str, rx_slen(str), 0) != null
}
function regex_exec(str: pointer, re: Prog) -> Match {
if re == null { return null }
let caps = rx_run(re, str, rx_slen(str), 0)
if caps == null { return null }
let m = new Match
m.str = str; m.ng = re.ngroups; m.caps = caps
return m
}
function regex_find(str: pointer, pattern: pointer) -> Match {
let p = regex_compile(pattern)
if p == null { return null }
return regex_exec(str, p)
}
function regex_next(str: pointer, re: Prog, from: int) -> Match {
if re == null { return null }
let n = rx_slen(str)
if from > n { return null }
let caps = rx_run(re, str, n, from)
if caps == null { return null }
let m = new Match
m.str = str; m.ng = re.ngroups; m.caps = caps
return m
}
function regex_ok(m: Match) -> bool { return m != null }
function regex_group_count(m: Match) -> int { if m == null { return 0 }; return m.ng }
function regex_start(m: Match, n: int) -> int {
if m == null or n < 0 or n > m.ng { return 0 - 1 }
return m.caps[2 * n]
}
function regex_end(m: Match, n: int) -> int {
if m == null or n < 0 or n > m.ng { return 0 - 1 }
return m.caps[2 * n + 1]
}
function regex_group(m: Match, n: int) -> pointer {
if m == null or n < 0 or n > m.ng { return "" }
let a = m.caps[2 * n]
let b = m.caps[2 * n + 1]
if a < 0 or b < 0 { return "" }
let out = bytes(b - a + 1)
var i = 0
while i < b - a { out[i] = m.str[a + i]; i = i + 1 }
out[b - a] = 0
return out
}
function regex_valid(pattern: pointer) -> bool { return regex_compile(pattern) != null }
# regex_replace(str, pattern, repl): replace all non-overlapping matches.
# repl expands \0..\9 (groups; \0 = whole match) and \\ (a literal backslash).
function rx_append(out: IVec, s: pointer, a: int, b: int) -> void {
var i = a
while i < b { iv_push(out, s[i] & 255); i = i + 1 }
}
function regex_replace(str: pointer, pattern: pointer, repl: pointer) -> pointer {
let p = regex_compile(pattern)
if p == null { return str }
let n = rx_slen(str)
let rn = rx_slen(repl)
let out = iv_new() # bytes accumulated as ints
var pos = 0
var prev = 0
while pos <= n {
let caps = rx_run(p, str, n, pos)
if caps == null { break }
let ms = caps[0]
let me = caps[1]
rx_append(out, str, prev, ms) # text before the match
# expand replacement
var i = 0
while i < rn {
if repl[i] == 92 and i + 1 < rn {
let d = repl[i + 1] & 255
if d >= 48 and d <= 57 {
let g = d - 48
if g <= p.ngroups { let a = caps[2 * g]; let b = caps[2 * g + 1]; if a >= 0 and b >= 0 { rx_append(out, str, a, b) } }
i = i + 2
} else if d == 92 { iv_push(out, 92); i = i + 2 }
else { iv_push(out, d); i = i + 2 }
} else {
iv_push(out, repl[i] & 255); i = i + 1
}
}
prev = me
if me > pos { pos = me } else { if me < n { iv_push(out, str[me] & 255) }; prev = me + 1; pos = me + 1 }
}
rx_append(out, str, prev, n) # trailing text
let res = bytes(out.n + 1)
var i = 0
while i < out.n { res[i] = out.d[i]; i = i + 1 }
res[out.n] = 0
return res
}