A regular-expression library with PCRE/PECL-compatible syntax, implemented as a
Thompson NFA / Pike VM so a bad pattern from a modder can NEVER cause
catastrophic backtracking — matching is O(n·m), never exponential. `(a+)+$` on
40 non-matching chars, `(a*)*b`, `(.*a){20}b` all run in microseconds; a 50 KB
input scans in ~7 ms.
The engine (runtime/native/regex.ludic + regex_vm.ludic, ~700 lines of Ludic, no
C) parses a pattern to a small bytecode program — an unanchored lazy `.*?` prefix
makes a plain search match anywhere — and the VM runs every alive thread in
lockstep per input byte, deduped by program counter and carrying capture slots
(save/restore, leftmost-greedy priority). Supported: literals, `.`, classes
`[...]` (ranges, negation, `\d \w \s` and their negations), anchors `^ $`,
alternation `|`, capturing and `(?:…)` groups, and `* + ? {n} {n,} {n,m}` in
greedy or lazy form, plus the common escapes; numbered capture groups. Errors are
values — an invalid pattern compiles to null, never a crash. Backreferences and
look-around are out of scope for a linear engine, and on the degenerate case of a
nullable subpattern under an unbounded quantifier positions may differ from a
backtracking engine (the price of the linear-time guarantee) — documented.
Surface (Regex.*, aliased in emit_call.ludic to the regex_* functions):
compile / valid / matches / test / find / exec / next / replace / group /
group_count / start / end / ok.
The runtime is spliced on demand: the parser sets a flag when it sees `Regex.`
and maybe_splice_runtime imports the engine — so it costs nothing in a program
that doesn't use it and works in a plain tool (not just an ECS game).
Verified against Python's `re` as an oracle: a 20k-case grammar fuzzer agrees
100% on realistic patterns (0 / 15000 with capture groups) and 99.8% on group-0
spans across the full pathological grammar, the residual being the documented
nullable-quantifier case. examples/library/regex.ludic asserts the behaviour
(wired into `x test`, now 60 passed); docs: a Regex section + 13 per-symbol
pages, inventory + coverage green. Seed reseeded; the C-free bootstrap fixpoint
holds.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
211 lines
7.2 KiB
Text
211 lines
7.2 KiB
Text
# ============================================================================
|
|
# regex_vm.ludic — the Pike VM (Thompson NFA execution) and the public Regex.*
|
|
# API for the engine compiled in regex.ludic. Splits the "run + query" concern
|
|
# out of regex.ludic's "parse + compile". Spliced together (regex.ludic imports
|
|
# this file), so the two share the Prog/RNode model and OP_* opcodes.
|
|
# ============================================================================
|
|
|
|
# ---- Pike VM ----------------------------------------------------------------
|
|
property TList { pc: words, caps: words, n: int }
|
|
var rx_seen: words = null
|
|
var rx_gen: int = 0
|
|
var rx_nc: int = 0
|
|
var rx_code: words = null
|
|
|
|
function rx_add(list: TList, pc: int, caps: words, sp: int, s: pointer, slen: int) -> void {
|
|
if rx_seen[pc] == rx_gen { return }
|
|
rx_seen[pc] = rx_gen
|
|
let op = rx_code[3 * pc]
|
|
if op == OP_JMP { rx_add(list, rx_code[3 * pc + 1], caps, sp, s, slen); return }
|
|
if op == OP_SPLIT {
|
|
rx_add(list, rx_code[3 * pc + 1], caps, sp, s, slen)
|
|
rx_add(list, rx_code[3 * pc + 2], caps, sp, s, slen)
|
|
return
|
|
}
|
|
if op == OP_SAVE {
|
|
let slot = rx_code[3 * pc + 1]
|
|
let old = caps[slot]
|
|
caps[slot] = sp
|
|
rx_add(list, pc + 1, caps, sp, s, slen)
|
|
caps[slot] = old
|
|
return
|
|
}
|
|
if op == OP_BOL {
|
|
if sp == 0 { rx_add(list, pc + 1, caps, sp, s, slen) }
|
|
return
|
|
}
|
|
if op == OP_EOL {
|
|
if sp == slen { rx_add(list, pc + 1, caps, sp, s, slen) }
|
|
else if sp == slen - 1 and (s[sp] & 255) == 10 { rx_add(list, pc + 1, caps, sp, s, slen) }
|
|
return
|
|
}
|
|
# a leaf that consumes (CHAR/ANY/ANYNL/CLASS) or MATCH: record it
|
|
let t = list.n
|
|
list.pc[t] = pc
|
|
var i = 0
|
|
while i < rx_nc { list.caps[t * rx_nc + i] = caps[i]; i = i + 1 }
|
|
list.n = list.n + 1
|
|
}
|
|
|
|
# run prog over s (length slen) from startpos; returns caps words or null
|
|
function rx_run(prog: Prog, s: pointer, slen: int, startpos: int) -> words {
|
|
rx_code = prog.code.d
|
|
rx_nc = 2 * (prog.ngroups + 1)
|
|
let ncode = prog.code.n / 3
|
|
rx_seen = words(ncode)
|
|
var i = 0
|
|
while i < ncode { rx_seen[i] = 0; i = i + 1 }
|
|
rx_gen = 0
|
|
var clist = new TList
|
|
clist.pc = words(ncode); clist.caps = words(ncode * rx_nc); clist.n = 0
|
|
var nlist = new TList
|
|
nlist.pc = words(ncode); nlist.caps = words(ncode * rx_nc); nlist.n = 0
|
|
let wcaps = words(rx_nc)
|
|
var matched: words = null
|
|
|
|
rx_gen = rx_gen + 1
|
|
i = 0
|
|
while i < rx_nc { wcaps[i] = 0 - 1; i = i + 1 }
|
|
rx_add(clist, 0, wcaps, startpos, s, slen)
|
|
|
|
var sp = startpos
|
|
while true {
|
|
if clist.n == 0 { break }
|
|
var c = 0 - 1
|
|
if sp < slen { c = s[sp] & 255 }
|
|
nlist.n = 0
|
|
rx_gen = rx_gen + 1
|
|
var ti = 0
|
|
var stop = false
|
|
while ti < clist.n and not stop {
|
|
let pc = clist.pc[ti]
|
|
var k = 0
|
|
while k < rx_nc { wcaps[k] = clist.caps[ti * rx_nc + k]; k = k + 1 }
|
|
let op = rx_code[3 * pc]
|
|
if op == OP_CHAR {
|
|
if c >= 0 and c == rx_code[3 * pc + 1] { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
|
|
} else if op == OP_ANY {
|
|
if c >= 0 and c != 10 { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
|
|
} else if op == OP_ANYNL {
|
|
if c >= 0 { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
|
|
} else if op == OP_CLASS {
|
|
if c >= 0 and rx_class_has(prog, rx_code[3 * pc + 1], c) { rx_add(nlist, pc + 1, wcaps, sp + 1, s, slen) }
|
|
} else if op == OP_MATCH {
|
|
if matched == null { matched = words(rx_nc) }
|
|
k = 0
|
|
while k < rx_nc { matched[k] = wcaps[k]; k = k + 1 }
|
|
stop = true
|
|
}
|
|
ti = ti + 1
|
|
}
|
|
let tmp = clist; clist = nlist; nlist = tmp
|
|
if sp >= slen { break }
|
|
sp = sp + 1
|
|
}
|
|
return matched
|
|
}
|
|
|
|
# ---- public API -------------------------------------------------------------
|
|
property Match { str: pointer = null, ng: int = 0, caps: words = null }
|
|
|
|
function regex_matches(str: pointer, pattern: pointer) -> bool {
|
|
let p = regex_compile(pattern)
|
|
if p == null { return false }
|
|
return rx_run(p, str, rx_slen(str), 0) != null
|
|
}
|
|
function regex_test(str: pointer, re: Prog) -> bool {
|
|
if re == null { return false }
|
|
return rx_run(re, str, rx_slen(str), 0) != null
|
|
}
|
|
function regex_exec(str: pointer, re: Prog) -> Match {
|
|
if re == null { return null }
|
|
let caps = rx_run(re, str, rx_slen(str), 0)
|
|
if caps == null { return null }
|
|
let m = new Match
|
|
m.str = str; m.ng = re.ngroups; m.caps = caps
|
|
return m
|
|
}
|
|
function regex_find(str: pointer, pattern: pointer) -> Match {
|
|
let p = regex_compile(pattern)
|
|
if p == null { return null }
|
|
return regex_exec(str, p)
|
|
}
|
|
function regex_next(str: pointer, re: Prog, from: int) -> Match {
|
|
if re == null { return null }
|
|
let n = rx_slen(str)
|
|
if from > n { return null }
|
|
let caps = rx_run(re, str, n, from)
|
|
if caps == null { return null }
|
|
let m = new Match
|
|
m.str = str; m.ng = re.ngroups; m.caps = caps
|
|
return m
|
|
}
|
|
function regex_ok(m: Match) -> bool { return m != null }
|
|
function regex_group_count(m: Match) -> int { if m == null { return 0 }; return m.ng }
|
|
function regex_start(m: Match, n: int) -> int {
|
|
if m == null or n < 0 or n > m.ng { return 0 - 1 }
|
|
return m.caps[2 * n]
|
|
}
|
|
function regex_end(m: Match, n: int) -> int {
|
|
if m == null or n < 0 or n > m.ng { return 0 - 1 }
|
|
return m.caps[2 * n + 1]
|
|
}
|
|
function regex_group(m: Match, n: int) -> pointer {
|
|
if m == null or n < 0 or n > m.ng { return "" }
|
|
let a = m.caps[2 * n]
|
|
let b = m.caps[2 * n + 1]
|
|
if a < 0 or b < 0 { return "" }
|
|
let out = bytes(b - a + 1)
|
|
var i = 0
|
|
while i < b - a { out[i] = m.str[a + i]; i = i + 1 }
|
|
out[b - a] = 0
|
|
return out
|
|
}
|
|
function regex_valid(pattern: pointer) -> bool { return regex_compile(pattern) != null }
|
|
|
|
# regex_replace(str, pattern, repl): replace all non-overlapping matches.
|
|
# repl expands \0..\9 (groups; \0 = whole match) and \\ (a literal backslash).
|
|
function rx_append(out: IVec, s: pointer, a: int, b: int) -> void {
|
|
var i = a
|
|
while i < b { iv_push(out, s[i] & 255); i = i + 1 }
|
|
}
|
|
function regex_replace(str: pointer, pattern: pointer, repl: pointer) -> pointer {
|
|
let p = regex_compile(pattern)
|
|
if p == null { return str }
|
|
let n = rx_slen(str)
|
|
let rn = rx_slen(repl)
|
|
let out = iv_new() # bytes accumulated as ints
|
|
var pos = 0
|
|
var prev = 0
|
|
while pos <= n {
|
|
let caps = rx_run(p, str, n, pos)
|
|
if caps == null { break }
|
|
let ms = caps[0]
|
|
let me = caps[1]
|
|
rx_append(out, str, prev, ms) # text before the match
|
|
# expand replacement
|
|
var i = 0
|
|
while i < rn {
|
|
if repl[i] == 92 and i + 1 < rn {
|
|
let d = repl[i + 1] & 255
|
|
if d >= 48 and d <= 57 {
|
|
let g = d - 48
|
|
if g <= p.ngroups { let a = caps[2 * g]; let b = caps[2 * g + 1]; if a >= 0 and b >= 0 { rx_append(out, str, a, b) } }
|
|
i = i + 2
|
|
} else if d == 92 { iv_push(out, 92); i = i + 2 }
|
|
else { iv_push(out, d); i = i + 2 }
|
|
} else {
|
|
iv_push(out, repl[i] & 255); i = i + 1
|
|
}
|
|
}
|
|
prev = me
|
|
if me > pos { pos = me } else { if me < n { iv_push(out, str[me] & 255) }; prev = me + 1; pos = me + 1 }
|
|
}
|
|
rx_append(out, str, prev, n) # trailing text
|
|
let res = bytes(out.n + 1)
|
|
var i = 0
|
|
while i < out.n { res[i] = out.d[i]; i = i + 1 }
|
|
res[out.n] = 0
|
|
return res
|
|
}
|
|
|