Phase 7h: string slicing s[a..b] (retire substr)
`s[a..b]` is a fresh substring of the bytes [a, b) — the modern, end-based form of the C-style `substr(s, start, count)`: substr(src, start, i - start) -> src[start..i] substr(t, 2, len(t) - 2) -> t[2..len(t)] substr(src, i, 2) -> src[i..i + 2] Mechanics: a new E_SLICE postfix (`base[lo..hi]`, distinct from `base[i]` indexing) lowers to a @fn_str_slice prelude (malloc + copy + terminate), emitted once into any program that slices. Two reseeds: add the syntax + prelude, then migrate the 22 substr calls and delete substr. The migrator recognises the common `count == end - start` shape and emits the clean `s[start..end]` rather than `s[start..start + (end - start)]`. examples/strings.ludic gains slicing (now prints 1..9). Reseeded (22673 lines); C-free fixpoint holds; goldens identical; 18/18; vocab + doc-fences clean. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
parent
0efa06dca2
commit
1b9789630a
14 changed files with 4049 additions and 3914 deletions
|
|
@ -598,7 +598,10 @@ let msg = `hello {name}, you have {count + 1} messages`
|
|||
```
|
||||
|
||||
`str(x)` is the same conversion on its own. Write a literal brace as `{{` / `}}`.
|
||||
`expr with { field: … }`
|
||||
|
||||
**Slicing.** `s[a..b]` is a fresh substring of the bytes `[a, b)`, and `len(s)`
|
||||
is a string's byte length — so `path[0..len(path) - 6]` trims an extension and
|
||||
`s[i]` still indexes a single byte. `expr with { field: … }`
|
||||
is not implemented; records appear only in `spawn`. Char literals (`'w'`) are
|
||||
`int` code points; colors are hex ints (`0xff8800`). `null` is the null-pointer
|
||||
literal; test any pointer/record/slice with `x == null` / `x != null` (an unset
|
||||
|
|
|
|||
|
|
@ -22,5 +22,10 @@ program Strings {
|
|||
let n = 42
|
||||
if `{greeting}, {who}!` == "hello, Ludic!" { print(6) }
|
||||
if `n={n}, next={n + 1}` == "n=42, next=43" { print(7) }
|
||||
|
||||
# slicing: s[a..b] is the substring of bytes [a, b), and len(s) its length
|
||||
let full = "hello world"
|
||||
if full[0..5] == "hello" { print(8) }
|
||||
if full[6..len(full)] == "world" { print(9) }
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -50,6 +50,7 @@ const E_FLOAT: int = 40
|
|||
const E_REC: int = 41
|
||||
const E_FINIT: int = 42
|
||||
const E_NULL: int = 43 # the `null` pointer literal
|
||||
const E_SLICE: int = 44 # s[a..b] — substring (a=base, b=start, c=end)
|
||||
|
||||
property Node {
|
||||
kind: int = 0
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ var loc_mut: []int # 1 = mutable (var / param / loop-var), 0 = immutab
|
|||
var nloc: int = 0
|
||||
var g_uses_str: bool = false # a `str + str` / `str == str` was emitted -> emit the prelude
|
||||
var g_uses_intstr: bool = false # `str(int)` was emitted -> emit the int->string prelude
|
||||
var g_uses_strslice: bool = false # `s[a..b]` was emitted -> emit the substring prelude
|
||||
|
||||
# loop targets for break/continue (innermost last)
|
||||
var brk_lbl: []ptr
|
||||
|
|
@ -57,7 +58,7 @@ fn llty(t: ptr) -> ptr {
|
|||
}
|
||||
|
||||
fn is_slice_ty(t: ptr) -> bool { return peek8(t, 0) == 91 and peek8(t, 1) == 93 } # "[]"
|
||||
fn slice_elem(t: ptr) -> ptr { return substr(t, 2, len(t) - 2) }
|
||||
fn slice_elem(t: ptr) -> ptr { return t[2..len(t)] }
|
||||
|
||||
fn find_arch(name: ptr) -> Node {
|
||||
var i = 0
|
||||
|
|
|
|||
|
|
@ -69,6 +69,7 @@ fn emit_program() -> void {
|
|||
code = buf_new()
|
||||
g_uses_str = false
|
||||
g_uses_intstr = false
|
||||
g_uses_strslice = false
|
||||
loc_name = new []ptr; loc_reg = new []ptr; loc_ty = new []ptr; loc_mut = new []int
|
||||
brk_lbl = new []ptr; cnt_lbl = new []ptr
|
||||
self_stk = new []ptr
|
||||
|
|
@ -86,6 +87,7 @@ fn emit_program() -> void {
|
|||
}
|
||||
if g_uses_str { emit_str_prelude() } # @fn_str_eq / @fn_str_concat, after all uses are seen
|
||||
if g_uses_intstr { emit_int_str() } # @fn_int_str, for str(int) in interpolation
|
||||
if g_uses_strslice { emit_str_slice() } # @fn_str_slice, for s[a..b]
|
||||
}
|
||||
|
||||
# Flush the emitted IR. With a null path it goes to stdout (the pipe the shell
|
||||
|
|
|
|||
|
|
@ -178,6 +178,13 @@ fn emit_expr(e: Node) -> Val {
|
|||
if e.kind == E_FLOAT { return val(itoa(e.ival), "fixed") }
|
||||
if e.kind == E_BOOL { return val(itoa(e.ival), "bool") }
|
||||
if e.kind == E_NULL { return val("null", "ptr") }
|
||||
if e.kind == E_SLICE { # s[a..b] -> a fresh substring
|
||||
let base = emit_expr(e.a)
|
||||
let lo = emit_expr(e.b)
|
||||
let hi = emit_expr(e.c)
|
||||
g_uses_strslice = true
|
||||
return val(emit_bind(`call ptr @fn_str_slice(ptr {base.code}, i32 {lo.code}, i32 {hi.code})`), "str")
|
||||
}
|
||||
if e.kind == E_STR { return val(emit_str_const(e.s), "str") }
|
||||
if e.kind == E_NEW {
|
||||
if is_slice_ty(e.s) { return emit_new_slice(e.s) }
|
||||
|
|
|
|||
|
|
@ -175,3 +175,16 @@ fn emit_int_str() -> void {
|
|||
emith("fin:\n %fpos = phi i32 [ %pos, %sign ], [ %posn, %addneg ]\n")
|
||||
emith(" %rpos = add i32 %fpos, 1\n %res = getelementptr inbounds i8, ptr %buf, i32 %rpos\n ret ptr %res\n}\n")
|
||||
}
|
||||
|
||||
# s[a..b] -> a fresh NUL-terminated copy of the bytes [a, b), emitted (once) into
|
||||
# any program that slices a string. Mallocs (b-a)+1, copies, terminates.
|
||||
fn emit_str_slice() -> void {
|
||||
emith("define ptr @fn_str_slice(ptr %s, i32 %start, i32 %end) {\n")
|
||||
emith("entry:\n %len = sub i32 %end, %start\n %sz = add i32 %len, 1\n")
|
||||
emith(" %sz64 = sext i32 %sz to i64\n %out = call ptr @malloc(i64 %sz64)\n br label %loop\n")
|
||||
emith("loop:\n %i = phi i32 [ 0, %entry ], [ %i1, %body ]\n")
|
||||
emith(" %d = icmp slt i32 %i, %len\n br i1 %d, label %body, label %fin\n")
|
||||
emith("body:\n %si = add i32 %start, %i\n %sp = getelementptr inbounds i8, ptr %s, i32 %si\n %c = load i8, ptr %sp\n")
|
||||
emith(" %op = getelementptr inbounds i8, ptr %out, i32 %i\n store i8 %c, ptr %op\n %i1 = add i32 %i, 1\n br label %loop\n")
|
||||
emith("fin:\n %tp = getelementptr inbounds i8, ptr %out, i32 %len\n store i8 0, ptr %tp\n ret ptr %out\n}\n")
|
||||
}
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ fn has_ui() -> bool {
|
|||
|
||||
# index of a UI_<name>: a ui block's root, or a widget's id=
|
||||
fn ui_index_of(nm: ptr) -> int {
|
||||
let s = substr(nm, 3, len(nm) - 3) # strip "UI_"
|
||||
let s = nm[3..len(nm)] # strip "UI_"
|
||||
let u = 0; var bi = 0
|
||||
var i = 0
|
||||
while i < len(prog) {
|
||||
|
|
|
|||
|
|
@ -127,25 +127,25 @@ fn lex(src: ptr) -> void {
|
|||
if char_is_alpha(c) {
|
||||
let start = i
|
||||
while i < n and char_is_alnum(peek8(src, i)) { i = i + 1 }
|
||||
tok_push(TK_ID, substr(src, start, i - start), 0, line)
|
||||
tok_push(TK_ID, src[start..i], 0, line)
|
||||
continue
|
||||
}
|
||||
if c == 59 { tok_push(TK_NL, null, 0, line); i = i + 1; continue } # ';'
|
||||
# two-character operators
|
||||
if two_at(src, i, 45, 62) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # ->
|
||||
if two_at(src, i, 61, 62) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # =>
|
||||
if two_at(src, i, 61, 61) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # ==
|
||||
if two_at(src, i, 33, 61) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # !=
|
||||
if two_at(src, i, 60, 61) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # <=
|
||||
if two_at(src, i, 62, 61) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # >=
|
||||
if two_at(src, i, 43, 61) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # +=
|
||||
if two_at(src, i, 45, 61) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # -=
|
||||
if two_at(src, i, 42, 61) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # *=
|
||||
if two_at(src, i, 47, 61) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # /=
|
||||
if two_at(src, i, 46, 46) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # ..
|
||||
if two_at(src, i, 60, 60) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # <<
|
||||
if two_at(src, i, 62, 62) { tok_push(TK_OP, substr(src, i, 2), 0, line); i = i + 2; continue } # >>
|
||||
if is_op1(c) { tok_push(TK_OP, substr(src, i, 1), 0, line); i = i + 1; continue }
|
||||
if two_at(src, i, 45, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # ->
|
||||
if two_at(src, i, 61, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # =>
|
||||
if two_at(src, i, 61, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # ==
|
||||
if two_at(src, i, 33, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # !=
|
||||
if two_at(src, i, 60, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # <=
|
||||
if two_at(src, i, 62, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # >=
|
||||
if two_at(src, i, 43, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # +=
|
||||
if two_at(src, i, 45, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # -=
|
||||
if two_at(src, i, 42, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # *=
|
||||
if two_at(src, i, 47, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # /=
|
||||
if two_at(src, i, 46, 46) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # ..
|
||||
if two_at(src, i, 60, 60) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # <<
|
||||
if two_at(src, i, 62, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i = i + 2; continue } # >>
|
||||
if is_op1(c) { tok_push(TK_OP, src[i..i + 1], 0, line); i = i + 1; continue }
|
||||
i = i + 1 # skip anything unrecognised
|
||||
}
|
||||
tok_push(TK_EOF, null, 0, line)
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -19,14 +19,14 @@ fn base_name(path: ptr) -> ptr {
|
|||
var last = 0 - 1
|
||||
var i = 0
|
||||
while peek8(path, i) != 0 { if peek8(path, i) == 47 { last = i }; i = i + 1 }
|
||||
return substr(path, last + 1, i - (last + 1))
|
||||
return path[last + 1..i]
|
||||
}
|
||||
|
||||
# drop a trailing ".ludic" if present
|
||||
fn strip_ludic(name: ptr) -> ptr {
|
||||
let n = len(name)
|
||||
if n > 6 {
|
||||
if (substr(name, n - 6, 6) == ".ludic") { return substr(name, 0, n - 6) }
|
||||
if (name[n - 6..n] == ".ludic") { return name[0..n - 6] }
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
|
|
|||
|
|
@ -57,7 +57,7 @@ fn args_call(call: Node) -> void {
|
|||
# ---- string interpolation --------------------------------------------------
|
||||
# `text {expr} text` desugars to a `+` chain of string literals and `str(expr)`
|
||||
# holes, so it reuses the string-concat operator and needs no new runtime.
|
||||
fn interp_lit(buf: ptr, len: int) -> Node { let n = node(E_STR); n.s = substr(buf, 0, len); return n }
|
||||
fn interp_lit(buf: ptr, len: int) -> Node { let n = node(E_STR); n.s = buf[0..0 + len]; return n }
|
||||
fn interp_add(acc: Node, part: Node) -> Node {
|
||||
if acc == null { return part }
|
||||
return mkbin("+", acc, part)
|
||||
|
|
@ -92,7 +92,7 @@ fn parse_interp(raw: ptr) -> Node {
|
|||
else { if d == 125 { depth = depth - 1; if depth == 0 { break } } }
|
||||
i = i + 1
|
||||
}
|
||||
acc = interp_add(acc, interp_str(parse_hole(substr(raw, hs, i - hs))))
|
||||
acc = interp_add(acc, interp_str(parse_hole(raw[hs..i])))
|
||||
i = i + 1 # skip the closing '}'
|
||||
} else {
|
||||
if c == 125 and peek8(raw, i + 1) == 125 { poke8(lit, lj, 125); lj = lj + 1; i = i + 2; continue } # }} -> }
|
||||
|
|
@ -131,7 +131,9 @@ fn p_postfix() -> Node {
|
|||
var e = p_primary()
|
||||
while true {
|
||||
if is_op(".") { pi = pi + 1; let m = node(E_MEMBER); m.a = e; m.s = eat_id(); e = m }
|
||||
else { if is_op("[") { pi = pi + 1; let ix = node(E_INDEX); ix.a = e; ix.b = expr(); eat_op("]"); e = ix }
|
||||
else { if is_op("[") { pi = pi + 1; let lo = expr()
|
||||
if is_op("..") { pi = pi + 1; let sl = node(E_SLICE); sl.a = e; sl.b = lo; sl.c = expr(); eat_op("]"); e = sl } # s[a..b] substring
|
||||
else { let ix = node(E_INDEX); ix.a = e; ix.b = lo; eat_op("]"); e = ix } }
|
||||
else { if is_op("(") { let c = node(E_CALL); c.a = e; args_call(c); e = c } else { break } } }
|
||||
}
|
||||
return e
|
||||
|
|
@ -311,7 +313,7 @@ fn dir_of(path: ptr) -> ptr {
|
|||
var i = 0
|
||||
while peek8(path, i) != 0 { if peek8(path, i) == 47 { last = i }; i = i + 1 }
|
||||
if last < 0 { return "" }
|
||||
return substr(path, 0, last + 1)
|
||||
return path[0..0 + (last + 1)]
|
||||
}
|
||||
fn path_join(dir: ptr, rel: ptr) -> ptr {
|
||||
if peek8(rel, 0) == 47 { return rel } # absolute
|
||||
|
|
|
|||
|
|
@ -3,13 +3,6 @@
|
|||
# byte buffers reached with peek8/poke8.
|
||||
|
||||
# a fresh NUL-terminated copy of src[start .. start+n]
|
||||
fn substr(src: ptr, start: int, n: int) -> ptr {
|
||||
let b = mem_alloc(n + 1)
|
||||
var i = 0
|
||||
while i < n { poke8(b, i, peek8(src, start + i)); i = i + 1 }
|
||||
poke8(b, n, 0)
|
||||
return b
|
||||
}
|
||||
|
||||
fn char_is_digit(c: int) -> bool { return c >= 48 and c <= 57 }
|
||||
fn char_is_alpha(c: int) -> bool {
|
||||
|
|
|
|||
4
test.sh
4
test.sh
|
|
@ -68,8 +68,8 @@ if ./selfhost/game-build.sh build/ludicc examples/toggle.ludic "/tmp/ludic_tog"
|
|||
else bad "toggle: $(tail -1 /tmp/tog.out)"; fi
|
||||
# String operators: ==/!= compare by content, + concatenates (str_* prelude).
|
||||
if ./selfhost/game-build.sh build/ludicc examples/strings.ludic "/tmp/ludic_str" >/tmp/str.out 2>&1 \
|
||||
&& [ "$(/tmp/ludic_str </dev/null | tr '\n' ' ')" = "1 2 3 4 5 6 7 " ]; then
|
||||
ok "strings.ludic (str ==/!=/+ and `interpolation`)"
|
||||
&& [ "$(/tmp/ludic_str </dev/null | tr '\n' ' ')" = "1 2 3 4 5 6 7 8 9 " ]; then
|
||||
ok "strings.ludic (str ops, interpolation, slicing)"
|
||||
else bad "strings: $(tail -1 /tmp/str.out)"; fi
|
||||
# --- Toolchain-agent CLI smoke tests append below this line ---
|
||||
echo "== self-hosted front-end binaries (ludicc / ludic) =="
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue