ludic/selfhost/backend/stdlib/emit_text.ludic
Orkuncakilkaya 254097657e runtime: Log, DateTime.format, Input.text, Path, Mime, Fs, Os and Text keep nothing per call
Found by reading every builtin (Os.platform's 8 KB per call started it). Log builds its line only at
or above the threshold and frees it; DateTime.format folds through + so its pieces go; Input.text
encodes into one buffer; Path/Mime/Fs/Os free their temporaries on every path; string results of
Text/Path/Mime/DateTime/Os dirs are fresh and Text frees a fresh argument. Reseeded.
runtime_temps.ludic: 19.8 MB -> 0 over 20,000 rounds, 64 KB -> 0 over 200 of file work; clean under
MallocScribble. string_temps still 0.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-28 13:55:49 +03:00

162 lines
7.4 KiB
Text

# emit_text.ludic — the Text.* namespace over `str` (null-terminated byte
# strings). The libc-backed queries (length/char_at/starts_with/ends_with/
# contains/index_of/to_int) allocate nothing; slice/from_int/equals/concat reuse
# the string preludes that the `+`, `s[a..b]` and string(int) operators emit.
function is_text_ns(meth: pointer) -> bool {
if (meth == "length") or (meth == "char_at") or (meth == "slice") { return true }
if (meth == "equals") or (meth == "concat") or (meth == "to_int") or (meth == "from_int") { return true }
if (meth == "to_float") { return true }
if (meth == "starts_with") or (meth == "ends_with") { return true }
if (meth == "contains") or (meth == "index_of") { return true }
if (meth == "upper") or (meth == "lower") or (meth == "trim") or (meth == "repeat") { return true }
if (meth == "pad_left") or (meth == "pad_right") { return true }
if (meth == "split") or (meth == "join") or (meth == "replace") { return true }
return false
}
# A string argument made for the call (a template, string(n), a slice) is the call's to give back once
# read: Text.to_int(string(n)) kept n's text. Every string a Text call returns is a copy of its own,
# so it is fresh and goes once a +, == or print has read it.
function text_gone(v: Val) -> void {
if v.fresh { emit(` call void @free(ptr {v.code})\n`) }
}
function emit_text_ns(meth: pointer, e: Node) -> Val {
if (meth == "from_int") { # int -> string, same as string(n)
let n = emit_expr(e.kids[0])
g_uses_intstr = true
return fresh_val(emit_bind(`call ptr @lp_int_str(i32 {n.code})`), "string")
}
if (meth == "slice") { # s[a..b], same substring helper
let s0 = emit_expr(e.kids[0]); let a = emit_expr(e.kids[1]); let b = emit_expr(e.kids[2])
g_uses_strslice = true
let r = fresh_val(emit_bind(`call ptr @lp_str_slice(ptr {s0.code}, i32 {a.code}, i32 {b.code})`), "string")
text_gone(s0)
return r
}
if (meth == "equals") { # byte-wise equality, same as ==
let a = emit_expr(e.kids[0]); let b = emit_expr(e.kids[1])
return emit_str_op("==", a, b)
}
if (meth == "concat") { # a + b, same as the + operator
let a = emit_expr(e.kids[0]); let b = emit_expr(e.kids[1])
return emit_str_op("+", a, b)
}
let s = emit_expr(e.kids[0])
if (meth == "length") { # byte length
let r = emit_bind(`call i64 @strlen(ptr {s.code})`)
text_gone(s)
return val(emit_bind(`trunc i64 {r} to i32`), "int")
}
if (meth == "char_at") { # the byte at index i, 0..255
let i = emit_expr(e.kids[1])
let a = emit_bind(`getelementptr inbounds i8, ptr {s.code}, i32 {i.code}`)
let c = emit_bind(`load i8, ptr {a}`)
text_gone(s)
return val(emit_bind(`zext i8 {c} to i32`), "int")
}
if (meth == "to_float") { # parse a leading decimal number, 0.0 if none
fp_declare("declare double @strtod(ptr, ptr)\n")
let d = emit_bind(`call double @strtod(ptr {s.code}, ptr null)`)
text_gone(s)
return val(emit_bind(`fptrunc double {d} to float`), "float")
}
if (meth == "to_int") { # parse a leading integer, 0 if none
let r = emit_bind(`call i32 @atoi(ptr {s.code})`)
text_gone(s)
return val(r, "int")
}
if (meth == "upper") or (meth == "lower") or (meth == "trim") { # a fresh copy, changed
g_uses_textrt = true
var f = "lp_str_trim"
if (meth == "upper") { f = "lp_str_upper" }
if (meth == "lower") { f = "lp_str_lower" }
let r = fresh_val(emit_bind(`call ptr @{f}(ptr {s.code})`), "string")
text_gone(s)
return r
}
if (meth == "repeat") { # s repeated n times
g_uses_textrt = true
let n = emit_expr(e.kids[1])
let r = fresh_val(emit_bind(`call ptr @lp_str_repeat(ptr {s.code}, i32 {n.code})`), "string")
text_gone(s)
return r
}
if (meth == "pad_left") or (meth == "pad_right") { # pad with spaces to width
g_uses_textrt = true
let w = emit_expr(e.kids[1])
var left = "1"
if (meth == "pad_right") { left = "0" }
let r = fresh_val(emit_bind(`call ptr @lp_str_pad(ptr {s.code}, i32 {w.code}, i1 {left})`), "string")
text_gone(s)
return r
}
if (meth == "replace") { # replace every `from` with `to`
g_uses_textrt2 = true
let from = emit_expr(e.kids[1]); let to = emit_expr(e.kids[2])
let r = fresh_val(emit_bind(`call ptr @lp_str_replace(ptr {s.code}, ptr {from.code}, ptr {to.code})`), "string")
text_gone(s); text_gone(from); text_gone(to)
return r
}
if (meth == "split") { # split on a separator -> []string
g_uses_textrt2 = true
let sep = emit_expr(e.kids[1])
let r = val(emit_bind(`call ptr @lp_str_split(ptr {s.code}, ptr {sep.code})`), "[]string")
text_gone(s); text_gone(sep)
return r
}
if (meth == "join") { # join a []string with a separator (s is the slice)
g_uses_textrt2 = true
let sep = emit_expr(e.kids[1])
let r = fresh_val(emit_bind(`call ptr @lp_str_join(ptr {s.code}, ptr {sep.code})`), "string")
text_gone(sep)
return r
}
if (meth == "contains") or (meth == "index_of") { # substring search
let sub = emit_expr(e.kids[1])
let p = emit_bind(`call ptr @strstr(ptr {s.code}, ptr {sub.code})`)
if (meth == "contains") {
let nn = emit_bind(`icmp ne ptr {p}, null`)
text_gone(s); text_gone(sub)
return val(emit_bind(`zext i1 {nn} to i32`), "bool")
}
let isnull = emit_bind(`icmp eq ptr {p}, null`) # index_of -> byte offset or -1
let pi = emit_bind(`ptrtoint ptr {p} to i64`)
let si = emit_bind(`ptrtoint ptr {s.code} to i64`)
let d = emit_bind(`sub i64 {pi}, {si}`)
let d32 = emit_bind(`trunc i64 {d} to i32`)
text_gone(s); text_gone(sub)
return val(emit_bind(`select i1 {isnull}, i32 -1, i32 {d32}`), "int")
}
# starts_with / ends_with: compare against the affix over its own length
let affix = emit_expr(e.kids[1])
let la = emit_bind(`call i64 @strlen(ptr {affix.code})`)
if (meth == "starts_with") { # strncmp of the head is null-safe
let cmp = emit_bind(`call i32 @strncmp(ptr {s.code}, ptr {affix.code}, i64 {la})`)
let eqz = emit_bind(`icmp eq i32 {cmp}, 0`)
text_gone(s); text_gone(affix)
return val(emit_bind(`zext i1 {eqz} to i32`), "bool")
}
# ends_with: compare the tail, but only when the affix fits (else a negative
# offset would read before the string) — branch so the strncmp never underflows
let ls = emit_bind(`call i64 @strlen(ptr {s.code})`)
let off = emit_bind(`sub i64 {ls}, {la}`)
let res = emit_alloca("i32")
store_at("i32", "0", res)
let neg = emit_bind(`icmp slt i64 {off}, 0`)
let cmpl = lbl("ew_cmp"); let en = lbl("ew_end")
emit(" br i1 "); emit(neg); emit(", label %"); emit(en); emit(", label %"); emit(cmpl); emit("\n")
emit(cmpl); emit(":\n")
let tail = emit_bind(`getelementptr inbounds i8, ptr {s.code}, i64 {off}`)
let cmp = emit_bind(`call i32 @strncmp(ptr {tail}, ptr {affix.code}, i64 {la})`)
let eqz = emit_bind(`icmp eq i32 {cmp}, 0`)
let z = emit_bind(`zext i1 {eqz} to i32`)
store_at("i32", z, res)
emit(" br label %"); emit(en); emit("\n")
emit(en); emit(":\n")
text_gone(s); text_gone(affix)
return val(emit_bind(`load i32, ptr {res}`), "bool")
}