feat(testing): line coverage via --coverage + bin/x test --coverage (#45)
All checks were successful
bootstrap / cfree-fixpoint (push) Successful in 17s
ci / build-and-test (push) Successful in 1m13s
commit-lint / conventional-commits (push) Successful in 3s
docs / build-and-deploy (push) Successful in 19s

Instrument each emitted statement with a per-source-line hit counter, gated
behind a new --coverage flag (default off) so ordinary builds — and the
compiler's own self-compile — stay byte-identical and the C-free bootstrap
fixpoint is untouched. A static line table plus a parallel hit-counter array
are dumped at exit through an atexit hook to $LUDIC_COVERAGE (default
ludic.cov) as a `FILE <name>` header and `<line> <hits>` rows.

bin/x test --coverage compiles the test specs with instrumentation, runs them
into per-file dumps, and aggregates a clean per-file line-coverage report that
names the unreached lines. Adds examples/library/coverage.ludic (a spec whose
tests deliberately miss one branch) and docs. Closes the last open acceptance
item of the testing framework (#12).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-08-31 14:51:14 +03:00
parent c7c8e2779c
commit 498593311f
11 changed files with 15317 additions and 14622 deletions

3
changes/coverage.md Normal file
View file

@ -0,0 +1,3 @@
bump: minor
type: feat
Line coverage for the testing framework — compile with `--coverage` and the compiler instruments each statement with a per-source-line hit counter, dumped at exit (via an `atexit` hook) to the file named by `$LUDIC_COVERAGE` (default `ludic.cov`) as a `FILE <name>` header plus one `<line> <hits>` row per instrumented line. The instrumentation is flag-gated and additive, so an ordinary build — and the compiler's own self-compile — stays byte-identical and the C-free bootstrap fixpoint is untouched. `bin/x test --coverage` compiles the test specs this way, runs them, and aggregates the dumps into a per-file line-coverage report that names the unreached lines. Closes the last open acceptance item of the testing framework (#12).

View file

@ -19,6 +19,16 @@ Assertions (each records a failure and prints <code>file:line: … failed</code>
Run a spec file directly with the compiler-runner — `ludic mymath_test.ludic` compiles it to a native binary, runs it, and forwards the pass/fail exit code — so it drops straight into `bin/x` and CI. Run a spec file directly with the compiler-runner — `ludic mymath_test.ludic` compiles it to a native binary, runs it, and forwards the pass/fail exit code — so it drops straight into `bin/x` and CI.
<strong>Line coverage.</strong> Compile with the <code>--coverage</code> flag and the compiler instruments every statement with a per-source-line hit counter; at exit the counts are written to the file named by <code>$LUDIC_COVERAGE</code> (default <code>ludic.cov</code>) as a <code>FILE &lt;name&gt;</code> header followed by one <code>&lt;line&gt; &lt;hits&gt;</code> row per instrumented line. The instrumentation is flag-gated and additive, so an ordinary build — and the compiler's own self-compile — stays byte-identical. <code>bin/x test --coverage</code> compiles the test specs this way, runs them, and aggregates the dumps into a per-file report that names the lines your tests never reached:
```
== line coverage (bin/x test --coverage) ==
examples/library/coverage.ludic 16/17 lines 94% uncovered: 33
examples/library/testing.ludic 17/17 lines 100%
----
total: 33/34 lines 97%
```
```ludic ```ludic
program MathSpec { program MathSpec {
function add(a: int, b: int) -> int { return a + b } function add(a: int, b: int) -> int { return a + b }

View file

@ -0,0 +1,51 @@
# coverage.ludic — a spec built to be measured by `bin/x test --coverage` (issue
# #45). Compiled with `--coverage`, every statement bumps a per-source-line hit
# counter; at exit the counts are dumped and `bin/x test --coverage` turns them
# into a per-file line-coverage report.
#
# The tests below exercise `sign` fully but only the taken branches of `grade`,
# so the report flags `grade`'s unreached line — the whole point of coverage:
# it shows you the branch your tests forgot.
#
# As an ordinary spec it still passes and prints:
# == 3 passed, 0 failed ==
program CoverageSpec {
# fully covered: the tests below hit all three returns.
function sign(n: int) -> int {
if n > 0 {
return 1
}
if n < 0 {
return 0 - 1
}
return 0
}
# partially covered: no test passes a score below 60, so the `return 0` (fail)
# line is never reached and shows up as uncovered in the report.
function grade(score: int) -> int {
if score >= 90 {
return 4
}
if score >= 60 {
return 2
}
return 0
}
test "sign covers every branch" {
expect_eq(sign(7), 1)
expect_eq(sign(0 - 3), 0 - 1)
expect_eq(sign(0), 0)
}
test "grade: high scores" {
expect_eq(grade(95), 4)
expect_eq(grade(75), 2)
}
test "grade: boundary" {
expect_eq(grade(90), 4)
expect_eq(grade(60), 2)
}
}

View file

@ -39,6 +39,39 @@ var g_tests: []Node # test "name" { ... } blocks collected by the p
var g_src_name: pointer = "?" # base name of the source file, for panic/expect file:line messages var g_src_name: pointer = "?" # base name of the source file, for panic/expect file:line messages
var g_uses_longstr: bool = false # string(long) / interpolating a long was emitted -> emit fn_long_str var g_uses_longstr: bool = false # string(long) / interpolating a long was emitted -> emit fn_long_str
# ---- line coverage (issue #45) --------------------------------------------
# --coverage instruments each emitted statement with a bump of a per-source-line
# hit counter. Everything here is gated behind g_coverage (default off), so an
# ordinary build — and the compiler's own self-compile — stays byte-identical
# and the C-free bootstrap fixpoint is untouched.
var g_coverage: bool = false # --coverage was passed
var g_cov_active: bool = true # instrument the code being emitted now (off for spliced runtime fns)
var g_cov_lines: []int # distinct source lines instrumented, first-seen order (slot = index)
var g_prog_user_end: int = 0 # count of prog decls from the user's source (before the runtime splice)
# slot index for a source line, appending on first sight so g_cov_lines doubles
# as the line table dumped at exit.
function cov_slot(line: int) -> int {
var i = 0
while i < len(g_cov_lines) { if g_cov_lines[i] == line { return i }; i = i + 1 }
push(g_cov_lines, line)
return len(g_cov_lines) - 1
}
# emit the per-statement hit bump: @L_cov_hits[slot] += 1. A no-op unless
# --coverage is on and we are in user code (not a spliced runtime function).
function emit_cov_hit(line: int) -> void {
if not g_coverage { return }
if not g_cov_active { return }
if line <= 0 { return }
let slot = cov_slot(line)
let p = nreg()
emit(" "); emit(p); emit(" = getelementptr inbounds i32, ptr @L_cov_hits, i32 "); emit(itoa(slot)); emit("\n")
let v = emit_bind(`load i32, ptr {p}`)
let v1 = emit_bind(`add i32 {v}, 1`)
emit(" store i32 "); emit(v1); emit(", ptr "); emit(p); emit("\n")
}
# loop targets for break/continue (innermost last) # loop targets for break/continue (innermost last)
var brk_lbl: []pointer var brk_lbl: []pointer
var cnt_lbl: []pointer var cnt_lbl: []pointer

View file

@ -140,6 +140,8 @@ function emit_program() -> void {
g_uses_loopback = false g_uses_loopback = false
g_uses_expect = false g_uses_expect = false
g_uses_panic = false g_uses_panic = false
g_cov_lines = new []int
g_cov_active = true
loc_name = new []pointer; loc_reg = new []pointer; loc_ty = new []pointer; loc_mut = new []int loc_name = new []pointer; loc_reg = new []pointer; loc_ty = new []pointer; loc_mut = new []int
brk_lbl = new []pointer; cnt_lbl = new []pointer brk_lbl = new []pointer; cnt_lbl = new []pointer
self_stk = new []pointer self_stk = new []pointer
@ -148,7 +150,14 @@ function emit_program() -> void {
emit_extern_decls() emit_extern_decls()
if has_ecs() { emit_ecs_storage() } if has_ecs() { emit_ecs_storage() }
var i = 0 var i = 0
while i < len(prog) { if prog[i].kind == N_FN { emit_fn(prog[i]) }; i = i + 1 } while i < len(prog) {
if prog[i].kind == N_FN {
g_cov_active = (i < g_prog_user_end) # don't instrument spliced runtime functions
emit_fn(prog[i])
g_cov_active = true
}
i = i + 1
}
if len(g_events) > 0 { emit_event_fns() } # EV0: @ev_<E> event-dispatch functions if len(g_events) > 0 { emit_event_fns() } # EV0: @ev_<E> event-dispatch functions
if has_ecs() and (len(g_events) > 0 or g_uses_query or g_uses_reflect) { emit_world_table() } # EV2/EV8: the mod reflection ABI (also powers Query.* / Reflect.*) if has_ecs() and (len(g_events) > 0 or g_uses_query or g_uses_reflect) { emit_world_table() } # EV2/EV8: the mod reflection ABI (also powers Query.* / Reflect.*)
if has_ecs() { emit_ecs_allocator(); emit_snapshot() } if has_ecs() { emit_ecs_allocator(); emit_snapshot() }
@ -189,6 +198,75 @@ function emit_program() -> void {
emith("declare i32 @fprintf(ptr, ptr, ...)\n") emith("declare i32 @fprintf(ptr, ptr, ...)\n")
emith("@.fmt_panic = private unnamed_addr constant [6 x i8] c\"%s%s\\0A\\00\"\n") emith("@.fmt_panic = private unnamed_addr constant [6 x i8] c\"%s%s\\0A\\00\"\n")
} }
emit_cov_runtime() # issue #45: --coverage tables + exit dump
}
# issue #45: the line-coverage runtime. Emits the static line table, a parallel
# hit-counter array, and @cov_dump — a function registered with atexit (via an
# LLVM global constructor) that writes `<line> <hits>` rows to the file named by
# $LUDIC_COVERAGE (default "ludic.cov"). All gated behind --coverage, so a normal
# build emits none of this and stays byte-identical.
function emit_cov_runtime() -> void {
if not g_coverage { return }
let n = len(g_cov_lines)
if n == 0 { return }
let sn = itoa(n)
# the line table + zeroed hit counters (module globals)
emith("@L_cov_lines = internal global ["); emith(sn); emith(" x i32] [")
var i = 0
while i < n {
if i > 0 { emith(", ") }
emith("i32 "); emith(itoa(g_cov_lines[i]))
i = i + 1
}
emith("]\n")
emith("@L_cov_hits = internal global ["); emith(sn); emith(" x i32] zeroinitializer\n")
emith(`@L_cov_n = internal global i32 {sn}\n`)
# the source name, the env-var name, the default path, fopen mode, and row format
let fnc = emit_str_const(g_src_name)
let envc = emit_str_const("LUDIC_COVERAGE")
let defc = emit_str_const("ludic.cov")
let modec = emit_str_const("w")
emith("@.cov_filefmt = private unnamed_addr constant [9 x i8] c\"FILE %s\\0A\\00\"\n")
emith("@.cov_rowfmt = private unnamed_addr constant [7 x i8] c\"%d %d\\0A\\00\"\n")
if not g_uses_panic { emith("declare i32 @fprintf(ptr, ptr, ...)\n") }
emith("declare i32 @atexit(ptr)\n")
# @cov_dump: open $LUDIC_COVERAGE (or "ludic.cov"), write a FILE header then one
# `<line> <hits>` row per instrumented line, and close.
emit("define void @cov_dump() {\nentry:\n")
emit(` %env = call ptr @getenv(ptr {envc})\n`)
emit(" %noenv = icmp eq ptr %env, null\n")
emit(` %path = select i1 %noenv, ptr {defc}, ptr %env\n`)
emit(` %f = call ptr @fopen(ptr %path, ptr {modec})\n`)
emit(" %bad = icmp eq ptr %f, null\n")
emit(" br i1 %bad, label %done, label %write\n")
emit("write:\n")
emit(` call i32 (ptr, ptr, ...) @fprintf(ptr %f, ptr @.cov_filefmt, ptr {fnc})\n`)
emit(" br label %loop\n")
emit("loop:\n")
emit(" %i = phi i32 [ 0, %write ], [ %i1, %body ]\n")
emit(` %go = icmp slt i32 %i, {sn}\n`)
emit(" br i1 %go, label %body, label %close\n")
emit("body:\n")
emit(" %lp = getelementptr inbounds ["); emit(sn); emit(" x i32], ptr @L_cov_lines, i32 0, i32 %i\n")
emit(" %lv = load i32, ptr %lp\n")
emit(" %hp = getelementptr inbounds ["); emit(sn); emit(" x i32], ptr @L_cov_hits, i32 0, i32 %i\n")
emit(" %hv = load i32, ptr %hp\n")
emit(" call i32 (ptr, ptr, ...) @fprintf(ptr %f, ptr @.cov_rowfmt, i32 %lv, i32 %hv)\n")
emit(" %i1 = add i32 %i, 1\n")
emit(" br label %loop\n")
emit("close:\n")
emit(" %rc = call i32 @fclose(ptr %f)\n")
emit(" br label %done\n")
emit("done:\n ret void\n}\n\n")
# register @cov_dump with atexit before main runs (an LLVM global constructor).
emit("define void @cov_init() {\nentry:\n")
emit(" %r = call i32 @atexit(ptr @cov_dump)\n")
emit(" ret void\n}\n\n")
emith("@llvm.global_ctors = appending global [1 x { i32, ptr, ptr }] [{ i32, ptr, ptr } { i32 65535, ptr @cov_init, ptr null }]\n")
} }
# Flush the emitted IR. With a null path it goes to stdout (the pipe the shell # Flush the emitted IR. With a null path it goes to stdout (the pipe the shell

View file

@ -222,6 +222,7 @@ function emit_emit(st: Node) -> Val {
} }
function emit_stmt(st: Node) -> void { function emit_stmt(st: Node) -> void {
emit_cov_hit(st.line) # --coverage: bump this line's hit counter (no-op otherwise)
if st.kind == S_LET { if st.kind == S_LET {
var ty = st.ty var ty = st.ty
if (ty == null) { let v0 = emit_expr(st.a); ty = v0.ty if (ty == null) { let v0 = emit_expr(st.a); ty = v0.ty

View file

@ -243,7 +243,17 @@ function block() -> Node {
return b return b
} }
# parse a statement and stamp it with its source line (the first token's line),
# unless the specific rule already set one. The line drives --coverage and the
# panic/expect file:line messages.
function stmt() -> Node { function stmt() -> Node {
let ln = toks[pi].line
let n = stmt_body()
if n.line == 0 { n.line = ln }
return n
}
function stmt_body() -> Node {
let t = toks[pi] let t = toks[pi]
if t.kind == TK_ID { if t.kind == TK_ID {
if (t.text == "let") or (t.text == "var") { if (t.text == "let") or (t.text == "var") {

File diff suppressed because it is too large Load diff

View file

@ -77,6 +77,7 @@ entry {
var fmt = false # --fmt: lex + parse only, then exit (the doc-check gate) var fmt = false # --fmt: lex + parse only, then exit (the doc-check gate)
var save = false # --save-temps: keep the intermediate .ll var save = false # --save-temps: keep the intermediate .ll
var run = false # compile then execute the result var run = false # compile then execute the result
g_coverage = false # --coverage: instrument each statement with a per-line hit counter
# multi-call: invoked as `ludic` -> run mode by default # multi-call: invoked as `ludic` -> run mode by default
if (base_name(arg(0)) == "ludic") { run = true } if (base_name(arg(0)) == "ludic") { run = true }
@ -91,6 +92,7 @@ entry {
else { if (a == ("--fmt")) { fmt = true } else { if (a == ("--fmt")) { fmt = true }
else { if (a == ("--save-temps")) { save = true } else { if (a == ("--save-temps")) { save = true }
else { if (a == ("--run")) { run = true } else { if (a == ("--run")) { run = true }
else { if (a == ("--coverage")) { g_coverage = true }
else { if (a == ("-o")) { ai = ai + 1; if ai < arg_count() { out = arg(ai) } } else { if (a == ("-o")) { ai = ai + 1; if ai < arg_count() { out = arg(ai) } }
else { else {
# an unknown flag is ignored (with a note) rather than mistaken for the # an unknown flag is ignored (with a note) rather than mistaken for the
@ -99,7 +101,7 @@ entry {
let m = `ludicc: ignoring unknown flag {a}\n` let m = `ludicc: ignoring unknown flag {a}\n`
file_write(file_stderr(), m, len(m)) file_write(file_stderr(), m, len(m))
} else { path = a } } else { path = a }
} } } } } } } } } } } } } } } } }
ai = ai + 1 ai = ai + 1
} }
@ -113,6 +115,7 @@ entry {
g_src_name = base_name(path) # for panic/expect file:line messages g_src_name = base_name(path) # for panic/expect file:line messages
lex(src) lex(src)
parse_program() parse_program()
g_prog_user_end = len(prog) # decls from the user's source; runtime splice appends after
# --fmt is the doc-check gate: reaching here means it lexed and parsed. A parse # --fmt is the doc-check gate: reaching here means it lexed and parsed. A parse
# error would already have exited nonzero, so a clean parse exits 0. (Canonical # error would already have exited nonzero, so a clean parse exits 0. (Canonical

View file

@ -37,6 +37,7 @@ program X {
print("") print("")
print("test:") print("test:")
print(" x test the full regression suite") print(" x test the full regression suite")
print(" x test --coverage per-file line coverage over the test specs")
print(" x selfhost-test the self-hosting suite (correctness + bootstrap fixpoints)") print(" x selfhost-test the self-hosting suite (correctness + bootstrap fixpoints)")
print(" x test-tools the editor-toolchain suite") print(" x test-tools the editor-toolchain suite")
print(" x golden regenerate selfhost/golden/renders.sha256 (review with git diff)") print(" x golden regenerate selfhost/golden/renders.sha256 (review with git diff)")
@ -79,7 +80,10 @@ program X {
if (cmd == "app") { exit(cmd_app()) } if (cmd == "app") { exit(cmd_app()) }
if (cmd == "tools") { exit(cmd_tools()) } if (cmd == "tools") { exit(cmd_tools()) }
if (cmd == "clean") { exit(cmd_clean()) } if (cmd == "clean") { exit(cmd_clean()) }
if (cmd == "test") { exit(cmd_test()) } if (cmd == "test") {
if (argn(2, "") == "--coverage") { exit(cmd_test_coverage()) }
exit(cmd_test())
}
if (cmd == "selfhost-test") { exit(cmd_selfhost_test()) } if (cmd == "selfhost-test") { exit(cmd_selfhost_test()) }
if (cmd == "test-tools") { exit(cmd_test_tools()) } if (cmd == "test-tools") { exit(cmd_test_tools()) }
if (cmd == "check-docs") { exit(cmd_check_docs()) } if (cmd == "check-docs") { exit(cmd_check_docs()) }

View file

@ -87,6 +87,66 @@ function panic_case() -> void {
else { bad2("panic", `rc={string(rc)} err=[{msg}]`) } else { bad2("panic", `rc={string(rc)} err=[{msg}]`) }
} }
# issue #45: `bin/x test --coverage`. Compile each test-spec with `--coverage`,
# run it with LUDIC_COVERAGE pointed at a per-file dump, then aggregate the dumps
# into a clean per-file line-coverage report. The instrumentation is flag-gated,
# so this reuses the same seed-built bin/ludicc the rest of the suite does.
function cov_one(path: pointer, dir: pointer) -> int {
let nm = flat(path)
let ll = `/tmp/x_cov_{nm}.ll`
let bin = `/tmp/x_cov_{nm}`
let cov = `{dir}/{nm}.cov`
if not shq(`bin/ludicc --coverage examples/{path}.ludic > {ll} 2>/tmp/x_cov.err`) {
bad2(path, capture_line("tail -1 /tmp/x_cov.err")); return 1
}
if not shq(`{cc()} -O2 {ll} -o {bin} 2>/tmp/x_cov.err`) {
bad2(path, capture_line("tail -1 /tmp/x_cov.err")); return 1
}
run(`rm -f {ll}`)
# run the spec; its atexit hook writes the dump to $LUDIC_COVERAGE
if not shq(`LUDIC_COVERAGE={cov} {bin} > /dev/null 2>&1`) {
bad2(path, "spec exited non-zero"); return 1
}
if not file_exists(cov) { bad2(path, "no coverage dump written"); return 1 }
# aggregate: total instrumented lines, how many were hit, and which were missed.
# each row is `<line> <hits>`, so an unhit line ends in " 0" — grep counts and
# lists them (avoiding awk, whose braces collide with string interpolation).
let total = str_to_int(capture_line(`grep -c '^[0-9]' {cov}`))
let nmiss = str_to_int(capture_line(`grep -c ' 0$' {cov}`))
let covered = total - nmiss
let missed = capture_line(`grep ' 0$' {cov} | cut -d' ' -f1 | tr '\n' ' '`)
var pct = 100
if total > 0 { pct = (covered * 100) / total }
var line = ` examples/{path}.ludic {string(covered)}/{string(total)} lines {string(pct)}%`
if not (missed == "") { line = `{line} uncovered: {missed}` }
print(line)
COV_COVERED = COV_COVERED + covered
COV_TOTAL = COV_TOTAL + total
return 0
}
var COV_COVERED: int = 0
var COV_TOTAL: int = 0
function cmd_test_coverage() -> int {
# the suite normally rebuilds bin/ludicc from the seed; do the same here so the
# instrumentation path is exactly what a clean checkout ships.
run("mkdir -p bin build /tmp/x_cov_out")
if (not is_exec("bin/ludicc")) or newer("selfhost/ludicc.seed.ll", "bin/ludicc") {
if not shq(`{cc()} selfhost/ludicc.seed.ll -o bin/ludicc`) { err("x: cannot build bin/ludicc from the seed\n"); return 1 }
}
COV_COVERED = 0
COV_TOTAL = 0
print("== line coverage (bin/x test --coverage) ==")
cov_one("library/coverage", "/tmp/x_cov_out")
cov_one("library/testing", "/tmp/x_cov_out")
var pct = 100
if COV_TOTAL > 0 { pct = (COV_COVERED * 100) / COV_TOTAL }
print(` ----`)
print(` total: {string(COV_COVERED)}/{string(COV_TOTAL)} lines {string(pct)}%`)
return 0
}
function cmd_test() -> int { function cmd_test() -> int {
PASS = 0 PASS = 0
FAIL = 0 FAIL = 0
@ -141,6 +201,7 @@ function cmd_test() -> int {
feat_case("library/render", "", "1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18", "render.ludic (Screen pixel/oval/camera/clip/blend_mode/measure_text + Camera set/follow/shake, verified by pixel readback)") feat_case("library/render", "", "1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18", "render.ludic (Screen pixel/oval/camera/clip/blend_mode/measure_text + Camera set/follow/shake, verified by pixel readback)")
feat_case("library/lighting", "", "1 2 3 4 5 6 7 8 9 10 11 12 13 14", "lighting.ludic (Light ambient/point radial falloff + occluder hard shadows — 2D light accumulation, verified by pixel readback)") feat_case("library/lighting", "", "1 2 3 4 5 6 7 8 9 10 11 12 13 14", "lighting.ludic (Light ambient/point radial falloff + occluder hard shadows — 2D light accumulation, verified by pixel readback)")
spec_case("library/testing", "== 6 passed, 0 failed ==") spec_case("library/testing", "== 6 passed, 0 failed ==")
spec_case("library/coverage", "== 3 passed, 0 failed ==")
feat_case("library/errors", "", "5 10 0 7 1", "errors.ludic (assert guards an invariant, holds -> runs to the end; issue #8 success path)") feat_case("library/errors", "", "5 10 0 7 1", "errors.ludic (assert guards an invariant, holds -> runs to the end; issue #8 success path)")
panic_case() panic_case()
feat_case("library/logging", "", "0 5 2 1", "logging.ludic (Log levels, set_level/level threshold, structured fields)") feat_case("library/logging", "", "0 5 2 1", "logging.ludic (Log levels, set_level/level threshold, structured fields)")