render3d: actors cast only into the cascades they can shade; Vulkan GPU timings and a mean frame split in R3D_PROF

Where the frame goes, before optimising it further. R3D_PROF now prints the average frame: CPU before
the swap (the game's share apart), GPU and swap, and the renderer's CPU phases in frame order - the
shadow pass split into cascade fit, scatter casters and actor casters. On Vulkan the per-pass GPU
table comes from timestamp queries (host query reset asked for where the device has it); MoltenVK's
attribution is tile-based and not to be trusted per pass, the PC's is.

What it showed on the PC (camp): 4.3 ms CPU and 5.3 ms GPU a frame; the shadow pass was the largest
CPU phase (1.8 ms) and actors half of that. Every actor within 300 m was drawn into all five cascades,
though the outer two only shade receivers from 212 and 935 m out: 160 actors and 300 draws into each.
cast_band_reaches - the flowers' reach test, now shared - skips an actor for a cascade it cannot shade
(receivers counted from 0.85 of the previous split, where sunShadow's cross-fade begins).

PC camp, two runs each: mean frame 9553/9601 -> 8664/8656 us; CPU 4.3 -> 3.7 ms; shadow GPU 1.14 ->
0.79 ms; actor-shadow CPU 1.02 -> 0.60 ms; 2039 -> 1411 draws; self-tests 59/59, validation 0.
Mac: OpenGL shot viewpoints and the camp byte-identical; town 19 px at <= 2/255 on two flower stems a
few metres from the camera - the accepted leftover-binding difference, no shadow; self-tests 59/59 on
OpenGL and Vulkan. R3D_CAST_ALL=1 draws every caster into every cascade.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-15 18:03:31 +03:00
parent ef745a141d
commit 5b9403c343
6 changed files with 134 additions and 10 deletions

View file

@ -149,6 +149,10 @@ var prof_up: []long = null
# the two are cleanly separable there.
var prof_pre_swap: long = 0
var prof_cpu_us: []long = null
var prof_sum_dt: long = 0 # sums over the frames past the first eight, for the mean split
var prof_sum_cpu: long = 0
var prof_sum_game: long = 0
var prof_sum_n: int = 0
function prof_before_swap() -> void { if prof_on { prof_pre_swap = gl_now_us() } }
# The single most expensive CPU phase of each frame, and what it was. A hitch is one
@ -160,12 +164,26 @@ var prof_mark_name: pointer = null
var prof_mark_us: []long = null
var prof_mark_who: []pointer = null
function prof_mark_start() -> void { if prof_on { prof_mark_last = gl_now_us(); prof_mark_best = 0; prof_mark_name = null } }
# ... and every phase's share of the average frame, past the first eight frames (loading, first fills)
var prof_mk_names: []pointer = null
var prof_mk_us: []long = null
var prof_mk_n: int = 0
function prof_cpu_mark(name: pointer) -> void {
if not prof_on { return }
let now = gl_now_us()
let d = now - prof_mark_last
prof_mark_last = now
if d > prof_mark_best { prof_mark_best = d; prof_mark_name = name }
var k = -1
var i = 0
while i < prof_mk_n { if prof_mk_names[i] == name { k = i }; i += 1 }
if k < 0 {
if prof_mk_names == null { prof_mk_names = new []pointer; prof_mk_us = new []long }
push(prof_mk_names, name); push(prof_mk_us, 0)
prof_mk_n += 1
k = prof_mk_n - 1
}
if prof_gen != null and len(prof_gen) > 8 { prof_mk_us[k] = prof_mk_us[k] + d }
}
# the game's own per-frame work, kept apart from the renderer's
@ -268,6 +286,10 @@ function prof_gen_frame() -> void {
if prof_pre_swap != 0 and dtf != 0 { cpu = prof_pre_swap - (now - dtf) }
push(prof_cpu_us, cpu)
push(prof_game_us, prof_game_us_cur)
if len(prof_gen) > 9 and dtf != 0 {
prof_sum_dt = prof_sum_dt + dtf; prof_sum_cpu = prof_sum_cpu + cpu; prof_sum_game = prof_sum_game + prof_game_us_cur
prof_sum_n += 1
}
push(prof_mark_us, prof_mark_best)
if prof_mark_name == null { push(prof_mark_who, "-") } else { push(prof_mark_who, prof_mark_name) }
prof_gen_us_cur = 0; prof_layer_us_cur = 0; prof_bake_us_cur = 0; prof_up_cur = 0; prof_game_us_cur = 0
@ -319,6 +341,21 @@ function prof_ft_report() -> void {
print(` fps at median {string(1000000 / med)} frames over 1.3x median: {string(o13)}, over 1.5x: {string(o15)}, over 2x: {string(over)} — of {string(n)}`)
print(` stutter: {string(excess / 1000)} ms of frame time beyond 1.2x median over the run`)
print(` streaming totals over the run: walk {string(stream_us_walk / 1000)} ms (generate {string(stream_us_gen / 1000)} ms, gather {string(stream_us_gather / 1000)} ms), {string(stream_walks)} stream-walks`)
# the average frame, split: what the process did before the swap (the game's share of it apart),
# the rest waiting on the GPU and the swap, and the renderer's CPU phases in frame order
if prof_sum_n > 0 {
let dt = prof_sum_dt / prof_sum_n
let cpu = prof_sum_cpu / prof_sum_n
let game = prof_sum_game / prof_sum_n
print("")
print(`MEAN frame {string(dt)} us over {string(prof_sum_n)} frames: CPU before the swap {string(cpu)} us (the game's own {string(game)} us), GPU and swap {string(dt - cpu)} us`)
print(" renderer CPU by phase, mean per frame:")
var i = 0
while i < prof_mk_n {
print(` {Text.pad_right(prof_mk_names[i], 20)} {Text.pad_left(string(prof_mk_us[i] / prof_sum_n), 7)} us`)
i += 1
}
}
}
# The slowest frames of the run, with the once-in-a-while CPU work that landed in them.