render3d: actors cast only into the cascades they can shade; Vulkan GPU timings and a mean frame split in R3D_PROF

Where the frame goes, before optimising it further. R3D_PROF now prints the average frame: CPU before
the swap (the game's share apart), GPU and swap, and the renderer's CPU phases in frame order - the
shadow pass split into cascade fit, scatter casters and actor casters. On Vulkan the per-pass GPU
table comes from timestamp queries (host query reset asked for where the device has it); MoltenVK's
attribution is tile-based and not to be trusted per pass, the PC's is.

What it showed on the PC (camp): 4.3 ms CPU and 5.3 ms GPU a frame; the shadow pass was the largest
CPU phase (1.8 ms) and actors half of that. Every actor within 300 m was drawn into all five cascades,
though the outer two only shade receivers from 212 and 935 m out: 160 actors and 300 draws into each.
cast_band_reaches - the flowers' reach test, now shared - skips an actor for a cascade it cannot shade
(receivers counted from 0.85 of the previous split, where sunShadow's cross-fade begins).

PC camp, two runs each: mean frame 9553/9601 -> 8664/8656 us; CPU 4.3 -> 3.7 ms; shadow GPU 1.14 ->
0.79 ms; actor-shadow CPU 1.02 -> 0.60 ms; 2039 -> 1411 draws; self-tests 59/59, validation 0.
Mac: OpenGL shot viewpoints and the camp byte-identical; town 19 px at <= 2/255 on two flower stems a
few metres from the camera - the accepted leftover-binding difference, no shadow; self-tests 59/59 on
OpenGL and Vulkan. R3D_CAST_ALL=1 draws every caster into every cascade.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-15 18:03:31 +03:00
parent ef745a141d
commit 5b9403c343
6 changed files with 134 additions and 10 deletions

View file

@ -370,10 +370,35 @@ function actor_draw_casters(light_vp: words) -> void {
let r = f_max(f_mul(a.radius, a.scale), hh)
if not ac_in_light(light_vp, a.x, f_add(a.y, hh), a.z, f_mul(r, fl(1.5))) { continue }
}
# Inside the light box is not the same as able to shade anything this cascade covers: every actor
# within 300 m sits inside the far cascades' boxes, whose receivers start 250 and 1100 m out - at
# the camp 160 actors cast 300 draws into each of those, against 41 into the nearest.
let dxa = f_sub(a.x, cam_pos[0]); let dza = f_sub(a.z, cam_pos[2])
let da = f_sqrt(f_add(f_mul(dxa, dxa), f_mul(dza, dza)))
let ra = f_mul(a.radius, a.scale)
if not cast_band_reaches(f_max(f_sub(da, ra), F_ZERO), f_add(da, ra), f_mul(a.model.height, a.scale)) { continue }
var ap = ac_sh
if a.cutout { ap = ac_sh_cut }
if ac_census_at > 0 and ac_frame == ac_census_at { ac_caster_count(a) }
actor_draw_one(a, ap, true)
}
if ac_census_at > 0 and ac_frame == ac_census_at {
print(`actor census, cascade {sh_cascade}: {ac_cc_actors} actors cast {ac_cc_draws} draws ({ac_cc_skinned} of them skinned, {ac_cc_cut} cutout)`)
ac_cc_actors = 0; ac_cc_draws = 0; ac_cc_skinned = 0; ac_cc_cut = 0
}
prof_cpu_mark("shadow actors")
}
var ac_cc_actors: int = 0
var ac_cc_draws: int = 0
var ac_cc_skinned: int = 0
var ac_cc_cut: int = 0
function ac_caster_count(a: Actor) -> void {
var shown = 0
for p in 0 .. len(a.model.prims) { if a.hide == null or a.hide[p] == 0 { shown += 1 } }
ac_cc_actors += 1
ac_cc_draws += shown
if a.skin != null or a.model.skin != null { ac_cc_skinned += shown }
if a.cutout { ac_cc_cut += shown }
}
# every actor off the stage at once, for a world being replaced. Actors own no GL objects;

View file

@ -216,12 +216,12 @@ function gpu_program_free(p: int) -> void {
}
# ---- GPU timers (R3D_PROF) ----------------------------------------------------------
function gpu_query_new(n: int, ids: words) -> void { if gpu_kind == GPU_VK { return }; gl_gen_queries(n, ids) }
function gpu_query_begin(id: int) -> void { if gpu_kind == GPU_VK { return }; gl_begin_query(GL_TIME_ELAPSED, id) }
function gpu_query_end() -> void { if gpu_kind == GPU_VK { return }; gl_end_query(GL_TIME_ELAPSED) }
function gpu_query_new(n: int, ids: words) -> void { if gpu_kind == GPU_VK { gvk_query_new(n, ids); return }; gl_gen_queries(n, ids) }
function gpu_query_begin(id: int) -> void { if gpu_kind == GPU_VK { gvk_query_begin(id); return }; gl_begin_query(GL_TIME_ELAPSED, id) }
function gpu_query_end() -> void { if gpu_kind == GPU_VK { gvk_query_end(); return }; gl_end_query(GL_TIME_ELAPSED) }
# true once the query has its result; the nanoseconds (low 32 bits) are then in out[0]
function gpu_query_result(id: int, out: words) -> bool {
if gpu_kind == GPU_VK { return false }
if gpu_kind == GPU_VK { return gvk_query_result(id, out) }
gl_get_query_objectiv(id, GL_QUERY_RESULT_AVAILABLE, out)
if out[0] == 0 { return false }
gl_get_query_objectui64v(id, GL_QUERY_RESULT, out)

View file

@ -177,6 +177,13 @@ function gvk_init() -> bool {
}
gvk_has_dic = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_drawIndirectCount) == 1
if gvk_has_dic { Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_drawIndirectCount, 1) }
# R3D_PROF: per-pass GPU time from timestamp queries (gvk_query_*). The slots are reused every few
# frames, so they are reset from the host - which is a feature to ask for.
gvk_has_hqr = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_hostQueryReset) == 1 and Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampComputeAndGraphics) == 1
if gvk_has_hqr {
Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_hostQueryReset, 1)
gvk_ts_period = Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampPeriod)
}
Vk.put_i32(cnt, 0, 0)
Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, null)
@ -520,3 +527,48 @@ function gvk_shutdown() -> void {
Vk.destroy_instance(gvk_inst, null)
gvk_ready = false
}
# ---- GPU timestamps (R3D_PROF) ------------------------------------------------------------
# prof.ludic's query slots, each a pair of timestamps: 2 * id where the pass starts, 2 * id + 1 where it
# ends. A slot is read back PROF_RING frames after it was written and reset just before it is written
# again. On a device without host query reset or timestamps, every read says "not yet" and the report
# stays empty.
var gvk_has_hqr: bool = false
var gvk_ts_period: int = 0 # float bits: nanoseconds per timestamp tick
var gvk_qpool: long = 0
var gvk_q_active: int = -1
function gvk_query_new(n: int, ids: words) -> void {
for i in 0 .. n { ids[i] = i }
if not gvk_has_hqr or gvk_dev == null { return }
let qci = bytes(VkQueryPoolCreateInfo_sizeof)
Vk.zero(qci, VkQueryPoolCreateInfo_sizeof)
Vk.put_i32(qci, VkQueryPoolCreateInfo_sType, VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO)
Vk.put_i32(qci, VkQueryPoolCreateInfo_queryType, VK_QUERY_TYPE_TIMESTAMP)
Vk.put_i32(qci, VkQueryPoolCreateInfo_queryCount, n * 2)
let out = bytes(8)
if Vk.create_query_pool(gvk_dev, qci, null, out) != VK_SUCCESS { return }
gvk_qpool = gvk_handle(out)
Vk.reset_query_pool(gvk_dev, gvk_qpool, 0, n * 2)
}
function gvk_query_begin(id: int) -> void {
if gvk_qpool == 0 { return }
Vk.reset_query_pool(gvk_dev, gvk_qpool, id * 2, 2)
Vk.cmd_write_timestamp(gvk_frame_cb(), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, gvk_qpool, id * 2)
gvk_q_active = id
}
function gvk_query_end() -> void {
if gvk_qpool == 0 or gvk_q_active < 0 { return }
Vk.cmd_write_timestamp(gvk_frame_cb(), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, gvk_qpool, gvk_q_active * 2 + 1)
gvk_q_active = -1
}
function gvk_query_result(id: int, out: words) -> bool {
if gvk_qpool == 0 { return false }
let data = bytes(16)
let size: long = 16
let stride: long = 8
if Vk.get_query_pool_results(gvk_dev, gvk_qpool, id * 2, 2, size, data, stride, VK_QUERY_RESULT_64_BIT) != VK_SUCCESS { return false }
let ticks = Text.to_int(string(Vk.get_i64(data, 8) - Vk.get_i64(data, 0)))
if ticks < 0 { return false }
out[0] = f_to_int(f_mul(fi(ticks), gvk_ts_period))
return true
}

View file

@ -149,6 +149,10 @@ var prof_up: []long = null
# the two are cleanly separable there.
var prof_pre_swap: long = 0
var prof_cpu_us: []long = null
var prof_sum_dt: long = 0 # sums over the frames past the first eight, for the mean split
var prof_sum_cpu: long = 0
var prof_sum_game: long = 0
var prof_sum_n: int = 0
function prof_before_swap() -> void { if prof_on { prof_pre_swap = gl_now_us() } }
# The single most expensive CPU phase of each frame, and what it was. A hitch is one
@ -160,12 +164,26 @@ var prof_mark_name: pointer = null
var prof_mark_us: []long = null
var prof_mark_who: []pointer = null
function prof_mark_start() -> void { if prof_on { prof_mark_last = gl_now_us(); prof_mark_best = 0; prof_mark_name = null } }
# ... and every phase's share of the average frame, past the first eight frames (loading, first fills)
var prof_mk_names: []pointer = null
var prof_mk_us: []long = null
var prof_mk_n: int = 0
function prof_cpu_mark(name: pointer) -> void {
if not prof_on { return }
let now = gl_now_us()
let d = now - prof_mark_last
prof_mark_last = now
if d > prof_mark_best { prof_mark_best = d; prof_mark_name = name }
var k = -1
var i = 0
while i < prof_mk_n { if prof_mk_names[i] == name { k = i }; i += 1 }
if k < 0 {
if prof_mk_names == null { prof_mk_names = new []pointer; prof_mk_us = new []long }
push(prof_mk_names, name); push(prof_mk_us, 0)
prof_mk_n += 1
k = prof_mk_n - 1
}
if prof_gen != null and len(prof_gen) > 8 { prof_mk_us[k] = prof_mk_us[k] + d }
}
# the game's own per-frame work, kept apart from the renderer's
@ -268,6 +286,10 @@ function prof_gen_frame() -> void {
if prof_pre_swap != 0 and dtf != 0 { cpu = prof_pre_swap - (now - dtf) }
push(prof_cpu_us, cpu)
push(prof_game_us, prof_game_us_cur)
if len(prof_gen) > 9 and dtf != 0 {
prof_sum_dt = prof_sum_dt + dtf; prof_sum_cpu = prof_sum_cpu + cpu; prof_sum_game = prof_sum_game + prof_game_us_cur
prof_sum_n += 1
}
push(prof_mark_us, prof_mark_best)
if prof_mark_name == null { push(prof_mark_who, "-") } else { push(prof_mark_who, prof_mark_name) }
prof_gen_us_cur = 0; prof_layer_us_cur = 0; prof_bake_us_cur = 0; prof_up_cur = 0; prof_game_us_cur = 0
@ -319,6 +341,21 @@ function prof_ft_report() -> void {
print(` fps at median {string(1000000 / med)} frames over 1.3x median: {string(o13)}, over 1.5x: {string(o15)}, over 2x: {string(over)} — of {string(n)}`)
print(` stutter: {string(excess / 1000)} ms of frame time beyond 1.2x median over the run`)
print(` streaming totals over the run: walk {string(stream_us_walk / 1000)} ms (generate {string(stream_us_gen / 1000)} ms, gather {string(stream_us_gather / 1000)} ms), {string(stream_walks)} stream-walks`)
# the average frame, split: what the process did before the swap (the game's share of it apart),
# the rest waiting on the GPU and the swap, and the renderer's CPU phases in frame order
if prof_sum_n > 0 {
let dt = prof_sum_dt / prof_sum_n
let cpu = prof_sum_cpu / prof_sum_n
let game = prof_sum_game / prof_sum_n
print("")
print(`MEAN frame {string(dt)} us over {string(prof_sum_n)} frames: CPU before the swap {string(cpu)} us (the game's own {string(game)} us), GPU and swap {string(dt - cpu)} us`)
print(" renderer CPU by phase, mean per frame:")
var i = 0
while i < prof_mk_n {
print(` {Text.pad_right(prof_mk_names[i], 20)} {Text.pad_left(string(prof_mk_us[i] / prof_sum_n), 7)} us`)
i += 1
}
}
}
# The slowest frames of the run, with the once-in-a-while CPU work that landed in them.

View file

@ -959,8 +959,17 @@ var sc_cast_gen: int = -1
var sc_cast_dy: int = 0
var sc_cast_k: int = 0
function layer_level_casts_here(l: Layer, k: int) -> bool {
if l.lod_dist == null { return true }
var dmin = F_ZERO
if k > 0 { dmin = l.lod_dist[k - 1] }
return cast_band_reaches(dmin, l.lod_dist[k], l.lods[k].height)
}
# Can something standing between dmin and dmax metres from the camera (dmax 0: no outer limit), this
# tall, put a shadow on anything the cascade being rendered covers? The flowers' levels ask it
# (layer_level_casts_here), and so does every actor (actor_draw_casters).
function cast_band_reaches(dmin: int, dmax: int, height: int) -> bool {
if sc_cast_all < 0 { sc_cast_all = 0; if Os.has_env("R3D_CAST_ALL") { sc_cast_all = 1 } }
if sc_cast_all == 1 or sh_split == null or l.lod_dist == null { return true }
if sc_cast_all == 1 or sh_split == null { return true }
if sc_cast_gen != sc_view_gen {
sc_cast_gen = sc_view_gen
sc_cast_dy = f_add(f_abs(f_sub(cam_pos[1], terrain_height(cam_pos[0], cam_pos[2]))), fi(5))
@ -971,12 +980,11 @@ function layer_level_casts_here(l: Layer, k: int) -> bool {
}
let c = sh_cascade
var near = cam_near
if c > 0 { near = sh_split[c - 1] }
# sunShadow cross-fades into this cascade from 0.85 of the previous one's split (lighting.glsl), so
# its receivers start there, not at the split: starting at the split changed 19 pixels in town
if c > 0 { near = f_mul(sh_split[c - 1], fl(0.85)) }
let far = sh_split[c]
var dmin = F_ZERO
if k > 0 { dmin = l.lod_dist[k - 1] }
let dmax = l.lod_dist[k] # 0: open, out to the layer's cull distance
let reach = f_max(f_min(f_mul(l.lods[k].height, fi(6)), fi(40)), fi(8))
let reach = f_max(f_min(f_mul(height, fi(6)), fi(40)), fi(8))
if dmax != 0 and f_ls(f_add(f_add(dmax, reach), sc_cast_dy), near) { return false }
if f_gt(f_sub(dmin, reach), f_mul(far, sc_cast_k)) { return false }
return true
@ -1224,6 +1232,7 @@ function scatter_draw_casters(light_vp: words) -> void {
}
else { layer_draw_near(l, true, light_vp, true) }
}
prof_cpu_mark("shadow scatter")
}

View file

@ -156,6 +156,7 @@ function shadow_pass() -> void {
for c in 0 .. SHADOW_CASCADES {
sh_cascade = c
shadow_fit(c, near, sh_split[c])
prof_cpu_mark("shadow fit")
near = sh_split[c]
gpu_fb_depth_layer(sh_tex, c)
if r3d_debug and c == 0 { let st = gpu_fb_status(); print(`shadow fbo status {st}`) }