From 5b9403c343478ac5620e658f2064e9a95a7595db Mon Sep 17 00:00:00 2001 From: Orkuncakilkaya Date: Tue, 15 Sep 2026 18:03:31 +0300 Subject: [PATCH] render3d: actors cast only into the cascades they can shade; Vulkan GPU timings and a mean frame split in R3D_PROF Where the frame goes, before optimising it further. R3D_PROF now prints the average frame: CPU before the swap (the game's share apart), GPU and swap, and the renderer's CPU phases in frame order - the shadow pass split into cascade fit, scatter casters and actor casters. On Vulkan the per-pass GPU table comes from timestamp queries (host query reset asked for where the device has it); MoltenVK's attribution is tile-based and not to be trusted per pass, the PC's is. What it showed on the PC (camp): 4.3 ms CPU and 5.3 ms GPU a frame; the shadow pass was the largest CPU phase (1.8 ms) and actors half of that. Every actor within 300 m was drawn into all five cascades, though the outer two only shade receivers from 212 and 935 m out: 160 actors and 300 draws into each. cast_band_reaches - the flowers' reach test, now shared - skips an actor for a cascade it cannot shade (receivers counted from 0.85 of the previous split, where sunShadow's cross-fade begins). PC camp, two runs each: mean frame 9553/9601 -> 8664/8656 us; CPU 4.3 -> 3.7 ms; shadow GPU 1.14 -> 0.79 ms; actor-shadow CPU 1.02 -> 0.60 ms; 2039 -> 1411 draws; self-tests 59/59, validation 0. Mac: OpenGL shot viewpoints and the camp byte-identical; town 19 px at <= 2/255 on two flower stems a few metres from the camera - the accepted leftover-binding difference, no shadow; self-tests 59/59 on OpenGL and Vulkan. R3D_CAST_ALL=1 draws every caster into every cascade. Co-Authored-By: Claude Opus 5 --- packages/ludic.render3d/actor.ludic | 25 +++++++++++++ packages/ludic.render3d/gpu.ludic | 8 ++--- packages/ludic.render3d/gpu_vk.ludic | 52 +++++++++++++++++++++++++++ packages/ludic.render3d/prof.ludic | 37 +++++++++++++++++++ packages/ludic.render3d/scatter.ludic | 21 +++++++---- packages/ludic.render3d/shadow.ludic | 1 + 6 files changed, 134 insertions(+), 10 deletions(-) diff --git a/packages/ludic.render3d/actor.ludic b/packages/ludic.render3d/actor.ludic index deb5a41e..f53b821b 100644 --- a/packages/ludic.render3d/actor.ludic +++ b/packages/ludic.render3d/actor.ludic @@ -370,10 +370,35 @@ function actor_draw_casters(light_vp: words) -> void { let r = f_max(f_mul(a.radius, a.scale), hh) if not ac_in_light(light_vp, a.x, f_add(a.y, hh), a.z, f_mul(r, fl(1.5))) { continue } } + # Inside the light box is not the same as able to shade anything this cascade covers: every actor + # within 300 m sits inside the far cascades' boxes, whose receivers start 250 and 1100 m out - at + # the camp 160 actors cast 300 draws into each of those, against 41 into the nearest. + let dxa = f_sub(a.x, cam_pos[0]); let dza = f_sub(a.z, cam_pos[2]) + let da = f_sqrt(f_add(f_mul(dxa, dxa), f_mul(dza, dza))) + let ra = f_mul(a.radius, a.scale) + if not cast_band_reaches(f_max(f_sub(da, ra), F_ZERO), f_add(da, ra), f_mul(a.model.height, a.scale)) { continue } var ap = ac_sh if a.cutout { ap = ac_sh_cut } + if ac_census_at > 0 and ac_frame == ac_census_at { ac_caster_count(a) } actor_draw_one(a, ap, true) } + if ac_census_at > 0 and ac_frame == ac_census_at { + print(`actor census, cascade {sh_cascade}: {ac_cc_actors} actors cast {ac_cc_draws} draws ({ac_cc_skinned} of them skinned, {ac_cc_cut} cutout)`) + ac_cc_actors = 0; ac_cc_draws = 0; ac_cc_skinned = 0; ac_cc_cut = 0 + } + prof_cpu_mark("shadow actors") +} +var ac_cc_actors: int = 0 +var ac_cc_draws: int = 0 +var ac_cc_skinned: int = 0 +var ac_cc_cut: int = 0 +function ac_caster_count(a: Actor) -> void { + var shown = 0 + for p in 0 .. len(a.model.prims) { if a.hide == null or a.hide[p] == 0 { shown += 1 } } + ac_cc_actors += 1 + ac_cc_draws += shown + if a.skin != null or a.model.skin != null { ac_cc_skinned += shown } + if a.cutout { ac_cc_cut += shown } } # every actor off the stage at once, for a world being replaced. Actors own no GL objects; diff --git a/packages/ludic.render3d/gpu.ludic b/packages/ludic.render3d/gpu.ludic index 740278d2..da690a3a 100644 --- a/packages/ludic.render3d/gpu.ludic +++ b/packages/ludic.render3d/gpu.ludic @@ -216,12 +216,12 @@ function gpu_program_free(p: int) -> void { } # ---- GPU timers (R3D_PROF) ---------------------------------------------------------- -function gpu_query_new(n: int, ids: words) -> void { if gpu_kind == GPU_VK { return }; gl_gen_queries(n, ids) } -function gpu_query_begin(id: int) -> void { if gpu_kind == GPU_VK { return }; gl_begin_query(GL_TIME_ELAPSED, id) } -function gpu_query_end() -> void { if gpu_kind == GPU_VK { return }; gl_end_query(GL_TIME_ELAPSED) } +function gpu_query_new(n: int, ids: words) -> void { if gpu_kind == GPU_VK { gvk_query_new(n, ids); return }; gl_gen_queries(n, ids) } +function gpu_query_begin(id: int) -> void { if gpu_kind == GPU_VK { gvk_query_begin(id); return }; gl_begin_query(GL_TIME_ELAPSED, id) } +function gpu_query_end() -> void { if gpu_kind == GPU_VK { gvk_query_end(); return }; gl_end_query(GL_TIME_ELAPSED) } # true once the query has its result; the nanoseconds (low 32 bits) are then in out[0] function gpu_query_result(id: int, out: words) -> bool { - if gpu_kind == GPU_VK { return false } + if gpu_kind == GPU_VK { return gvk_query_result(id, out) } gl_get_query_objectiv(id, GL_QUERY_RESULT_AVAILABLE, out) if out[0] == 0 { return false } gl_get_query_objectui64v(id, GL_QUERY_RESULT, out) diff --git a/packages/ludic.render3d/gpu_vk.ludic b/packages/ludic.render3d/gpu_vk.ludic index c818ed97..265c7cd7 100644 --- a/packages/ludic.render3d/gpu_vk.ludic +++ b/packages/ludic.render3d/gpu_vk.ludic @@ -177,6 +177,13 @@ function gvk_init() -> bool { } gvk_has_dic = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_drawIndirectCount) == 1 if gvk_has_dic { Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_drawIndirectCount, 1) } + # R3D_PROF: per-pass GPU time from timestamp queries (gvk_query_*). The slots are reused every few + # frames, so they are reset from the host - which is a feature to ask for. + gvk_has_hqr = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_hostQueryReset) == 1 and Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampComputeAndGraphics) == 1 + if gvk_has_hqr { + Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_hostQueryReset, 1) + gvk_ts_period = Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampPeriod) + } Vk.put_i32(cnt, 0, 0) Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, null) @@ -520,3 +527,48 @@ function gvk_shutdown() -> void { Vk.destroy_instance(gvk_inst, null) gvk_ready = false } + +# ---- GPU timestamps (R3D_PROF) ------------------------------------------------------------ +# prof.ludic's query slots, each a pair of timestamps: 2 * id where the pass starts, 2 * id + 1 where it +# ends. A slot is read back PROF_RING frames after it was written and reset just before it is written +# again. On a device without host query reset or timestamps, every read says "not yet" and the report +# stays empty. +var gvk_has_hqr: bool = false +var gvk_ts_period: int = 0 # float bits: nanoseconds per timestamp tick +var gvk_qpool: long = 0 +var gvk_q_active: int = -1 +function gvk_query_new(n: int, ids: words) -> void { + for i in 0 .. n { ids[i] = i } + if not gvk_has_hqr or gvk_dev == null { return } + let qci = bytes(VkQueryPoolCreateInfo_sizeof) + Vk.zero(qci, VkQueryPoolCreateInfo_sizeof) + Vk.put_i32(qci, VkQueryPoolCreateInfo_sType, VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO) + Vk.put_i32(qci, VkQueryPoolCreateInfo_queryType, VK_QUERY_TYPE_TIMESTAMP) + Vk.put_i32(qci, VkQueryPoolCreateInfo_queryCount, n * 2) + let out = bytes(8) + if Vk.create_query_pool(gvk_dev, qci, null, out) != VK_SUCCESS { return } + gvk_qpool = gvk_handle(out) + Vk.reset_query_pool(gvk_dev, gvk_qpool, 0, n * 2) +} +function gvk_query_begin(id: int) -> void { + if gvk_qpool == 0 { return } + Vk.reset_query_pool(gvk_dev, gvk_qpool, id * 2, 2) + Vk.cmd_write_timestamp(gvk_frame_cb(), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, gvk_qpool, id * 2) + gvk_q_active = id +} +function gvk_query_end() -> void { + if gvk_qpool == 0 or gvk_q_active < 0 { return } + Vk.cmd_write_timestamp(gvk_frame_cb(), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, gvk_qpool, gvk_q_active * 2 + 1) + gvk_q_active = -1 +} +function gvk_query_result(id: int, out: words) -> bool { + if gvk_qpool == 0 { return false } + let data = bytes(16) + let size: long = 16 + let stride: long = 8 + if Vk.get_query_pool_results(gvk_dev, gvk_qpool, id * 2, 2, size, data, stride, VK_QUERY_RESULT_64_BIT) != VK_SUCCESS { return false } + let ticks = Text.to_int(string(Vk.get_i64(data, 8) - Vk.get_i64(data, 0))) + if ticks < 0 { return false } + out[0] = f_to_int(f_mul(fi(ticks), gvk_ts_period)) + return true +} diff --git a/packages/ludic.render3d/prof.ludic b/packages/ludic.render3d/prof.ludic index 3b5dcc2a..8d8f101c 100644 --- a/packages/ludic.render3d/prof.ludic +++ b/packages/ludic.render3d/prof.ludic @@ -149,6 +149,10 @@ var prof_up: []long = null # the two are cleanly separable there. var prof_pre_swap: long = 0 var prof_cpu_us: []long = null +var prof_sum_dt: long = 0 # sums over the frames past the first eight, for the mean split +var prof_sum_cpu: long = 0 +var prof_sum_game: long = 0 +var prof_sum_n: int = 0 function prof_before_swap() -> void { if prof_on { prof_pre_swap = gl_now_us() } } # The single most expensive CPU phase of each frame, and what it was. A hitch is one @@ -160,12 +164,26 @@ var prof_mark_name: pointer = null var prof_mark_us: []long = null var prof_mark_who: []pointer = null function prof_mark_start() -> void { if prof_on { prof_mark_last = gl_now_us(); prof_mark_best = 0; prof_mark_name = null } } +# ... and every phase's share of the average frame, past the first eight frames (loading, first fills) +var prof_mk_names: []pointer = null +var prof_mk_us: []long = null +var prof_mk_n: int = 0 function prof_cpu_mark(name: pointer) -> void { if not prof_on { return } let now = gl_now_us() let d = now - prof_mark_last prof_mark_last = now if d > prof_mark_best { prof_mark_best = d; prof_mark_name = name } + var k = -1 + var i = 0 + while i < prof_mk_n { if prof_mk_names[i] == name { k = i }; i += 1 } + if k < 0 { + if prof_mk_names == null { prof_mk_names = new []pointer; prof_mk_us = new []long } + push(prof_mk_names, name); push(prof_mk_us, 0) + prof_mk_n += 1 + k = prof_mk_n - 1 + } + if prof_gen != null and len(prof_gen) > 8 { prof_mk_us[k] = prof_mk_us[k] + d } } # the game's own per-frame work, kept apart from the renderer's @@ -268,6 +286,10 @@ function prof_gen_frame() -> void { if prof_pre_swap != 0 and dtf != 0 { cpu = prof_pre_swap - (now - dtf) } push(prof_cpu_us, cpu) push(prof_game_us, prof_game_us_cur) + if len(prof_gen) > 9 and dtf != 0 { + prof_sum_dt = prof_sum_dt + dtf; prof_sum_cpu = prof_sum_cpu + cpu; prof_sum_game = prof_sum_game + prof_game_us_cur + prof_sum_n += 1 + } push(prof_mark_us, prof_mark_best) if prof_mark_name == null { push(prof_mark_who, "-") } else { push(prof_mark_who, prof_mark_name) } prof_gen_us_cur = 0; prof_layer_us_cur = 0; prof_bake_us_cur = 0; prof_up_cur = 0; prof_game_us_cur = 0 @@ -319,6 +341,21 @@ function prof_ft_report() -> void { print(` fps at median {string(1000000 / med)} frames over 1.3x median: {string(o13)}, over 1.5x: {string(o15)}, over 2x: {string(over)} — of {string(n)}`) print(` stutter: {string(excess / 1000)} ms of frame time beyond 1.2x median over the run`) print(` streaming totals over the run: walk {string(stream_us_walk / 1000)} ms (generate {string(stream_us_gen / 1000)} ms, gather {string(stream_us_gather / 1000)} ms), {string(stream_walks)} stream-walks`) + # the average frame, split: what the process did before the swap (the game's share of it apart), + # the rest waiting on the GPU and the swap, and the renderer's CPU phases in frame order + if prof_sum_n > 0 { + let dt = prof_sum_dt / prof_sum_n + let cpu = prof_sum_cpu / prof_sum_n + let game = prof_sum_game / prof_sum_n + print("") + print(`MEAN frame {string(dt)} us over {string(prof_sum_n)} frames: CPU before the swap {string(cpu)} us (the game's own {string(game)} us), GPU and swap {string(dt - cpu)} us`) + print(" renderer CPU by phase, mean per frame:") + var i = 0 + while i < prof_mk_n { + print(` {Text.pad_right(prof_mk_names[i], 20)} {Text.pad_left(string(prof_mk_us[i] / prof_sum_n), 7)} us`) + i += 1 + } + } } # The slowest frames of the run, with the once-in-a-while CPU work that landed in them. diff --git a/packages/ludic.render3d/scatter.ludic b/packages/ludic.render3d/scatter.ludic index 7f911ddf..3f007e73 100644 --- a/packages/ludic.render3d/scatter.ludic +++ b/packages/ludic.render3d/scatter.ludic @@ -959,8 +959,17 @@ var sc_cast_gen: int = -1 var sc_cast_dy: int = 0 var sc_cast_k: int = 0 function layer_level_casts_here(l: Layer, k: int) -> bool { + if l.lod_dist == null { return true } + var dmin = F_ZERO + if k > 0 { dmin = l.lod_dist[k - 1] } + return cast_band_reaches(dmin, l.lod_dist[k], l.lods[k].height) +} +# Can something standing between dmin and dmax metres from the camera (dmax 0: no outer limit), this +# tall, put a shadow on anything the cascade being rendered covers? The flowers' levels ask it +# (layer_level_casts_here), and so does every actor (actor_draw_casters). +function cast_band_reaches(dmin: int, dmax: int, height: int) -> bool { if sc_cast_all < 0 { sc_cast_all = 0; if Os.has_env("R3D_CAST_ALL") { sc_cast_all = 1 } } - if sc_cast_all == 1 or sh_split == null or l.lod_dist == null { return true } + if sc_cast_all == 1 or sh_split == null { return true } if sc_cast_gen != sc_view_gen { sc_cast_gen = sc_view_gen sc_cast_dy = f_add(f_abs(f_sub(cam_pos[1], terrain_height(cam_pos[0], cam_pos[2]))), fi(5)) @@ -971,12 +980,11 @@ function layer_level_casts_here(l: Layer, k: int) -> bool { } let c = sh_cascade var near = cam_near - if c > 0 { near = sh_split[c - 1] } + # sunShadow cross-fades into this cascade from 0.85 of the previous one's split (lighting.glsl), so + # its receivers start there, not at the split: starting at the split changed 19 pixels in town + if c > 0 { near = f_mul(sh_split[c - 1], fl(0.85)) } let far = sh_split[c] - var dmin = F_ZERO - if k > 0 { dmin = l.lod_dist[k - 1] } - let dmax = l.lod_dist[k] # 0: open, out to the layer's cull distance - let reach = f_max(f_min(f_mul(l.lods[k].height, fi(6)), fi(40)), fi(8)) + let reach = f_max(f_min(f_mul(height, fi(6)), fi(40)), fi(8)) if dmax != 0 and f_ls(f_add(f_add(dmax, reach), sc_cast_dy), near) { return false } if f_gt(f_sub(dmin, reach), f_mul(far, sc_cast_k)) { return false } return true @@ -1224,6 +1232,7 @@ function scatter_draw_casters(light_vp: words) -> void { } else { layer_draw_near(l, true, light_vp, true) } } + prof_cpu_mark("shadow scatter") } diff --git a/packages/ludic.render3d/shadow.ludic b/packages/ludic.render3d/shadow.ludic index b9be103a..379ee27b 100644 --- a/packages/ludic.render3d/shadow.ludic +++ b/packages/ludic.render3d/shadow.ludic @@ -156,6 +156,7 @@ function shadow_pass() -> void { for c in 0 .. SHADOW_CASCADES { sh_cascade = c shadow_fit(c, near, sh_split[c]) + prof_cpu_mark("shadow fit") near = sh_split[c] gpu_fb_depth_layer(sh_tex, c) if r3d_debug and c == 0 { let st = gpu_fb_status(); print(`shadow fbo status {st}`) }