render3d: actors cast only into the cascades they can shade; Vulkan GPU timings and a mean frame split in R3D_PROF
Where the frame goes, before optimising it further. R3D_PROF now prints the average frame: CPU before the swap (the game's share apart), GPU and swap, and the renderer's CPU phases in frame order - the shadow pass split into cascade fit, scatter casters and actor casters. On Vulkan the per-pass GPU table comes from timestamp queries (host query reset asked for where the device has it); MoltenVK's attribution is tile-based and not to be trusted per pass, the PC's is. What it showed on the PC (camp): 4.3 ms CPU and 5.3 ms GPU a frame; the shadow pass was the largest CPU phase (1.8 ms) and actors half of that. Every actor within 300 m was drawn into all five cascades, though the outer two only shade receivers from 212 and 935 m out: 160 actors and 300 draws into each. cast_band_reaches - the flowers' reach test, now shared - skips an actor for a cascade it cannot shade (receivers counted from 0.85 of the previous split, where sunShadow's cross-fade begins). PC camp, two runs each: mean frame 9553/9601 -> 8664/8656 us; CPU 4.3 -> 3.7 ms; shadow GPU 1.14 -> 0.79 ms; actor-shadow CPU 1.02 -> 0.60 ms; 2039 -> 1411 draws; self-tests 59/59, validation 0. Mac: OpenGL shot viewpoints and the camp byte-identical; town 19 px at <= 2/255 on two flower stems a few metres from the camera - the accepted leftover-binding difference, no shadow; self-tests 59/59 on OpenGL and Vulkan. R3D_CAST_ALL=1 draws every caster into every cascade. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
ef745a141d
commit
5b9403c343
6 changed files with 134 additions and 10 deletions
|
|
@ -177,6 +177,13 @@ function gvk_init() -> bool {
|
|||
}
|
||||
gvk_has_dic = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_drawIndirectCount) == 1
|
||||
if gvk_has_dic { Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_drawIndirectCount, 1) }
|
||||
# R3D_PROF: per-pass GPU time from timestamp queries (gvk_query_*). The slots are reused every few
|
||||
# frames, so they are reset from the host - which is a feature to ask for.
|
||||
gvk_has_hqr = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_hostQueryReset) == 1 and Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampComputeAndGraphics) == 1
|
||||
if gvk_has_hqr {
|
||||
Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_hostQueryReset, 1)
|
||||
gvk_ts_period = Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampPeriod)
|
||||
}
|
||||
|
||||
Vk.put_i32(cnt, 0, 0)
|
||||
Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, null)
|
||||
|
|
@ -520,3 +527,48 @@ function gvk_shutdown() -> void {
|
|||
Vk.destroy_instance(gvk_inst, null)
|
||||
gvk_ready = false
|
||||
}
|
||||
|
||||
# ---- GPU timestamps (R3D_PROF) ------------------------------------------------------------
|
||||
# prof.ludic's query slots, each a pair of timestamps: 2 * id where the pass starts, 2 * id + 1 where it
|
||||
# ends. A slot is read back PROF_RING frames after it was written and reset just before it is written
|
||||
# again. On a device without host query reset or timestamps, every read says "not yet" and the report
|
||||
# stays empty.
|
||||
var gvk_has_hqr: bool = false
|
||||
var gvk_ts_period: int = 0 # float bits: nanoseconds per timestamp tick
|
||||
var gvk_qpool: long = 0
|
||||
var gvk_q_active: int = -1
|
||||
function gvk_query_new(n: int, ids: words) -> void {
|
||||
for i in 0 .. n { ids[i] = i }
|
||||
if not gvk_has_hqr or gvk_dev == null { return }
|
||||
let qci = bytes(VkQueryPoolCreateInfo_sizeof)
|
||||
Vk.zero(qci, VkQueryPoolCreateInfo_sizeof)
|
||||
Vk.put_i32(qci, VkQueryPoolCreateInfo_sType, VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO)
|
||||
Vk.put_i32(qci, VkQueryPoolCreateInfo_queryType, VK_QUERY_TYPE_TIMESTAMP)
|
||||
Vk.put_i32(qci, VkQueryPoolCreateInfo_queryCount, n * 2)
|
||||
let out = bytes(8)
|
||||
if Vk.create_query_pool(gvk_dev, qci, null, out) != VK_SUCCESS { return }
|
||||
gvk_qpool = gvk_handle(out)
|
||||
Vk.reset_query_pool(gvk_dev, gvk_qpool, 0, n * 2)
|
||||
}
|
||||
function gvk_query_begin(id: int) -> void {
|
||||
if gvk_qpool == 0 { return }
|
||||
Vk.reset_query_pool(gvk_dev, gvk_qpool, id * 2, 2)
|
||||
Vk.cmd_write_timestamp(gvk_frame_cb(), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, gvk_qpool, id * 2)
|
||||
gvk_q_active = id
|
||||
}
|
||||
function gvk_query_end() -> void {
|
||||
if gvk_qpool == 0 or gvk_q_active < 0 { return }
|
||||
Vk.cmd_write_timestamp(gvk_frame_cb(), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, gvk_qpool, gvk_q_active * 2 + 1)
|
||||
gvk_q_active = -1
|
||||
}
|
||||
function gvk_query_result(id: int, out: words) -> bool {
|
||||
if gvk_qpool == 0 { return false }
|
||||
let data = bytes(16)
|
||||
let size: long = 16
|
||||
let stride: long = 8
|
||||
if Vk.get_query_pool_results(gvk_dev, gvk_qpool, id * 2, 2, size, data, stride, VK_QUERY_RESULT_64_BIT) != VK_SUCCESS { return false }
|
||||
let ticks = Text.to_int(string(Vk.get_i64(data, 8) - Vk.get_i64(data, 0)))
|
||||
if ticks < 0 { return false }
|
||||
out[0] = f_to_int(f_mul(fi(ticks), gvk_ts_period))
|
||||
return true
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue