feat(render3d): compute and indirect draws on Vulkan, and tree layers culled on the GPU (opt-in)

- Device: multiDrawIndirect, drawIndirectFirstInstance and drawIndirectCount where present.
- Buffers carry storage and indirect usage; a GPU-owned buffer is never swapped under a draw.
- Compute programs from shaders/compute.list (binding 0 parameters, 1.. storage buffers),
  built by `ludic-dev shaders`; gpu_compute / gpu_dispatch / gpu_draw_mesh_indirect in gpu.ludic.
- R3D_VK_PROBE=1: a dispatch read back (OK on the RTX 3070 Ti).
- scatter_cull.comp: a tree layer's frustum test and LOD split on the GPU, with the lit, prepass,
  impostor and shadow-LOD draws reading its records. Behind R3D_GPU_CULL=1 and off by default:
  at the camp it is slower (43.0 fps against 53.3), because the frame's cost is per-draw
  descriptor sets and it adds empty-level draws. Validation-clean; OpenGL frames unchanged.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-15 14:19:43 +03:00
parent 04cda22d19
commit c1eb5f399f
11 changed files with 463 additions and 15 deletions

View file

@ -509,6 +509,34 @@ function gpu_buffer_free(buf: int) -> void {
gl_delete_buffers(1, ids)
}
# ---- compute and indirect draws (Vulkan) ----------------------------------------------------
# The GPU-driven path: a compute program writes instance lists and draw commands into buffers
# the draws then read. OpenGL here is 4.1 (macOS) with no compute, so gpu_compute is 0 there and
# a caller keeps its CPU path. A compute program's binding 0 is its parameter block (params,
# copied at the dispatch); bindings 1.. are `bufs`, gpu buffers.
function gpu_has_compute() -> bool { return gpu_kind == GPU_VK }
function gpu_compute(name: string, n_bufs: int) -> int {
if gpu_kind != GPU_VK { return 0 }
return gvk_compute_new(name, n_bufs)
}
function gpu_dispatch(c: int, params: pointer, n_params: int, bufs: words, groups: int) -> void {
if gpu_kind == GPU_VK and c > 0 { gvk_dispatch(c, params, n_params, bufs, groups, 1, 1) }
}
# a buffer a compute pass writes (never reallocated under a draw that reads it)
function gpu_buffer_gpu_owned(buf: int) -> void { if gpu_kind == GPU_VK { gvk_buf_gpu_owned(buf) } }
# the host-visible contents of a buffer, for a readback after gpu_finish; null on OpenGL
function gpu_buffer_map(buf: int) -> pointer {
if gpu_kind != GPU_VK or buf <= 0 { return null }
return gvk_buf_map[buf]
}
function gpu_finish() -> void { if gpu_kind == GPU_VK { gvk_flush() } }
# n indexed draws of mesh m from VkDrawIndexedIndirectCommand records in buffer cmds at offset
# (bytes); each record's firstInstance selects its instances out of the bound instance buffer.
# With count_buf > 0 the GPU's own count (a uint at count_off) is used, up to n.
function gpu_draw_mesh_indirect(m: Mesh, cmds: int, offset: int, n: int, count_buf: int, count_off: int) -> void {
if gpu_kind == GPU_VK { gvk_draw_indirect_now(m, cmds, offset, n, count_buf, count_off) }
}
# drawing
function gpu_mesh_bind(m: Mesh) -> void { if gpu_kind == GPU_VK { return }; gl_bind_vertex_array(m.vao) }
function gpu_mesh_unbind() -> void { if gpu_kind == GPU_VK { return }; gl_bind_vertex_array(0) }

View file

@ -37,6 +37,9 @@ function gvk_ext_in(props: bytes, n: int, want: string) -> bool {
# Instance, the first discrete GPU (else the first listed), its first graphics queue, and a
# device with the Tier 1 floor switched on. False, with gvk_why set, if any of it is missing:
# the caller stays on OpenGL.
var gvk_has_mdi: bool = false # multiDrawIndirect + drawIndirectFirstInstance
var gvk_has_dic: bool = false # drawIndirectCount
function gvk_init() -> bool {
if gvk_ready { return true }
if Vk.open() == 0 { gvk_why = "no Vulkan loader"; return false }
@ -145,6 +148,16 @@ function gvk_init() -> bool {
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2)
Vk.put_ptr(want2, VkPhysicalDeviceFeatures2_pNext, want12)
if aniso { Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_samplerAnisotropy, 1) }
# GPU-driven drawing: one indirect buffer holds a layer's draws, and a compute pass may write
# how many of them there are. Asked for where the device has them; the renderer checks
# gvk_has_mdi / gvk_has_dic before it takes that path.
gvk_has_mdi = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect) == 1 and Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance) == 1
if gvk_has_mdi {
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect, 1)
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance, 1)
}
gvk_has_dic = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_drawIndirectCount) == 1
if gvk_has_dic { Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_drawIndirectCount, 1) }
Vk.put_i32(cnt, 0, 0)
Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, null)
@ -186,7 +199,7 @@ function gvk_init() -> bool {
if not gvk_cmd_init() { return false }
gvk_ready = true
print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x`)
print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x, multi-draw indirect {gvk_has_mdi}, indirect count {gvk_has_dic}`)
return true
}

View file

@ -118,6 +118,141 @@ function gvk_program(p: int, key: string, spv_dir: string) -> bool {
return true
}
# ---- compute ------------------------------------------------------------------------------
# A compute program is <name>.comp.spv beside the manifest, with bindings fixed by convention:
# 0 the parameter block (a uniform buffer, copied into the frame's ring at the dispatch) and
# 1 .. n storage buffers. Handles start at 1.
var gvk_cp_pipe: []long = null
var gvk_cp_layout: []long = null
var gvk_cp_dsl: []long = null
var gvk_cp_nbuf: []int = null
function gvk_compute_new(name: string, n_bufs: int) -> int {
let zero: long = 0
if gvk_cp_pipe == null {
gvk_cp_pipe = new []long; gvk_cp_layout = new []long; gvk_cp_dsl = new []long; gvk_cp_nbuf = new []int
push(gvk_cp_pipe, zero); push(gvk_cp_layout, zero); push(gvk_cp_dsl, zero); push(gvk_cp_nbuf, 0)
}
let module = gvk_module(`{gvk_spv_dir}/{name}.comp.spv`)
if module == 0 { return 0 }
let nb = n_bufs + 1
let bw = VkDescriptorSetLayoutBinding_sizeof
let binds = bytes(bw * nb)
Vk.zero(binds, bw * nb)
for k in 0 .. nb {
var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER
if k == 0 { kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER }
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, k)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, kind)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_COMPUTE_BIT)
}
let dslci = bytes(VkDescriptorSetLayoutCreateInfo_sizeof)
Vk.zero(dslci, VkDescriptorSetLayoutCreateInfo_sizeof)
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO)
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_bindingCount, nb)
Vk.put_ptr(dslci, VkDescriptorSetLayoutCreateInfo_pBindings, binds)
let dsl = bytes(8)
var r = Vk.create_descriptor_set_layout(gvk_dev, dslci, null, dsl)
if r != VK_SUCCESS { gvk_fail(`vkCreateDescriptorSetLayout for compute {name}`, r); return 0 }
let plci = bytes(VkPipelineLayoutCreateInfo_sizeof)
Vk.zero(plci, VkPipelineLayoutCreateInfo_sizeof)
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO)
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_setLayoutCount, 1)
Vk.put_ptr(plci, VkPipelineLayoutCreateInfo_pSetLayouts, dsl)
let out = bytes(8)
r = Vk.create_pipeline_layout(gvk_dev, plci, null, out)
if r != VK_SUCCESS { gvk_fail(`vkCreatePipelineLayout for compute {name}`, r); return 0 }
let layout = gvk_handle(out)
let cpci = bytes(VkComputePipelineCreateInfo_sizeof)
Vk.zero(cpci, VkComputePipelineCreateInfo_sizeof)
Vk.put_i32(cpci, VkComputePipelineCreateInfo_sType, VK_STRUCTURE_TYPE_COMPUTE_PIPELINE_CREATE_INFO)
let so = VkComputePipelineCreateInfo_stage
Vk.put_i32(cpci, so + VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
Vk.put_i32(cpci, so + VkPipelineShaderStageCreateInfo_stage, VK_SHADER_STAGE_COMPUTE_BIT)
Vk.put_i64(cpci, so + VkPipelineShaderStageCreateInfo_module, module)
Vk.put_ptr(cpci, so + VkPipelineShaderStageCreateInfo_pName, "main")
Vk.put_i64(cpci, VkComputePipelineCreateInfo_layout, layout)
r = Vk.create_compute_pipelines(gvk_dev, zero, 1, cpci, null, out)
if r != VK_SUCCESS { gvk_fail(`vkCreateComputePipelines {name}`, r); return 0 }
push(gvk_cp_pipe, gvk_handle(out)); push(gvk_cp_layout, layout); push(gvk_cp_dsl, Vk.get_i64(dsl, 0)); push(gvk_cp_nbuf, n_bufs)
return len(gvk_cp_pipe) - 1
}
# Record a dispatch of compute program c into the frame, between passes: params (n_params
# bytes, std140) as binding 0 and bufs[0 .. n) as bindings 1 .. n.
function gvk_dispatch(c: int, params: pointer, n_params: int, bufs: words, gx: int, gy: int, gz: int) -> void {
let zero: long = 0
if gvk_cp_pipe == null or c <= 0 or c >= len(gvk_cp_pipe) { return }
gvk_pass_end()
let cb = gvk_frame_cb()
let nb = gvk_cp_nbuf[c]
let dsai = bytes(VkDescriptorSetAllocateInfo_sizeof)
Vk.zero(dsai, VkDescriptorSetAllocateInfo_sizeof)
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO)
Vk.put_i64(dsai, VkDescriptorSetAllocateInfo_descriptorPool, gvk_dpool)
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_descriptorSetCount, 1)
let layouts = bytes(8)
Vk.put_i64(layouts, 0, gvk_cp_dsl[c])
Vk.put_ptr(dsai, VkDescriptorSetAllocateInfo_pSetLayouts, layouts)
let sets = bytes(8)
let r = Vk.allocate_descriptor_sets(gvk_dev, dsai, sets)
if r != VK_SUCCESS { gvk_fail("vkAllocateDescriptorSets (compute)", r); return }
let set = Vk.get_i64(sets, 0)
let at = gvk_ring_put(params, n_params)
if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return }
let ww = VkWriteDescriptorSet_sizeof
let bw = VkDescriptorBufferInfo_sizeof
let writes = bytes(ww * (nb + 1))
Vk.zero(writes, ww * (nb + 1))
let infos = bytes(bw * (nb + 1))
Vk.zero(infos, bw * (nb + 1))
for k in 0 .. nb + 1 {
var buf = gvk_ring_buf
var off: long = 0
var range: long = 0
var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER
if k == 0 { off = at; range = n_params; kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER }
else { buf = bufs[k - 1]; range = gvk_buf_size[buf]; gvk_buf_used[buf] = gvk_frame_no }
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_buffer, gvk_buf[buf])
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_offset, off)
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_range, range)
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET)
Vk.put_i64(writes, k * ww + VkWriteDescriptorSet_dstSet, set)
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_dstBinding, k)
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorCount, 1)
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorType, kind)
Vk.put_ptr(writes, k * ww + VkWriteDescriptorSet_pBufferInfo, mem_off(infos, k * bw))
}
Vk.update_descriptor_sets(gvk_dev, nb + 1, writes, 0, null)
Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_COMPUTE, gvk_cp_pipe[c])
Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_COMPUTE, gvk_cp_layout[c], 0, 1, sets, 0, null)
Vk.cmd_dispatch(cb, gx, gy, gz)
}
# R3D_VK_PROBE=1: one dispatch over a GPU-owned buffer, read back - the compute path end to end
function gvk_compute_probe() -> void {
let c = gvk_compute_new("probe", 1)
if c == 0 { print("r3d: vulkan compute probe: FAILED (no program)"); return }
let n = 1000
let b = gvk_buf_new()
let data = bytes(n * 4)
for i in 0 .. n { Vk.put_i32(data, i * 4, fi(i)) }
gvk_buf_upload(b, n * 4, data)
gvk_buf_gpu_owned(b)
let pr = bytes(8)
Vk.put_i32(pr, 0, n)
Vk.put_i32(pr, 4, fi(3))
let bufs = words(1)
bufs[0] = b
gvk_dispatch(c, pr, 8, bufs, (n + 63) / 64, 1, 1)
gvk_flush()
var bad = 0
let mp = gvk_buf_map[b]
for i in 0 .. n { if Vk.get_i32(mp, i * 4) != fi(i * 3) { bad += 1 } }
if bad == 0 { print(`r3d: vulkan compute probe OK ({n} values)`) } else { print(`r3d: vulkan compute probe: FAILED ({bad} of {n} wrong)`) }
}
# ---- pipelines ----------------------------------------------------------------------------
# A pipeline is everything OpenGL decides at the draw: the program, the vertex layout the mesh
# recorded, the render state the renderer set, and the formats and sample count of the pass it
@ -480,16 +615,18 @@ function gvk_frame_init() -> bool {
if gvk_ring_align < 16 { gvk_ring_align = 16 }
gvk_ring_buf = gvk_buf_new()
if not gvk_buf_reserve(gvk_ring_buf, GVK_RING_BYTES) { return false }
let sizes = bytes(VkDescriptorPoolSize_sizeof * 2)
let sizes = bytes(VkDescriptorPoolSize_sizeof * 3)
Vk.put_i32(sizes, VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER)
Vk.put_i32(sizes, VkDescriptorPoolSize_descriptorCount, 32768)
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_descriptorCount, 131072)
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof * 2 + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_STORAGE_BUFFER)
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof * 2 + VkDescriptorPoolSize_descriptorCount, 16384)
let dpci = bytes(VkDescriptorPoolCreateInfo_sizeof)
Vk.zero(dpci, VkDescriptorPoolCreateInfo_sizeof)
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO)
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_maxSets, 16384)
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_poolSizeCount, 2)
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_poolSizeCount, 3)
Vk.put_ptr(dpci, VkDescriptorPoolCreateInfo_pPoolSizes, sizes)
let out = bytes(8)
let r = Vk.create_descriptor_pool(gvk_dev, dpci, null, out)
@ -507,7 +644,7 @@ function gvk_frame_reset() -> void {
}
# a block into the ring; its offset, or -1 when the frame has used the whole ring
function gvk_ring_put(blk: bytes, n: int) -> int {
function gvk_ring_put(blk: pointer, n: int) -> int {
let at = (gvk_ring_off + gvk_ring_align - 1) / gvk_ring_align * gvk_ring_align
if at + n > GVK_RING_BYTES { return -1 }
mem_copy(mem_off(gvk_buf_map[gvk_ring_buf], at), blk, n)
@ -914,7 +1051,20 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc
if m.itype == GL_UNSIGNED_SHORT { itype = VK_INDEX_TYPE_UINT16 }
Vk.cmd_bind_index_buffer(cb, gvk_buf[m.ebo], zero, itype)
gvk_buf_used[m.ebo] = gvk_frame_no
Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0)
if gvk_ind_buf > 0 {
# the draws are records in a buffer (gvk_draw_indirect_now); the GPU may have written them
let ioff: long = gvk_ind_off
gvk_buf_used[gvk_ind_buf] = gvk_frame_no
if gvk_ind_cbuf > 0 and gvk_has_dic {
let coff: long = gvk_ind_coff
gvk_buf_used[gvk_ind_cbuf] = gvk_frame_no
Vk.cmd_draw_indexed_indirect_count(cb, gvk_buf[gvk_ind_buf], ioff, gvk_buf[gvk_ind_cbuf], coff, gvk_ind_n, VkDrawIndexedIndirectCommand_sizeof)
} else {
Vk.cmd_draw_indexed_indirect(cb, gvk_buf[gvk_ind_buf], ioff, gvk_ind_n, VkDrawIndexedIndirectCommand_sizeof)
}
} else {
Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0)
}
} else {
Vk.cmd_draw(cb, n, instances, first, 0)
}
@ -1002,6 +1152,7 @@ function gvk_open(w: int, h: int, title: string) -> bool {
}
if not gvk_frame_init() { return false }
if not gvk_screen_make(gl_w, gl_h) { return false }
if Os.has_env("R3D_VK_PROBE") { gvk_compute_probe() }
if is_windowed() { return gvk_swap_make(gl_w, gl_h) }
return true
}
@ -1073,6 +1224,20 @@ function gvk_state_now() -> GvkState {
st.wireframe = gvk_wireframe
return st
}
# An indirect draw goes through gvk_draw like any other - the same pipeline, set and buffers -
# and only its last call differs, so the record is handed over in these for that one draw.
var gvk_ind_buf: int = 0
var gvk_ind_off: int = 0
var gvk_ind_n: int = 0
var gvk_ind_cbuf: int = 0
var gvk_ind_coff: int = 0
function gvk_draw_indirect_now(m: Mesh, cmds: int, offset: int, n: int, count_buf: int, count_off: int) -> void {
if m == null or m.ebo == 0 or cmds <= 0 or n <= 0 { return }
gvk_ind_buf = cmds; gvk_ind_off = offset; gvk_ind_n = n; gvk_ind_cbuf = count_buf; gvk_ind_coff = count_off
gvk_draw_now(m, 0, 0, 1)
gvk_ind_buf = 0; gvk_ind_cbuf = 0
}
function gvk_draw_now(m: Mesh, first: int, count: int, instances: int) -> void {
gvk_draw(gpu_prog_cur, m, gvk_state_now(), first, count, instances, gpu_tx, GPU_TX_W, gpu_tx_cap, gpu_fb, gpu_fb_at(gvk_fb_cur))
}

View file

@ -561,12 +561,22 @@ function gvk_retire_flush() -> void {
gvk_retired_mem = new []long
}
# A buffer a compute pass writes: a draw later in the same frame reads what the GPU put there,
# so it is never swapped for fresh storage because it was used this frame (gvk_buf_reserve).
var gvk_buf_gpu: []int = null
function gvk_buf_gpu_owned(b: int) -> void {
if gvk_buf_gpu == null { gvk_buf_gpu = new []int }
while len(gvk_buf_gpu) <= b { push(gvk_buf_gpu, 0) }
gvk_buf_gpu[b] = 1
}
function gvk_buf_is_gpu(b: int) -> bool { return gvk_buf_gpu != null and b < len(gvk_buf_gpu) and gvk_buf_gpu[b] == 1 }
# Room for at least n bytes behind handle b; a buffer that is already big enough is kept, so a
# stream re-filled every frame allocates once.
function gvk_buf_reserve(b: int, n: int) -> bool {
if b <= 0 or b >= len(gvk_buf) { return false }
# already big enough, and no draw this frame reads what is there: fill it in place
if gvk_buf[b] != 0 and gvk_buf_size[b] >= n and gvk_buf_used[b] != gvk_frame_no { return true }
if gvk_buf[b] != 0 and gvk_buf_size[b] >= n and (gvk_buf_used[b] != gvk_frame_no or gvk_buf_is_gpu(b)) { return true }
gvk_buf_release(b)
var size = n
if size < 64 { size = 64 }
@ -575,7 +585,7 @@ function gvk_buf_reserve(b: int, n: int) -> bool {
Vk.zero(bci, VkBufferCreateInfo_sizeof)
Vk.put_i32(bci, VkBufferCreateInfo_sType, VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO)
Vk.put_i64(bci, VkBufferCreateInfo_size, size_l)
Vk.put_i32(bci, VkBufferCreateInfo_usage, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT)
Vk.put_i32(bci, VkBufferCreateInfo_usage, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT)
Vk.put_i32(bci, VkBufferCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE)
let out = bytes(8)
var r = Vk.create_buffer(gvk_dev, bci, null, out)

View file

@ -61,6 +61,15 @@ property Layer {
# crossed card carrying its atlas (cover keeps its baked card as the far level).
lods: []Model,
n_lods: int = 0,
# The GPU-culled path (Vulkan, phase 38): every instance in g_src, a compute pass packs the
# visible ones per bucket into g_dst and writes the instance counts of the draw records in
# g_cmds (see scatter_cull.comp for the record layout). g_on once it is set up for g_n instances.
g_on: bool = false,
g_n: int = 0,
g_src: int = 0,
g_dst: int = 0,
g_cmds: int = 0,
g_counts: int = 0,
lod_dist: words,
lod_card: words,
lod_buf: words,
@ -525,6 +534,88 @@ function scatter_begin_frame() -> void {
v3_copy(sc_view_pos, cam_pos)
v3_copy(sc_view_fwd, cam_fwd)
}
# GPU-culled layers dispatch before the first pass of the frame, so no pass is split for it
if sc_layers != null {
for i in 0 .. len(sc_layers) { if sc_layers[i].n_lods > 1 and sc_layers[i].imp != null { layer_update(sc_layers[i]) } }
}
}
# ---- the GPU-culled path ------------------------------------------------------------------
# R3D_GPU_CULL=1 on Vulkan: a tree layer (a LOD chain of up to four levels of up to four prims,
# with an impostor, not streamed) is culled and split into its buckets by scatter_cull.comp, and
# its lit, prepass, impostor and shadow-LOD draws read the records that pass wrote. Nothing is
# partitioned or uploaded on the CPU when the view moves.
const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand
const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp)
var sc_cull_prog: int = 0
var sc_cull_tried: bool = false
var sc_ind_base: int = -1 # >= 0: layer_draw_model / layer_draw_depth draw records from here
function layer_gpu_eligible(l: Layer) -> bool {
if l.n_lods < 2 or l.n_lods > 4 or l.imp == null or l.streamed or l.flower or l.blade or l.count == 0 { return false }
for k in 0 .. l.n_lods { if len(l.lods[k].prims) > 4 { return false } }
return true
}
function layer_gpu_prepare(l: Layer) -> bool {
if not sc_cull_tried {
sc_cull_tried = true
# Os.env is null when the variable is unset, and a compare reads through it: ask first
if gpu_has_compute() and Os.has_env("R3D_GPU_CULL") and Os.env("R3D_GPU_CULL") == "1" {
sc_cull_prog = gpu_compute("scatter_cull", 4)
if sc_cull_prog > 0 { print("r3d: scatter: tree layers are culled on the GPU") }
}
}
if sc_cull_prog == 0 or not layer_gpu_eligible(l) { return false }
if l.g_on and l.g_n == l.count { return true }
if l.g_src == 0 {
l.g_src = gpu_buffer_new(); l.g_dst = gpu_buffer_new(); l.g_cmds = gpu_buffer_new(); l.g_counts = gpu_buffer_new()
gpu_buffer_gpu_owned(l.g_dst); gpu_buffer_gpu_owned(l.g_cmds); gpu_buffer_gpu_owned(l.g_counts)
}
let n = l.n_lods
let cap = l.count
gpu_buffer_upload(l.g_src, cap * INST_FLOATS * 4, l.inst, GPU_STATIC)
gpu_buffer_upload(l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC)
let rec = words(SC_RECS * 5)
for i in 0 .. SC_RECS * 5 { rec[i] = 0 }
for k in 0 .. n {
let m = l.lods[k]
for j in 0 .. len(m.prims) { rec[(k * 4 + j) * 5] = m.prims[j].mesh.count; rec[(k * 4 + j) * 5 + 4] = k * cap }
}
rec[16 * 5] = sc_card.count; rec[16 * 5 + 4] = n * cap
if n > 2 {
let m2 = l.lods[2]
for b in 0 .. 3 {
for j in 0 .. len(m2.prims) { rec[(17 + b * 4 + j) * 5] = m2.prims[j].mesh.count; rec[(17 + b * 4 + j) * 5 + 4] = b * cap }
}
}
gpu_buffer_upload(l.g_cmds, SC_RECS * SC_REC_W, rec, GPU_DYNAMIC)
let zeros = words(5)
for i in 0 .. 5 { zeros[i] = 0 }
gpu_buffer_upload(l.g_counts, 20, zeros, GPU_DYNAMIC)
free(rec); free(zeros)
# the card casts every instance, as on the CPU path (layer_grid_build uploads this there)
l.n_sh = l.count
gpu_buffer_upload(l.sh_buf, l.count * INST_FLOATS * 4, l.inst, GPU_STATIC)
l.g_n = l.count
l.g_on = true
return true
}
# the dispatch for the view as it stands: frustum, camera, distances, the layer's shape
function layer_gpu_cull(l: Layer) -> void {
let pr = words(36)
for i in 0 .. 36 { pr[i] = 0 }
if cam_planes != null { for i in 0 .. 16 { pr[i] = cam_planes[i] } }
pr[16] = cam_pos[0]; pr[17] = cam_pos[1]; pr[18] = cam_pos[2]; pr[19] = l.cull
for k in 0 .. l.n_lods { pr[20 + k] = l.lod_dist[k]; pr[24 + k] = len(l.lods[k].prims) }
pr[28] = l.count; pr[29] = l.count; pr[30] = l.n_lods; pr[31] = 1
# as layer_grid_gather pads a cell: the tallest instance, plus a margin
pr[32] = f_mul(l.lods[0].height, F_TWO); pr[33] = fi(4)
let bufs = words(4)
bufs[0] = l.g_src; bufs[1] = l.g_dst; bufs[2] = l.g_cmds; bufs[3] = l.g_counts
gpu_dispatch(sc_cull_prog, pr, 144, bufs, 1)
free(pr); free(bufs)
}
# Sort a static layer's instances into square cells (call once, after placement; a
@ -667,6 +758,7 @@ function layer_update(l: Layer) -> void {
if sc_freeze { return }
if l.view_gen == sc_view_gen { return }
l.view_gen = sc_view_gen
if layer_gpu_prepare(l) { layer_gpu_cull(l); return }
let t_lu = gl_now_us()
let n_lu = l.count
# A streamed layer's instances were already gathered per visible chunk: no split, no
@ -764,6 +856,11 @@ var sc_dbg_tint: words = null
# no cheap stand-in to cast from, so without this its shadow simply began at the near
# distance — which is the crown shadow that appeared as you walked up to a tree.
function layer_draw_near(l: Layer, shadow: bool, light_vp: words, full: bool) -> void {
if l.n_lods > 1 and l.g_on and not shadow {
for k in 0 .. l.n_lods { sc_dbg_level = k; sc_ind_base = k * 4; layer_draw_model(l, l.lods[k], l.g_dst, 1, l.lod_card[k] == 1, shadow, light_vp) }
sc_ind_base = -1; sc_dbg_level = -1
return
}
if l.n_lods > 1 {
# a LOD chain: every level from its own bucket (casters are what is drawn)
for k in 0 .. l.n_lods { sc_dbg_level = k; layer_draw_model(l, l.lods[k], l.lod_buf[k], l.n_lod[k], l.lod_card[k] == 1, shadow, light_vp) }
@ -829,14 +926,15 @@ function layer_draw_model(l: Layer, model: Model, vb: int, cnt: int, card: bool,
r3d_bind_2d(p, "u_diff", 0, pr.diff)
if not shadow { r3d_bind_2d(p, "u_nrm", 1, pr.nrm); r3d_bind_2d(p, "u_arm", 2, pr.arm) }
}
mesh_draw_instanced(pr.mesh, cnt)
if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i) * SC_REC_W, 1, 0, 0) }
else { mesh_draw_instanced(pr.mesh, cnt) }
}
gpu_alpha_to_coverage(false)
if p == sc_prog_fol_eq { gpu_depth_func(GL_LESS) }
}
function layer_draw_far(l: Layer, shadow: bool, light_vp: words) -> void {
if l.n_far == 0 or l.imp == null { return }
if l.imp == null or (l.n_far == 0 and not l.g_on) { return }
var p = sc_imp_prog
if shadow { p = sc_imp_prog_shadow }
gpu_use_program(p)
@ -863,8 +961,13 @@ function layer_draw_far(l: Layer, shadow: bool, light_vp: words) -> void {
}
gpu_cull(false)
if not shadow and sc_a2c { gpu_alpha_to_coverage(true) }
scatter_attach(sc_card, l.imp_buf)
mesh_draw_instanced(sc_card, l.n_far)
if l.g_on {
scatter_attach(sc_card, l.g_dst)
gpu_draw_mesh_indirect(sc_card, l.g_cmds, 16 * SC_REC_W, 1, 0, 0)
} else {
scatter_attach(sc_card, l.imp_buf)
mesh_draw_instanced(sc_card, l.n_far)
}
gpu_alpha_to_coverage(false)
}
@ -915,7 +1018,8 @@ function layer_draw_depth(l: Layer, model: Model, vb: int, cnt: int) -> void {
let pr = model.prims[i]
scatter_attach(pr.mesh, vb)
r3d_bind_2d(p, "u_diff", 0, pr.diff)
mesh_draw_instanced(pr.mesh, cnt)
if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i) * SC_REC_W, 1, 0, 0) }
else { mesh_draw_instanced(pr.mesh, cnt) }
}
}
@ -928,7 +1032,10 @@ function scatter_draw_depth() -> void {
if not l.foliage or l.blade or l.flower { continue }
if r3d_no_trees and l.imp != null { continue }
layer_update(l)
if l.n_lods > 1 {
if l.n_lods > 1 and l.g_on {
for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { sc_ind_base = k * 4; layer_draw_depth(l, l.lods[k], l.g_dst, 1) } }
sc_ind_base = -1
} else if l.n_lods > 1 {
# (a tree layer is flagged `card` for its distant level; its mesh levels still count)
for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { layer_draw_depth(l, l.lods[k], l.lod_buf[k], l.n_lod[k]) } }
} else if not l.card {
@ -969,7 +1076,10 @@ function scatter_draw_casters(light_vp: words) -> void {
# crown of drooping needle cards is mostly slivers from the side, so the sun, which
# sees the crown from above, cast a trunk line with a few blobs. The near levels
# now also cast their LOD2 mesh, alpha-tested, on top of the card.
if l.n_lods > 2 {
if l.n_lods > 2 and l.g_on {
for k in 0 .. 3 { sc_ind_base = 17 + k * 4; layer_draw_model(l, l.lods[2], l.g_dst, 1, false, true, light_vp) }
sc_ind_base = -1
} else if l.n_lods > 2 {
for k in 0 .. 3 { layer_draw_model(l, l.lods[2], l.lod_buf[k], l.n_lod[k], false, true, light_vp) }
}
}

View file

@ -0,0 +1,4 @@
# compute.list - render3d's compute programs: name|file.comp. Built by `ludic-dev shaders` into
# spv/<name>.comp.spv. Binding 0 is the uniform block of parameters, 1.. storage buffers.
probe|probe.comp
scatter_cull|scatter_cull.comp

View file

@ -0,0 +1,10 @@
// probe.comp - the compute path's own check (R3D_VK_PROBE=1): every float in the buffer times
// the parameter, so a readback proves the dispatch, the bindings and the GPU-owned buffer.
layout(local_size_x = 64) in;
layout(set = 0, binding = 0) uniform Params { uint count; float mul; } pr;
layout(set = 0, binding = 1) buffer Data { float v[]; } data;
void main() {
uint i = gl_GlobalInvocationID.x;
if (i >= pr.count) { return; }
data.v[i] = data.v[i] * pr.mul;
}

View file

@ -0,0 +1,82 @@
// scatter_cull.comp - a scatter layer's per-frame culling on the GPU (phase 38, stage 2).
//
// Every instance of the layer (8 floats: x, y, z, scale, sin yaw, cos yaw, seed, wind) is tested
// against the view frustum and sorted into the layer's LOD buckets by distance - the same rule as
// layer_partition_lods on the CPU. Bucket k's instances are packed from out[k * cap], and the
// instanceCount of each indirect draw command for that bucket is written here, so the draws that
// follow read exactly what survived. Bucket n_lods is the impostor bucket.
//
// One invocation walks the list in order: packing needs a running count per bucket, and a walk of
// tens of thousands of instances is a few hundred microseconds of GPU time - far below the CPU
// partition and upload it replaces. A parallel prefix-sum version can come later.
layout(local_size_x = 1) in;
layout(set = 0, binding = 0) uniform Params {
vec4 planes[4]; // the side frustum planes (cam_planes): xyz in, w distance; inside when dot >= -r
vec4 cam; // xyz the camera, w the cull distance (0 = none)
vec4 lod_dist; // outer distance of each level (0 = open: runs out to the cull distance)
uvec4 prims; // the prims each level draws (at most 4)
uint count; // instances in the layer
uint cap; // instances each bucket can hold
uint n_lods; // levels (1 .. 4)
uint has_imp; // 1 when the layer has an impostor bucket
float pad; // sphere radius per unit of instance scale (the model's bound)
float pad_abs; // added to every radius (m)
} pr;
layout(set = 0, binding = 1) readonly buffer Src { float src[]; };
layout(set = 0, binding = 2) buffer Dst { float dst[]; };
// VkDrawIndexedIndirectCommand records, 5 uints each; the CPU fills indexCount / firstIndex /
// vertexOffset / firstInstance once, this writes instanceCount:
// 0 .. 15 level * 4 + prim the lit pass and the depth prepass
// 16 the impostor card bucket n_lods
// 17 .. 28 17 + bucket * 4 + prim level 2's mesh cast from buckets 0 .. 2 (the shadow LOD)
layout(set = 0, binding = 3) buffer Cmds { uint cmds[]; };
// how many instances landed in each bucket (levels, then the impostor), for the CPU to read
layout(set = 0, binding = 4) buffer Counts { uint counts[5]; };
void main() {
uint n = pr.n_lods;
uint cnt[5] = uint[5](0u, 0u, 0u, 0u, 0u);
float cull2 = pr.cam.w * pr.cam.w;
bool open = pr.lod_dist[n - 1u] == 0.0;
for (uint i = 0u; i < pr.count; i++) {
uint o = i * 8u;
vec3 p = vec3(src[o], src[o + 1u], src[o + 2u]);
float dx = p.x - pr.cam.x;
float dz = p.z - pr.cam.z;
float d2 = dx * dx + dz * dz;
if (pr.cam.w != 0.0 && d2 > cull2) { continue; }
float r = pr.pad * src[o + 3u] + pr.pad_abs;
bool inside = true;
for (int k = 0; k < 4; k++) {
if (dot(pr.planes[k].xyz, p) + pr.planes[k].w < -r) { inside = false; break; }
}
if (!inside) { continue; }
float d = sqrt(d2);
uint lv = n;
for (uint k = 0u; k < n; k++) {
if (pr.lod_dist[k] != 0.0 && d < pr.lod_dist[k]) { lv = k; break; }
}
if (lv == n && open) { lv = n - 1u; }
if (lv == n && pr.has_imp == 0u) { continue; }
if (cnt[lv] >= pr.cap) { continue; }
uint q = (lv * pr.cap + cnt[lv]) * 8u;
for (uint k = 0u; k < 8u; k++) { dst[q + k] = src[o + k]; }
cnt[lv] += 1u;
}
for (uint k = 0u; k < 4u; k++) {
for (uint j = 0u; j < 4u; j++) {
uint c = (k * 4u + j) * 5u;
cmds[c + 1u] = (k < n && j < pr.prims[k]) ? cnt[k] : 0u;
}
}
cmds[16u * 5u + 1u] = (pr.has_imp != 0u) ? cnt[n] : 0u;
for (uint b = 0u; b < 3u; b++) {
for (uint j = 0u; j < 4u; j++) {
uint c = (17u + b * 4u + j) * 5u;
cmds[c + 1u] = (n > 2u && b < n && j < pr.prims[2]) ? cnt[b] : 0u;
}
}
for (uint k = 0u; k < 5u; k++) { counts[k] = cnt[k]; }
}

Binary file not shown.

View file

@ -280,6 +280,32 @@ function cmd_shaders() -> int {
}
made += 1
}
# Compute programs: compute.list names each one (name|file.comp). The renderer loads
# <name>.comp.spv straight from the directory; its bindings are fixed by convention - binding 0
# a uniform block of parameters, 1.. storage buffers - so there is nothing to reflect.
var made_c = 0
let clist = read_file("packages/ludic.render3d/shaders/compute.list")
if clist != null {
let clines = Text.split(clist, "\n")
var ci = 0
while ci < len(clines) {
let cl = Text.trim(clines[ci])
ci += 1
if slen(cl) == 0 or cl[0] == '#' { continue }
let cp = Text.split(cl, "|")
if len(cp) < 2 { err(`shaders: bad compute line: {cl}\n`); return 1 }
let glsl = `{tmp}/{cp[0]}.comp.glsl`
if not write_file(glsl, "#version 460\n" + shd_file(cp[1])) { err(`shaders: cannot write {glsl}\n`); return 1 }
let spv = `{outdir}/{cp[0]}.comp.spv`
let logf = `{tmp}/{cp[0]}.comp.log`
if not shq(`{shd_q(shd_bin + "/glslangValidator")} -V -S comp {shd_q(glsl)} -o {shd_q(spv)} > {shd_q(logf)} 2>&1`) {
err(`shaders: {cp[1]} (compute) did not compile:\n{capture("grep ERROR " + shd_q(logf) + " | head -5")}\n`)
return 1
}
if not shq(`{shd_q(shd_bin + "/spirv-val")} --target-env vulkan1.3 {shd_q(spv)} > /dev/null 2>&1`) { err(`shaders: {spv} fails spirv-val\n`); return 1 }
made_c += 1
}
}
if not write_file(`{outdir}/manifest.txt`, sb_str(mf)) { err("shaders: cannot write manifest.txt\n"); return 1 }
if check {
let real = "packages/ludic.render3d/shaders/spv"
@ -287,6 +313,6 @@ function cmd_shaders() -> int {
}
var note = ""
if check { note = " (unchanged)" }
print(`OK {string(made)} programs, {string(made * 2)} SPIR-V stages{note}`)
print(`OK {string(made)} programs, {string(made * 2 + made_c)} SPIR-V stages ({string(made_c)} compute){note}`)
return 0
}