feat(render3d): the meadow's blades are culled on the GPU - under 1.5 ms of a MoltenVK frame for all the grass

- grass_cull.comp decides each candidate blade once a frame (place, ground, density, water,
  slope, frustum, colour field) and writes survivors into three bands by distance, one indirect
  draw each; grass_inst.vert only bends and places the vertices. OpenGL keeps grass.vert's
  per-vertex path; R3D_GRASS_GPU=0 compares
- compute programs take sampled textures after their buffers (gpu_compute_tex / gpu_dispatch_tex),
  and every dispatch now records a compute-to-draw memory barrier
- the blades as a sward: 4 m cells, bands with five, three and one-quad blades, spacing doubling
  every 18 m to 70 m, never narrower than a pixel; lit facing the sun and leaning to the sky,
  shadow looked up above the ground (it read the terrain as its caster), a colour ramp that
  leaves only the sheath dark, clumps, dry patches and a tussock shade worked out per blade
- scatter layers flagged grass are skipped while the blades draw (and under R3D_NOGRASS);
  R3D_BLADES, R3D_NOBLADES win over the game's setting; R3D_GRASS_S0/D0 for measuring

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-27 15:38:43 +03:00
parent f815f3e3cf
commit cc384efee0
34 changed files with 798 additions and 212 deletions

View file

@ -120,24 +120,28 @@ function gvk_program(render3d_st: mut Render3dState, p: int, key: string, spv_di
# ---- compute ------------------------------------------------------------------------------
# A compute program is <name>.comp.spv beside the manifest, with bindings fixed by convention:
# 0 the parameter block (a uniform buffer, copied into the frame's ring at the dispatch) and
# 1 .. n storage buffers. Handles start at 1.
# 0 the parameter block (a uniform buffer, copied into the frame's ring at the dispatch),
# 1 .. n storage buffers, then n+1 .. n+m sampled textures (linear, clamped). Handles start at 1.
function gvk_compute_new(render3d_st: mut Render3dState, name: string, n_bufs: int) -> int {
function gvk_compute_new(render3d_st: mut Render3dState, name: string, n_bufs: int) -> int { return gvk_compute_new_tex(render3d_st, name, n_bufs, 0) }
function gvk_compute_new_tex(render3d_st: mut Render3dState, name: string, n_bufs: int, n_tex: int) -> int {
let zero: long = 0
if render3d_st.gvk_cp_pipe == null {
render3d_st.gvk_cp_pipe = new []long; render3d_st.gvk_cp_layout = new []long; render3d_st.gvk_cp_dsl = new []long; render3d_st.gvk_cp_nbuf = new []int
push(render3d_st.gvk_cp_pipe, zero); push(render3d_st.gvk_cp_layout, zero); push(render3d_st.gvk_cp_dsl, zero); push(render3d_st.gvk_cp_nbuf, 0)
render3d_st.gvk_cp_ntex = new []int
push(render3d_st.gvk_cp_pipe, zero); push(render3d_st.gvk_cp_layout, zero); push(render3d_st.gvk_cp_dsl, zero); push(render3d_st.gvk_cp_nbuf, 0); push(render3d_st.gvk_cp_ntex, 0)
}
let module = gvk_module(render3d_st, `{render3d_st.gvk_spv_dir}/{name}.comp.spv`)
if module == 0 { return 0 }
let nb = n_bufs + 1
let nb = n_bufs + 1 + n_tex
let bw = VkDescriptorSetLayoutBinding_sizeof
let binds = bytes(bw * nb)
Vk.zero(binds, bw * nb)
for k in 0 .. nb {
var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER
if k == 0 { kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER }
if k > n_bufs { kind = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER }
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, k)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, kind)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
@ -172,17 +176,25 @@ function gvk_compute_new(render3d_st: mut Render3dState, name: string, n_bufs: i
r = Vk.create_compute_pipelines(render3d_st.gvk_dev, zero, 1, cpci, null, out)
if r != VK_SUCCESS { gvk_fail(render3d_st, `vkCreateComputePipelines {name}`, r); return 0 }
push(render3d_st.gvk_cp_pipe, gvk_handle(out)); push(render3d_st.gvk_cp_layout, layout); push(render3d_st.gvk_cp_dsl, Vk.get_i64(dsl, 0)); push(render3d_st.gvk_cp_nbuf, n_bufs)
push(render3d_st.gvk_cp_ntex, n_tex)
return len(render3d_st.gvk_cp_pipe) - 1
}
# Record a dispatch of compute program c into the frame, between passes: params (n_params
# bytes, std140) as binding 0 and bufs[0 .. n) as bindings 1 .. n.
function gvk_dispatch(render3d_st: mut Render3dState, c: int, params: pointer, n_params: int, bufs: words, gx: int, gy: int, gz: int) -> void {
function gvk_dispatch(render3d_st: mut Render3dState, c: int, params: pointer, n_params: int, bufs: words, gx: int, gy: int, gz: int) -> void { gvk_dispatch_tex(render3d_st, c, params, n_params, bufs, null, gx, gy, gz) }
# ... and texs[0 .. m) as the sampled textures after the buffers. What it writes is made visible to
# the indirect draws, vertex attributes and shaders that follow (Metal's own hazard tracking was all
# that ordered the tree culling's output before this)
function gvk_dispatch_tex(render3d_st: mut Render3dState, c: int, params: pointer, n_params: int, bufs: words, texs: words, gx: int, gy: int, gz: int) -> void {
let zero: long = 0
if render3d_st.gvk_cp_pipe == null or c <= 0 or c >= len(render3d_st.gvk_cp_pipe) { return }
gvk_pass_end(render3d_st)
let cb = gvk_frame_cb(render3d_st)
let nb = render3d_st.gvk_cp_nbuf[c]
var nt = 0
if render3d_st.gvk_cp_ntex != null and c < len(render3d_st.gvk_cp_ntex) { nt = render3d_st.gvk_cp_ntex[c] }
let dsai = bytes(VkDescriptorSetAllocateInfo_sizeof)
Vk.zero(dsai, VkDescriptorSetAllocateInfo_sizeof)
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO)
@ -199,10 +211,13 @@ function gvk_dispatch(render3d_st: mut Render3dState, c: int, params: pointer, n
if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return }
let ww = VkWriteDescriptorSet_sizeof
let bw = VkDescriptorBufferInfo_sizeof
let writes = bytes(ww * (nb + 1))
Vk.zero(writes, ww * (nb + 1))
let iw = VkDescriptorImageInfo_sizeof
let writes = bytes(ww * (nb + 1 + nt))
Vk.zero(writes, ww * (nb + 1 + nt))
let infos = bytes(bw * (nb + 1))
Vk.zero(infos, bw * (nb + 1))
let iis = bytes(iw * (nt + 1))
Vk.zero(iis, iw * (nt + 1))
for k in 0 .. nb + 1 {
var buf = render3d_st.gvk_ring_buf
var off: long = 0
@ -220,10 +235,31 @@ function gvk_dispatch(render3d_st: mut Render3dState, c: int, params: pointer, n
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorType, kind)
Vk.put_ptr(writes, k * ww + VkWriteDescriptorSet_pBufferInfo, mem_off(infos, k * bw))
}
Vk.update_descriptor_sets(render3d_st.gvk_dev, nb + 1, writes, 0, null)
let smp = gvk_sampler(render3d_st, GL_LINEAR, GL_LINEAR, GL_CLAMP_TO_EDGE, GL_CLAMP_TO_EDGE, 0, 0)
for t in 0 .. nt {
var tex = render3d_st.gvk_white
if texs != null and t < len(texs) and texs[t] > 0 { tex = texs[t] }
Vk.put_i64(iis, t * iw + VkDescriptorImageInfo_sampler, smp)
Vk.put_i64(iis, t * iw + VkDescriptorImageInfo_imageView, render3d_st.gvk_tex_view[tex])
Vk.put_i32(iis, t * iw + VkDescriptorImageInfo_imageLayout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
let k = nb + 1 + t
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET)
Vk.put_i64(writes, k * ww + VkWriteDescriptorSet_dstSet, set)
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_dstBinding, k)
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorCount, 1)
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorType, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
Vk.put_ptr(writes, k * ww + VkWriteDescriptorSet_pImageInfo, mem_off(iis, t * iw))
}
Vk.update_descriptor_sets(render3d_st.gvk_dev, nb + 1 + nt, writes, 0, null)
Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_COMPUTE, render3d_st.gvk_cp_pipe[c])
Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_COMPUTE, render3d_st.gvk_cp_layout[c], 0, 1, sets, 0, null)
Vk.cmd_dispatch(cb, gx, gy, gz)
let mb = bytes(VkMemoryBarrier_sizeof)
Vk.zero(mb, VkMemoryBarrier_sizeof)
Vk.put_i32(mb, VkMemoryBarrier_sType, VK_STRUCTURE_TYPE_MEMORY_BARRIER)
Vk.put_i32(mb, VkMemoryBarrier_srcAccessMask, VK_ACCESS_SHADER_WRITE_BIT)
Vk.put_i32(mb, VkMemoryBarrier_dstAccessMask, VK_ACCESS_INDIRECT_COMMAND_READ_BIT | VK_ACCESS_VERTEX_ATTRIBUTE_READ_BIT | VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT)
Vk.cmd_pipeline_barrier(cb, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, VK_PIPELINE_STAGE_DRAW_INDIRECT_BIT | VK_PIPELINE_STAGE_VERTEX_INPUT_BIT | VK_PIPELINE_STAGE_VERTEX_SHADER_BIT | VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, 1, mb, 0, null, 0, null)
}
# R3D_VK_PROBE=1: one dispatch over a GPU-owned buffer, read back - the compute path end to end