feat(render3d): compute and indirect draws on Vulkan, and tree layers culled on the GPU (opt-in)

- Device: multiDrawIndirect, drawIndirectFirstInstance and drawIndirectCount where present.
- Buffers carry storage and indirect usage; a GPU-owned buffer is never swapped under a draw.
- Compute programs from shaders/compute.list (binding 0 parameters, 1.. storage buffers),
  built by `ludic-dev shaders`; gpu_compute / gpu_dispatch / gpu_draw_mesh_indirect in gpu.ludic.
- R3D_VK_PROBE=1: a dispatch read back (OK on the RTX 3070 Ti).
- scatter_cull.comp: a tree layer's frustum test and LOD split on the GPU, with the lit, prepass,
  impostor and shadow-LOD draws reading its records. Behind R3D_GPU_CULL=1 and off by default:
  at the camp it is slower (43.0 fps against 53.3), because the frame's cost is per-draw
  descriptor sets and it adds empty-level draws. Validation-clean; OpenGL frames unchanged.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-15 14:19:43 +03:00
parent 04cda22d19
commit c1eb5f399f
11 changed files with 463 additions and 15 deletions

View file

@ -61,6 +61,15 @@ property Layer {
# crossed card carrying its atlas (cover keeps its baked card as the far level).
lods: []Model,
n_lods: int = 0,
# The GPU-culled path (Vulkan, phase 38): every instance in g_src, a compute pass packs the
# visible ones per bucket into g_dst and writes the instance counts of the draw records in
# g_cmds (see scatter_cull.comp for the record layout). g_on once it is set up for g_n instances.
g_on: bool = false,
g_n: int = 0,
g_src: int = 0,
g_dst: int = 0,
g_cmds: int = 0,
g_counts: int = 0,
lod_dist: words,
lod_card: words,
lod_buf: words,
@ -525,6 +534,88 @@ function scatter_begin_frame() -> void {
v3_copy(sc_view_pos, cam_pos)
v3_copy(sc_view_fwd, cam_fwd)
}
# GPU-culled layers dispatch before the first pass of the frame, so no pass is split for it
if sc_layers != null {
for i in 0 .. len(sc_layers) { if sc_layers[i].n_lods > 1 and sc_layers[i].imp != null { layer_update(sc_layers[i]) } }
}
}
# ---- the GPU-culled path ------------------------------------------------------------------
# R3D_GPU_CULL=1 on Vulkan: a tree layer (a LOD chain of up to four levels of up to four prims,
# with an impostor, not streamed) is culled and split into its buckets by scatter_cull.comp, and
# its lit, prepass, impostor and shadow-LOD draws read the records that pass wrote. Nothing is
# partitioned or uploaded on the CPU when the view moves.
const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand
const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp)
var sc_cull_prog: int = 0
var sc_cull_tried: bool = false
var sc_ind_base: int = -1 # >= 0: layer_draw_model / layer_draw_depth draw records from here
function layer_gpu_eligible(l: Layer) -> bool {
if l.n_lods < 2 or l.n_lods > 4 or l.imp == null or l.streamed or l.flower or l.blade or l.count == 0 { return false }
for k in 0 .. l.n_lods { if len(l.lods[k].prims) > 4 { return false } }
return true
}
function layer_gpu_prepare(l: Layer) -> bool {
if not sc_cull_tried {
sc_cull_tried = true
# Os.env is null when the variable is unset, and a compare reads through it: ask first
if gpu_has_compute() and Os.has_env("R3D_GPU_CULL") and Os.env("R3D_GPU_CULL") == "1" {
sc_cull_prog = gpu_compute("scatter_cull", 4)
if sc_cull_prog > 0 { print("r3d: scatter: tree layers are culled on the GPU") }
}
}
if sc_cull_prog == 0 or not layer_gpu_eligible(l) { return false }
if l.g_on and l.g_n == l.count { return true }
if l.g_src == 0 {
l.g_src = gpu_buffer_new(); l.g_dst = gpu_buffer_new(); l.g_cmds = gpu_buffer_new(); l.g_counts = gpu_buffer_new()
gpu_buffer_gpu_owned(l.g_dst); gpu_buffer_gpu_owned(l.g_cmds); gpu_buffer_gpu_owned(l.g_counts)
}
let n = l.n_lods
let cap = l.count
gpu_buffer_upload(l.g_src, cap * INST_FLOATS * 4, l.inst, GPU_STATIC)
gpu_buffer_upload(l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC)
let rec = words(SC_RECS * 5)
for i in 0 .. SC_RECS * 5 { rec[i] = 0 }
for k in 0 .. n {
let m = l.lods[k]
for j in 0 .. len(m.prims) { rec[(k * 4 + j) * 5] = m.prims[j].mesh.count; rec[(k * 4 + j) * 5 + 4] = k * cap }
}
rec[16 * 5] = sc_card.count; rec[16 * 5 + 4] = n * cap
if n > 2 {
let m2 = l.lods[2]
for b in 0 .. 3 {
for j in 0 .. len(m2.prims) { rec[(17 + b * 4 + j) * 5] = m2.prims[j].mesh.count; rec[(17 + b * 4 + j) * 5 + 4] = b * cap }
}
}
gpu_buffer_upload(l.g_cmds, SC_RECS * SC_REC_W, rec, GPU_DYNAMIC)
let zeros = words(5)
for i in 0 .. 5 { zeros[i] = 0 }
gpu_buffer_upload(l.g_counts, 20, zeros, GPU_DYNAMIC)
free(rec); free(zeros)
# the card casts every instance, as on the CPU path (layer_grid_build uploads this there)
l.n_sh = l.count
gpu_buffer_upload(l.sh_buf, l.count * INST_FLOATS * 4, l.inst, GPU_STATIC)
l.g_n = l.count
l.g_on = true
return true
}
# the dispatch for the view as it stands: frustum, camera, distances, the layer's shape
function layer_gpu_cull(l: Layer) -> void {
let pr = words(36)
for i in 0 .. 36 { pr[i] = 0 }
if cam_planes != null { for i in 0 .. 16 { pr[i] = cam_planes[i] } }
pr[16] = cam_pos[0]; pr[17] = cam_pos[1]; pr[18] = cam_pos[2]; pr[19] = l.cull
for k in 0 .. l.n_lods { pr[20 + k] = l.lod_dist[k]; pr[24 + k] = len(l.lods[k].prims) }
pr[28] = l.count; pr[29] = l.count; pr[30] = l.n_lods; pr[31] = 1
# as layer_grid_gather pads a cell: the tallest instance, plus a margin
pr[32] = f_mul(l.lods[0].height, F_TWO); pr[33] = fi(4)
let bufs = words(4)
bufs[0] = l.g_src; bufs[1] = l.g_dst; bufs[2] = l.g_cmds; bufs[3] = l.g_counts
gpu_dispatch(sc_cull_prog, pr, 144, bufs, 1)
free(pr); free(bufs)
}
# Sort a static layer's instances into square cells (call once, after placement; a
@ -667,6 +758,7 @@ function layer_update(l: Layer) -> void {
if sc_freeze { return }
if l.view_gen == sc_view_gen { return }
l.view_gen = sc_view_gen
if layer_gpu_prepare(l) { layer_gpu_cull(l); return }
let t_lu = gl_now_us()
let n_lu = l.count
# A streamed layer's instances were already gathered per visible chunk: no split, no
@ -764,6 +856,11 @@ var sc_dbg_tint: words = null
# no cheap stand-in to cast from, so without this its shadow simply began at the near
# distance — which is the crown shadow that appeared as you walked up to a tree.
function layer_draw_near(l: Layer, shadow: bool, light_vp: words, full: bool) -> void {
if l.n_lods > 1 and l.g_on and not shadow {
for k in 0 .. l.n_lods { sc_dbg_level = k; sc_ind_base = k * 4; layer_draw_model(l, l.lods[k], l.g_dst, 1, l.lod_card[k] == 1, shadow, light_vp) }
sc_ind_base = -1; sc_dbg_level = -1
return
}
if l.n_lods > 1 {
# a LOD chain: every level from its own bucket (casters are what is drawn)
for k in 0 .. l.n_lods { sc_dbg_level = k; layer_draw_model(l, l.lods[k], l.lod_buf[k], l.n_lod[k], l.lod_card[k] == 1, shadow, light_vp) }
@ -829,14 +926,15 @@ function layer_draw_model(l: Layer, model: Model, vb: int, cnt: int, card: bool,
r3d_bind_2d(p, "u_diff", 0, pr.diff)
if not shadow { r3d_bind_2d(p, "u_nrm", 1, pr.nrm); r3d_bind_2d(p, "u_arm", 2, pr.arm) }
}
mesh_draw_instanced(pr.mesh, cnt)
if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i) * SC_REC_W, 1, 0, 0) }
else { mesh_draw_instanced(pr.mesh, cnt) }
}
gpu_alpha_to_coverage(false)
if p == sc_prog_fol_eq { gpu_depth_func(GL_LESS) }
}
function layer_draw_far(l: Layer, shadow: bool, light_vp: words) -> void {
if l.n_far == 0 or l.imp == null { return }
if l.imp == null or (l.n_far == 0 and not l.g_on) { return }
var p = sc_imp_prog
if shadow { p = sc_imp_prog_shadow }
gpu_use_program(p)
@ -863,8 +961,13 @@ function layer_draw_far(l: Layer, shadow: bool, light_vp: words) -> void {
}
gpu_cull(false)
if not shadow and sc_a2c { gpu_alpha_to_coverage(true) }
scatter_attach(sc_card, l.imp_buf)
mesh_draw_instanced(sc_card, l.n_far)
if l.g_on {
scatter_attach(sc_card, l.g_dst)
gpu_draw_mesh_indirect(sc_card, l.g_cmds, 16 * SC_REC_W, 1, 0, 0)
} else {
scatter_attach(sc_card, l.imp_buf)
mesh_draw_instanced(sc_card, l.n_far)
}
gpu_alpha_to_coverage(false)
}
@ -915,7 +1018,8 @@ function layer_draw_depth(l: Layer, model: Model, vb: int, cnt: int) -> void {
let pr = model.prims[i]
scatter_attach(pr.mesh, vb)
r3d_bind_2d(p, "u_diff", 0, pr.diff)
mesh_draw_instanced(pr.mesh, cnt)
if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i) * SC_REC_W, 1, 0, 0) }
else { mesh_draw_instanced(pr.mesh, cnt) }
}
}
@ -928,7 +1032,10 @@ function scatter_draw_depth() -> void {
if not l.foliage or l.blade or l.flower { continue }
if r3d_no_trees and l.imp != null { continue }
layer_update(l)
if l.n_lods > 1 {
if l.n_lods > 1 and l.g_on {
for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { sc_ind_base = k * 4; layer_draw_depth(l, l.lods[k], l.g_dst, 1) } }
sc_ind_base = -1
} else if l.n_lods > 1 {
# (a tree layer is flagged `card` for its distant level; its mesh levels still count)
for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { layer_draw_depth(l, l.lods[k], l.lod_buf[k], l.n_lod[k]) } }
} else if not l.card {
@ -969,7 +1076,10 @@ function scatter_draw_casters(light_vp: words) -> void {
# crown of drooping needle cards is mostly slivers from the side, so the sun, which
# sees the crown from above, cast a trunk line with a few blobs. The near levels
# now also cast their LOD2 mesh, alpha-tested, on top of the card.
if l.n_lods > 2 {
if l.n_lods > 2 and l.g_on {
for k in 0 .. 3 { sc_ind_base = 17 + k * 4; layer_draw_model(l, l.lods[2], l.g_dst, 1, false, true, light_vp) }
sc_ind_base = -1
} else if l.n_lods > 2 {
for k in 0 .. 3 { layer_draw_model(l, l.lods[2], l.lod_buf[k], l.n_lod[k], false, true, light_vp) }
}
}