Every level of a kit tree or rock carries the same materials, so a GPU-culled layer merges each material's levels into one mesh (layer_arena_build) and scatter_cull.comp writes a record per (material, level) naming that level's index and vertex range; one indirect draw covers them all. A conifer's lit pass and prepass go from 8 draws to 2, its shadow LOD from 6 to 2. Layers that do not fit (card levels, other materials or attributes, 32-bit indices) keep the CPU path. GPU culling is now the Vulkan default (R3D_GPU_CULL=0 turns it off). Camp benchmark, 400 frames: PC 2791 -> 2657 draws, 4.3 -> 4.1 s (both runs); Mac 2791 -> 2657, 8.0/7.4 -> 7.7/7.2 s. Self-tests 59/59 on both machines, PC validation 0 errors; frames within run-to-run noise; OpenGL frames unchanged. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
84 lines
4 KiB
Text
84 lines
4 KiB
Text
// scatter_cull.comp - a scatter layer's per-frame culling on the GPU (phase 38, stage 2).
|
|
//
|
|
// Every instance of the layer (8 floats: x, y, z, scale, sin yaw, cos yaw, seed, wind) is tested
|
|
// against the view frustum and sorted into the layer's LOD buckets by distance - the same rule as
|
|
// layer_partition_lods on the CPU. Bucket k's instances are packed from out[k * cap], and the
|
|
// instanceCount of each indirect draw command for that bucket is written here, so the draws that
|
|
// follow read exactly what survived. Bucket n_lods is the impostor bucket.
|
|
//
|
|
// One invocation walks the list in order: packing needs a running count per bucket, and a walk of
|
|
// tens of thousands of instances is a few hundred microseconds of GPU time - far below the CPU
|
|
// partition and upload it replaces. A parallel prefix-sum version can come later.
|
|
layout(local_size_x = 1) in;
|
|
|
|
layout(set = 0, binding = 0) uniform Params {
|
|
vec4 planes[4]; // the side frustum planes (cam_planes): xyz in, w distance; inside when dot >= -r
|
|
vec4 cam; // xyz the camera, w the cull distance (0 = none)
|
|
vec4 lod_dist; // outer distance of each level (0 = open: runs out to the cull distance)
|
|
uvec4 prims; // the prims each level draws (at most 4)
|
|
uint count; // instances in the layer
|
|
uint cap; // instances each bucket can hold
|
|
uint n_lods; // levels (1 .. 4)
|
|
uint has_imp; // 1 when the layer has an impostor bucket
|
|
float pad; // sphere radius per unit of instance scale (the model's bound)
|
|
float pad_abs; // added to every radius (m)
|
|
} pr;
|
|
|
|
layout(set = 0, binding = 1) readonly buffer Src { float src[]; };
|
|
layout(set = 0, binding = 2) buffer Dst { float dst[]; };
|
|
// VkDrawIndexedIndirectCommand records, 5 uints each; the CPU fills indexCount / firstIndex /
|
|
// vertexOffset / firstInstance once, this writes instanceCount. A prim is a material of the layer's
|
|
// merged meshes (scatter.ludic, layer_arena_build), and one material's records sit together so a
|
|
// single draw covers every level of it:
|
|
// 0 .. 15 prim * 4 + level the lit pass and the depth prepass
|
|
// 16 the impostor card bucket n_lods
|
|
// 17 .. 28 17 + prim * 3 + bucket level 2's range cast from buckets 0 .. 2 (the shadow LOD)
|
|
layout(set = 0, binding = 3) buffer Cmds { uint cmds[]; };
|
|
// how many instances landed in each bucket (levels, then the impostor), for the CPU to read
|
|
layout(set = 0, binding = 4) buffer Counts { uint counts[5]; };
|
|
|
|
void main() {
|
|
uint n = pr.n_lods;
|
|
uint cnt[5] = uint[5](0u, 0u, 0u, 0u, 0u);
|
|
float cull2 = pr.cam.w * pr.cam.w;
|
|
bool open = pr.lod_dist[n - 1u] == 0.0;
|
|
for (uint i = 0u; i < pr.count; i++) {
|
|
uint o = i * 8u;
|
|
vec3 p = vec3(src[o], src[o + 1u], src[o + 2u]);
|
|
float dx = p.x - pr.cam.x;
|
|
float dz = p.z - pr.cam.z;
|
|
float d2 = dx * dx + dz * dz;
|
|
if (pr.cam.w != 0.0 && d2 > cull2) { continue; }
|
|
float r = pr.pad * src[o + 3u] + pr.pad_abs;
|
|
bool inside = true;
|
|
for (int k = 0; k < 4; k++) {
|
|
if (dot(pr.planes[k].xyz, p) + pr.planes[k].w < -r) { inside = false; break; }
|
|
}
|
|
if (!inside) { continue; }
|
|
float d = sqrt(d2);
|
|
uint lv = n;
|
|
for (uint k = 0u; k < n; k++) {
|
|
if (pr.lod_dist[k] != 0.0 && d < pr.lod_dist[k]) { lv = k; break; }
|
|
}
|
|
if (lv == n && open) { lv = n - 1u; }
|
|
if (lv == n && pr.has_imp == 0u) { continue; }
|
|
if (cnt[lv] >= pr.cap) { continue; }
|
|
uint q = (lv * pr.cap + cnt[lv]) * 8u;
|
|
for (uint k = 0u; k < 8u; k++) { dst[q + k] = src[o + k]; }
|
|
cnt[lv] += 1u;
|
|
}
|
|
for (uint k = 0u; k < 4u; k++) {
|
|
for (uint j = 0u; j < 4u; j++) {
|
|
uint c = (j * 4u + k) * 5u;
|
|
cmds[c + 1u] = (k < n && j < pr.prims[k]) ? cnt[k] : 0u;
|
|
}
|
|
}
|
|
cmds[16u * 5u + 1u] = (pr.has_imp != 0u) ? cnt[n] : 0u;
|
|
for (uint b = 0u; b < 3u; b++) {
|
|
for (uint j = 0u; j < 4u; j++) {
|
|
uint c = (17u + j * 3u + b) * 5u;
|
|
cmds[c + 1u] = (n > 2u && b < n && j < pr.prims[2]) ? cnt[b] : 0u;
|
|
}
|
|
}
|
|
for (uint k = 0u; k < 5u; k++) { counts[k] = cnt[k]; }
|
|
}
|