diff --git a/packages/ludic.render3d/gltf.ludic b/packages/ludic.render3d/gltf.ludic index 219d2c2d..81b0f867 100644 --- a/packages/ludic.render3d/gltf.ludic +++ b/packages/ludic.render3d/gltf.ludic @@ -10,6 +10,7 @@ property Prim { diff: int = 0, nrm: int = 0, arm: int = 0, + verts: int = 0, # vertices in the mesh (each attribute buffer holds this many) name: string # the material's name (a part to tint: "hk_jacket") } property Model { @@ -134,6 +135,7 @@ function gltf_prim(p: Val) -> Prim { let m = gpu_mesh_new() let attrs = value_get(p, "attributes") gltf_attrib(m, attrs, "POSITION", 0) + let nverts = gltf_count gltf_attrib(m, attrs, "NORMAL", 1) gltf_attrib(m, attrs, "TEXCOORD_0", 2) skin_attribs(m, attrs) # JOINTS_0 / WEIGHTS_0 onto 5 / 6, when the mesh has them @@ -145,6 +147,7 @@ function gltf_prim(p: Val) -> Prim { m.count = gltf_count gpu_mesh_done(m) pr.mesh = m + pr.verts = nverts # material textures if gltf_white == 0 { gltf_white = tex_solid(200, 200, 200, 255); gltf_flat = tex_solid(128, 128, 255, 255) } pr.diff = gltf_white; pr.nrm = gltf_flat; pr.arm = gltf_white diff --git a/packages/ludic.render3d/scatter.ludic b/packages/ludic.render3d/scatter.ludic index 81dc205a..53e55161 100644 --- a/packages/ludic.render3d/scatter.ludic +++ b/packages/ludic.render3d/scatter.ludic @@ -70,6 +70,11 @@ property Layer { g_dst: int = 0, g_cmds: int = 0, g_counts: int = 0, + g_arena: []Prim, # one merged mesh per material, holding every level's copy (layer_arena_build) + g_first: words, # material * 4 + level: that level's first index in the merged mesh + g_base: words, # ... and its first vertex + g_model: Model, # the merged meshes as a model the draws take (level 0's height) + g_model_sh: Model, # the same with level 2's height, for the shadow LOD lod_dist: words, lod_card: words, lod_buf: words, @@ -541,27 +546,108 @@ function scatter_begin_frame() -> void { } # ---- the GPU-culled path ------------------------------------------------------------------ -# R3D_GPU_CULL=1 on Vulkan: a tree layer (a LOD chain of up to four levels of up to four prims, -# with an impostor, not streamed) is culled and split into its buckets by scatter_cull.comp, and -# its lit, prepass, impostor and shadow-LOD draws read the records that pass wrote. Nothing is -# partitioned or uploaded on the CPU when the view moves. +# On Vulkan (R3D_GPU_CULL=0 turns it off): a tree layer (a LOD chain of up to four levels sharing up +# to four materials, with an impostor, not streamed) is culled and split into its buckets by +# scatter_cull.comp, and its lit, prepass, impostor and shadow-LOD draws read the records that pass +# wrote - one draw per material covering every level. Nothing is partitioned or uploaded on the CPU +# when the view moves. PC camp benchmark: 2791 -> 2657 draws, 4.3 -> 4.1 s for 400 frames. const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp) var sc_cull_prog: int = 0 var sc_cull_tried: bool = false var sc_ind_base: int = -1 # >= 0: layer_draw_model / layer_draw_depth draw records from here +var sc_ind_n: int = 1 # records each of those draws covers: one per level of a material +var sc_ind_stride: int = 4 # records between one material's first and the next's (3 for the shadow LOD) function layer_gpu_eligible(l: Layer) -> bool { if l.n_lods < 2 or l.n_lods > 4 or l.imp == null or l.streamed or l.flower or l.blade or l.count == 0 { return false } - for k in 0 .. l.n_lods { if len(l.lods[k].prims) > 4 { return false } } + return layer_arena_ok(l) +} + +# The merged meshes. Every level of a kit tree or rock carries the same materials in the same order +# (bark then needles; the rock's one), so each material becomes ONE mesh holding all its levels, and +# one indirect draw of several records draws every level of it: record (material, level) names that +# level's index and vertex range and its bucket's instances. A conifer's lit pass goes from eight +# draws to two. Anything that does not fit - a card level, a level with other materials, other +# attributes or 32-bit indices - keeps the CPU path. +function layer_arena_ok(l: Layer) -> bool { + let n_mat = len(l.lods[0].prims) + if n_mat == 0 or n_mat > 4 { return false } + for k in 0 .. l.n_lods { + if l.lod_card[k] == 1 { return false } + let m = l.lods[k] + if len(m.prims) != n_mat { return false } + for j in 0 .. n_mat { + let pm = m.prims[j].mesh + let p0 = l.lods[0].prims[j] + if m.prims[j].diff != p0.diff or m.prims[j].verts == 0 or pm.ebo == 0 or pm.itype != GL_UNSIGNED_SHORT { return false } + if gpu_buffer_map(pm.ebo) == null { return false } + for a in 0 .. 3 { + let o = a * GPU_ATTR_W + if p0.mesh.attrs[o + 1] == 0 or pm.attrs[o + 1] != p0.mesh.attrs[o + 1] or pm.attrs[o + 3] != pm.attrs[o + 1] * 4 { return false } + if gpu_buffer_map(pm.attrs[o]) == null { return false } + } + } + } return true } +function layer_arena_build(l: Layer) -> void { + let n_mat = len(l.lods[0].prims) + l.g_arena = new []Prim + l.g_first = words(16); l.g_base = words(16) + for i in 0 .. 16 { l.g_first[i] = 0; l.g_base[i] = 0 } + for j in 0 .. n_mat { + var nv = 0 + var ni = 0 + for k in 0 .. l.n_lods { + let pr = l.lods[k].prims[j] + l.g_first[j * 4 + k] = ni; l.g_base[j * 4 + k] = nv + nv += pr.verts; ni += pr.mesh.count + } + let p0 = l.lods[0].prims[j] + let m = gpu_mesh_new() + for a in 0 .. 3 { + let comps = p0.mesh.attrs[a * GPU_ATTR_W + 1] + let vb = bytes(nv * comps * 4 + 8) + for k in 0 .. l.n_lods { + let pr = l.lods[k].prims[j] + mem_copy(mem_off(vb, l.g_base[j * 4 + k] * comps * 4), gpu_buffer_map(pr.mesh.attrs[a * GPU_ATTR_W]), pr.verts * comps * 4) + } + gpu_mesh_vertices(m, vb, nv * comps * 4, GPU_STATIC) + gpu_mesh_attr(m, a, comps, GPU_F32, 0, 0, false) + free(vb) + } + let ib = bytes(ni * 2 + 8) + for k in 0 .. l.n_lods { + let pm = l.lods[k].prims[j].mesh + mem_copy(mem_off(ib, l.g_first[j * 4 + k] * 2), gpu_buffer_map(pm.ebo), pm.count * 2) + } + gpu_mesh_indices(m, ib, ni * 2, 2) + free(ib) + m.count = ni + gpu_mesh_done(m) + scatter_attach(m, l.g_dst) + let ap = new Prim + ap.mesh = m; ap.diff = p0.diff; ap.nrm = p0.nrm; ap.arm = p0.arm; ap.verts = nv; ap.name = p0.name + push(l.g_arena, ap) + } + l.g_model = new Model + l.g_model.prims = l.g_arena; l.g_model.height = l.lods[0].height; l.g_model.radius = l.lods[0].radius; l.g_model.ymin = l.lods[0].ymin + l.g_model_sh = new Model + var sh = l.n_lods - 1 + if sh > 2 { sh = 2 } + l.g_model_sh.prims = l.g_arena; l.g_model_sh.height = l.lods[sh].height; l.g_model_sh.radius = l.lods[sh].radius; l.g_model_sh.ymin = l.lods[sh].ymin +} + function layer_gpu_prepare(l: Layer) -> bool { if not sc_cull_tried { sc_cull_tried = true # Os.env is null when the variable is unset, and a compare reads through it: ask first - if gpu_has_compute() and Os.has_env("R3D_GPU_CULL") and Os.env("R3D_GPU_CULL") == "1" { + # on by default wherever there is compute; R3D_GPU_CULL=0 keeps the CPU partition (for comparing) + var off = false + if Os.has_env("R3D_GPU_CULL") { off = Os.env("R3D_GPU_CULL") == "0" } + if gpu_has_compute() and not off { sc_cull_prog = gpu_compute("scatter_cull", 4) if sc_cull_prog > 0 { print("r3d: scatter: tree layers are culled on the GPU") } } @@ -576,17 +662,25 @@ function layer_gpu_prepare(l: Layer) -> bool { let cap = l.count gpu_buffer_upload(l.g_src, cap * INST_FLOATS * 4, l.inst, GPU_STATIC) gpu_buffer_upload(l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC) + if l.g_arena == null { layer_arena_build(l) } + let n_mat = len(l.g_arena) let rec = words(SC_RECS * 5) for i in 0 .. SC_RECS * 5 { rec[i] = 0 } - for k in 0 .. n { - let m = l.lods[k] - for j in 0 .. len(m.prims) { rec[(k * 4 + j) * 5] = m.prims[j].mesh.count; rec[(k * 4 + j) * 5 + 4] = k * cap } + # material j, level k: that level's range of the merged mesh, its instances from bucket k + for j in 0 .. n_mat { + for k in 0 .. n { + let r = (j * 4 + k) * 5 + rec[r] = l.lods[k].prims[j].mesh.count; rec[r + 2] = l.g_first[j * 4 + k]; rec[r + 3] = l.g_base[j * 4 + k]; rec[r + 4] = k * cap + } } rec[16 * 5] = sc_card.count; rec[16 * 5 + 4] = n * cap + # the shadow LOD: level 2's range, from buckets 0 .. 2 if n > 2 { - let m2 = l.lods[2] - for b in 0 .. 3 { - for j in 0 .. len(m2.prims) { rec[(17 + b * 4 + j) * 5] = m2.prims[j].mesh.count; rec[(17 + b * 4 + j) * 5 + 4] = b * cap } + for j in 0 .. n_mat { + for b in 0 .. 3 { + let r = (17 + j * 3 + b) * 5 + rec[r] = l.lods[2].prims[j].mesh.count; rec[r + 2] = l.g_first[j * 4 + 2]; rec[r + 3] = l.g_base[j * 4 + 2]; rec[r + 4] = b * cap + } } } gpu_buffer_upload(l.g_cmds, SC_RECS * SC_REC_W, rec, GPU_DYNAMIC) @@ -857,8 +951,10 @@ var sc_dbg_tint: words = null # distance — which is the crown shadow that appeared as you walked up to a tree. function layer_draw_near(l: Layer, shadow: bool, light_vp: words, full: bool) -> void { if l.n_lods > 1 and l.g_on and not shadow { - for k in 0 .. l.n_lods { sc_dbg_level = k; sc_ind_base = k * 4; layer_draw_model(l, l.lods[k], l.g_dst, 1, l.lod_card[k] == 1, shadow, light_vp) } - sc_ind_base = -1; sc_dbg_level = -1 + # one draw per material covering all of its levels (the merged meshes) + sc_ind_base = 0; sc_ind_n = l.n_lods + layer_draw_model(l, l.g_model, l.g_dst, 1, false, shadow, light_vp) + sc_ind_base = -1; sc_ind_n = 1 return } if l.n_lods > 1 { @@ -926,7 +1022,7 @@ function layer_draw_model(l: Layer, model: Model, vb: int, cnt: int, card: bool, r3d_bind_2d(p, "u_diff", 0, pr.diff) if not shadow { r3d_bind_2d(p, "u_nrm", 1, pr.nrm); r3d_bind_2d(p, "u_arm", 2, pr.arm) } } - if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i) * SC_REC_W, 1, 0, 0) } + if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i * sc_ind_stride) * SC_REC_W, sc_ind_n, 0, 0) } else { mesh_draw_instanced(pr.mesh, cnt) } } gpu_alpha_to_coverage(false) @@ -1018,7 +1114,7 @@ function layer_draw_depth(l: Layer, model: Model, vb: int, cnt: int) -> void { let pr = model.prims[i] scatter_attach(pr.mesh, vb) r3d_bind_2d(p, "u_diff", 0, pr.diff) - if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i) * SC_REC_W, 1, 0, 0) } + if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i * sc_ind_stride) * SC_REC_W, sc_ind_n, 0, 0) } else { mesh_draw_instanced(pr.mesh, cnt) } } } @@ -1033,8 +1129,9 @@ function scatter_draw_depth() -> void { if r3d_no_trees and l.imp != null { continue } layer_update(l) if l.n_lods > 1 and l.g_on { - for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { sc_ind_base = k * 4; layer_draw_depth(l, l.lods[k], l.g_dst, 1) } } - sc_ind_base = -1 + sc_ind_base = 0; sc_ind_n = l.n_lods + layer_draw_depth(l, l.g_model, l.g_dst, 1) + sc_ind_base = -1; sc_ind_n = 1 } else if l.n_lods > 1 { # (a tree layer is flagged `card` for its distant level; its mesh levels still count) for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { layer_draw_depth(l, l.lods[k], l.lod_buf[k], l.n_lod[k]) } } @@ -1077,8 +1174,9 @@ function scatter_draw_casters(light_vp: words) -> void { # sees the crown from above, cast a trunk line with a few blobs. The near levels # now also cast their LOD2 mesh, alpha-tested, on top of the card. if l.n_lods > 2 and l.g_on { - for k in 0 .. 3 { sc_ind_base = 17 + k * 4; layer_draw_model(l, l.lods[2], l.g_dst, 1, false, true, light_vp) } - sc_ind_base = -1 + sc_ind_base = 17; sc_ind_n = 3; sc_ind_stride = 3 + layer_draw_model(l, l.g_model_sh, l.g_dst, 1, false, true, light_vp) + sc_ind_base = -1; sc_ind_n = 1; sc_ind_stride = 4 } else if l.n_lods > 2 { for k in 0 .. 3 { layer_draw_model(l, l.lods[2], l.lod_buf[k], l.n_lod[k], false, true, light_vp) } } @@ -1203,6 +1301,7 @@ function scatter_clear_all() -> void { if l.vis != null { free(l.vis) } if l.lod_dist != null { free(l.lod_dist); free(l.lod_card); free(l.lod_buf); free(l.n_lod) } if l.lvl != null { free(l.lvl) } + if l.g_arena != null { for k in 0 .. len(l.g_arena) { mesh_free(l.g_arena[k].mesh) } } # layer_cards built this layer's crossed card and its atlas itself if l.card and l.n_lods == 0 and l.model != null { for k in 0 .. len(l.model.prims) { mesh_free(l.model.prims[k].mesh) } } if l.card and l.atlas != null { diff --git a/packages/ludic.render3d/shaders/scatter_cull.comp b/packages/ludic.render3d/shaders/scatter_cull.comp index 2d12cfa0..78b0cf10 100644 --- a/packages/ludic.render3d/shaders/scatter_cull.comp +++ b/packages/ludic.render3d/shaders/scatter_cull.comp @@ -27,10 +27,12 @@ layout(set = 0, binding = 0) uniform Params { layout(set = 0, binding = 1) readonly buffer Src { float src[]; }; layout(set = 0, binding = 2) buffer Dst { float dst[]; }; // VkDrawIndexedIndirectCommand records, 5 uints each; the CPU fills indexCount / firstIndex / -// vertexOffset / firstInstance once, this writes instanceCount: -// 0 .. 15 level * 4 + prim the lit pass and the depth prepass -// 16 the impostor card bucket n_lods -// 17 .. 28 17 + bucket * 4 + prim level 2's mesh cast from buckets 0 .. 2 (the shadow LOD) +// vertexOffset / firstInstance once, this writes instanceCount. A prim is a material of the layer's +// merged meshes (scatter.ludic, layer_arena_build), and one material's records sit together so a +// single draw covers every level of it: +// 0 .. 15 prim * 4 + level the lit pass and the depth prepass +// 16 the impostor card bucket n_lods +// 17 .. 28 17 + prim * 3 + bucket level 2's range cast from buckets 0 .. 2 (the shadow LOD) layout(set = 0, binding = 3) buffer Cmds { uint cmds[]; }; // how many instances landed in each bucket (levels, then the impostor), for the CPU to read layout(set = 0, binding = 4) buffer Counts { uint counts[5]; }; @@ -67,14 +69,14 @@ void main() { } for (uint k = 0u; k < 4u; k++) { for (uint j = 0u; j < 4u; j++) { - uint c = (k * 4u + j) * 5u; + uint c = (j * 4u + k) * 5u; cmds[c + 1u] = (k < n && j < pr.prims[k]) ? cnt[k] : 0u; } } cmds[16u * 5u + 1u] = (pr.has_imp != 0u) ? cnt[n] : 0u; for (uint b = 0u; b < 3u; b++) { for (uint j = 0u; j < 4u; j++) { - uint c = (17u + b * 4u + j) * 5u; + uint c = (17u + j * 3u + b) * 5u; cmds[c + 1u] = (n > 2u && b < n && j < pr.prims[2]) ? cnt[b] : 0u; } } diff --git a/packages/ludic.render3d/shaders/spv/scatter_cull.comp.spv b/packages/ludic.render3d/shaders/spv/scatter_cull.comp.spv index 3d742bc0..07f965a3 100644 Binary files a/packages/ludic.render3d/shaders/spv/scatter_cull.comp.spv and b/packages/ludic.render3d/shaders/spv/scatter_cull.comp.spv differ