From c1eb5f399f4081b1f5995edbfc3b9a20a4ba05a0 Mon Sep 17 00:00:00 2001 From: Orkuncakilkaya Date: Tue, 15 Sep 2026 14:19:43 +0300 Subject: [PATCH] feat(render3d): compute and indirect draws on Vulkan, and tree layers culled on the GPU (opt-in) - Device: multiDrawIndirect, drawIndirectFirstInstance and drawIndirectCount where present. - Buffers carry storage and indirect usage; a GPU-owned buffer is never swapped under a draw. - Compute programs from shaders/compute.list (binding 0 parameters, 1.. storage buffers), built by `ludic-dev shaders`; gpu_compute / gpu_dispatch / gpu_draw_mesh_indirect in gpu.ludic. - R3D_VK_PROBE=1: a dispatch read back (OK on the RTX 3070 Ti). - scatter_cull.comp: a tree layer's frustum test and LOD split on the GPU, with the lit, prepass, impostor and shadow-LOD draws reading its records. Behind R3D_GPU_CULL=1 and off by default: at the camp it is slower (43.0 fps against 53.3), because the frame's cost is per-draw descriptor sets and it adds empty-level draws. Validation-clean; OpenGL frames unchanged. Co-Authored-By: Claude Opus 5 --- packages/ludic.render3d/gpu.ludic | 28 +++ packages/ludic.render3d/gpu_vk.ludic | 15 +- packages/ludic.render3d/gpu_vk_draw.ludic | 173 +++++++++++++++++- packages/ludic.render3d/gpu_vk_res.ludic | 14 +- packages/ludic.render3d/scatter.ludic | 124 ++++++++++++- packages/ludic.render3d/shaders/compute.list | 4 + packages/ludic.render3d/shaders/probe.comp | 10 + .../ludic.render3d/shaders/scatter_cull.comp | 82 +++++++++ .../ludic.render3d/shaders/spv/probe.comp.spv | Bin 0 -> 1268 bytes .../shaders/spv/scatter_cull.comp.spv | Bin 0 -> 9540 bytes tools/ludic-cli/shaders.ludic | 28 ++- 11 files changed, 463 insertions(+), 15 deletions(-) create mode 100644 packages/ludic.render3d/shaders/compute.list create mode 100644 packages/ludic.render3d/shaders/probe.comp create mode 100644 packages/ludic.render3d/shaders/scatter_cull.comp create mode 100644 packages/ludic.render3d/shaders/spv/probe.comp.spv create mode 100644 packages/ludic.render3d/shaders/spv/scatter_cull.comp.spv diff --git a/packages/ludic.render3d/gpu.ludic b/packages/ludic.render3d/gpu.ludic index 03eff7da..dc350696 100644 --- a/packages/ludic.render3d/gpu.ludic +++ b/packages/ludic.render3d/gpu.ludic @@ -509,6 +509,34 @@ function gpu_buffer_free(buf: int) -> void { gl_delete_buffers(1, ids) } +# ---- compute and indirect draws (Vulkan) ---------------------------------------------------- +# The GPU-driven path: a compute program writes instance lists and draw commands into buffers +# the draws then read. OpenGL here is 4.1 (macOS) with no compute, so gpu_compute is 0 there and +# a caller keeps its CPU path. A compute program's binding 0 is its parameter block (params, +# copied at the dispatch); bindings 1.. are `bufs`, gpu buffers. +function gpu_has_compute() -> bool { return gpu_kind == GPU_VK } +function gpu_compute(name: string, n_bufs: int) -> int { + if gpu_kind != GPU_VK { return 0 } + return gvk_compute_new(name, n_bufs) +} +function gpu_dispatch(c: int, params: pointer, n_params: int, bufs: words, groups: int) -> void { + if gpu_kind == GPU_VK and c > 0 { gvk_dispatch(c, params, n_params, bufs, groups, 1, 1) } +} +# a buffer a compute pass writes (never reallocated under a draw that reads it) +function gpu_buffer_gpu_owned(buf: int) -> void { if gpu_kind == GPU_VK { gvk_buf_gpu_owned(buf) } } +# the host-visible contents of a buffer, for a readback after gpu_finish; null on OpenGL +function gpu_buffer_map(buf: int) -> pointer { + if gpu_kind != GPU_VK or buf <= 0 { return null } + return gvk_buf_map[buf] +} +function gpu_finish() -> void { if gpu_kind == GPU_VK { gvk_flush() } } +# n indexed draws of mesh m from VkDrawIndexedIndirectCommand records in buffer cmds at offset +# (bytes); each record's firstInstance selects its instances out of the bound instance buffer. +# With count_buf > 0 the GPU's own count (a uint at count_off) is used, up to n. +function gpu_draw_mesh_indirect(m: Mesh, cmds: int, offset: int, n: int, count_buf: int, count_off: int) -> void { + if gpu_kind == GPU_VK { gvk_draw_indirect_now(m, cmds, offset, n, count_buf, count_off) } +} + # drawing function gpu_mesh_bind(m: Mesh) -> void { if gpu_kind == GPU_VK { return }; gl_bind_vertex_array(m.vao) } function gpu_mesh_unbind() -> void { if gpu_kind == GPU_VK { return }; gl_bind_vertex_array(0) } diff --git a/packages/ludic.render3d/gpu_vk.ludic b/packages/ludic.render3d/gpu_vk.ludic index cd146bfa..045769b7 100644 --- a/packages/ludic.render3d/gpu_vk.ludic +++ b/packages/ludic.render3d/gpu_vk.ludic @@ -37,6 +37,9 @@ function gvk_ext_in(props: bytes, n: int, want: string) -> bool { # Instance, the first discrete GPU (else the first listed), its first graphics queue, and a # device with the Tier 1 floor switched on. False, with gvk_why set, if any of it is missing: # the caller stays on OpenGL. +var gvk_has_mdi: bool = false # multiDrawIndirect + drawIndirectFirstInstance +var gvk_has_dic: bool = false # drawIndirectCount + function gvk_init() -> bool { if gvk_ready { return true } if Vk.open() == 0 { gvk_why = "no Vulkan loader"; return false } @@ -145,6 +148,16 @@ function gvk_init() -> bool { Vk.put_i32(want2, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2) Vk.put_ptr(want2, VkPhysicalDeviceFeatures2_pNext, want12) if aniso { Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_samplerAnisotropy, 1) } + # GPU-driven drawing: one indirect buffer holds a layer's draws, and a compute pass may write + # how many of them there are. Asked for where the device has them; the renderer checks + # gvk_has_mdi / gvk_has_dic before it takes that path. + gvk_has_mdi = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect) == 1 and Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance) == 1 + if gvk_has_mdi { + Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect, 1) + Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance, 1) + } + gvk_has_dic = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_drawIndirectCount) == 1 + if gvk_has_dic { Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_drawIndirectCount, 1) } Vk.put_i32(cnt, 0, 0) Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, null) @@ -186,7 +199,7 @@ function gvk_init() -> bool { if not gvk_cmd_init() { return false } gvk_ready = true - print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x`) + print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x, multi-draw indirect {gvk_has_mdi}, indirect count {gvk_has_dic}`) return true } diff --git a/packages/ludic.render3d/gpu_vk_draw.ludic b/packages/ludic.render3d/gpu_vk_draw.ludic index 7df701d2..40372449 100644 --- a/packages/ludic.render3d/gpu_vk_draw.ludic +++ b/packages/ludic.render3d/gpu_vk_draw.ludic @@ -118,6 +118,141 @@ function gvk_program(p: int, key: string, spv_dir: string) -> bool { return true } +# ---- compute ------------------------------------------------------------------------------ +# A compute program is .comp.spv beside the manifest, with bindings fixed by convention: +# 0 the parameter block (a uniform buffer, copied into the frame's ring at the dispatch) and +# 1 .. n storage buffers. Handles start at 1. +var gvk_cp_pipe: []long = null +var gvk_cp_layout: []long = null +var gvk_cp_dsl: []long = null +var gvk_cp_nbuf: []int = null + +function gvk_compute_new(name: string, n_bufs: int) -> int { + let zero: long = 0 + if gvk_cp_pipe == null { + gvk_cp_pipe = new []long; gvk_cp_layout = new []long; gvk_cp_dsl = new []long; gvk_cp_nbuf = new []int + push(gvk_cp_pipe, zero); push(gvk_cp_layout, zero); push(gvk_cp_dsl, zero); push(gvk_cp_nbuf, 0) + } + let module = gvk_module(`{gvk_spv_dir}/{name}.comp.spv`) + if module == 0 { return 0 } + let nb = n_bufs + 1 + let bw = VkDescriptorSetLayoutBinding_sizeof + let binds = bytes(bw * nb) + Vk.zero(binds, bw * nb) + for k in 0 .. nb { + var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER + if k == 0 { kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER } + Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, k) + Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, kind) + Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1) + Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_COMPUTE_BIT) + } + let dslci = bytes(VkDescriptorSetLayoutCreateInfo_sizeof) + Vk.zero(dslci, VkDescriptorSetLayoutCreateInfo_sizeof) + Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO) + Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_bindingCount, nb) + Vk.put_ptr(dslci, VkDescriptorSetLayoutCreateInfo_pBindings, binds) + let dsl = bytes(8) + var r = Vk.create_descriptor_set_layout(gvk_dev, dslci, null, dsl) + if r != VK_SUCCESS { gvk_fail(`vkCreateDescriptorSetLayout for compute {name}`, r); return 0 } + let plci = bytes(VkPipelineLayoutCreateInfo_sizeof) + Vk.zero(plci, VkPipelineLayoutCreateInfo_sizeof) + Vk.put_i32(plci, VkPipelineLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO) + Vk.put_i32(plci, VkPipelineLayoutCreateInfo_setLayoutCount, 1) + Vk.put_ptr(plci, VkPipelineLayoutCreateInfo_pSetLayouts, dsl) + let out = bytes(8) + r = Vk.create_pipeline_layout(gvk_dev, plci, null, out) + if r != VK_SUCCESS { gvk_fail(`vkCreatePipelineLayout for compute {name}`, r); return 0 } + let layout = gvk_handle(out) + let cpci = bytes(VkComputePipelineCreateInfo_sizeof) + Vk.zero(cpci, VkComputePipelineCreateInfo_sizeof) + Vk.put_i32(cpci, VkComputePipelineCreateInfo_sType, VK_STRUCTURE_TYPE_COMPUTE_PIPELINE_CREATE_INFO) + let so = VkComputePipelineCreateInfo_stage + Vk.put_i32(cpci, so + VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO) + Vk.put_i32(cpci, so + VkPipelineShaderStageCreateInfo_stage, VK_SHADER_STAGE_COMPUTE_BIT) + Vk.put_i64(cpci, so + VkPipelineShaderStageCreateInfo_module, module) + Vk.put_ptr(cpci, so + VkPipelineShaderStageCreateInfo_pName, "main") + Vk.put_i64(cpci, VkComputePipelineCreateInfo_layout, layout) + r = Vk.create_compute_pipelines(gvk_dev, zero, 1, cpci, null, out) + if r != VK_SUCCESS { gvk_fail(`vkCreateComputePipelines {name}`, r); return 0 } + push(gvk_cp_pipe, gvk_handle(out)); push(gvk_cp_layout, layout); push(gvk_cp_dsl, Vk.get_i64(dsl, 0)); push(gvk_cp_nbuf, n_bufs) + return len(gvk_cp_pipe) - 1 +} + +# Record a dispatch of compute program c into the frame, between passes: params (n_params +# bytes, std140) as binding 0 and bufs[0 .. n) as bindings 1 .. n. +function gvk_dispatch(c: int, params: pointer, n_params: int, bufs: words, gx: int, gy: int, gz: int) -> void { + let zero: long = 0 + if gvk_cp_pipe == null or c <= 0 or c >= len(gvk_cp_pipe) { return } + gvk_pass_end() + let cb = gvk_frame_cb() + let nb = gvk_cp_nbuf[c] + let dsai = bytes(VkDescriptorSetAllocateInfo_sizeof) + Vk.zero(dsai, VkDescriptorSetAllocateInfo_sizeof) + Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO) + Vk.put_i64(dsai, VkDescriptorSetAllocateInfo_descriptorPool, gvk_dpool) + Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_descriptorSetCount, 1) + let layouts = bytes(8) + Vk.put_i64(layouts, 0, gvk_cp_dsl[c]) + Vk.put_ptr(dsai, VkDescriptorSetAllocateInfo_pSetLayouts, layouts) + let sets = bytes(8) + let r = Vk.allocate_descriptor_sets(gvk_dev, dsai, sets) + if r != VK_SUCCESS { gvk_fail("vkAllocateDescriptorSets (compute)", r); return } + let set = Vk.get_i64(sets, 0) + let at = gvk_ring_put(params, n_params) + if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return } + let ww = VkWriteDescriptorSet_sizeof + let bw = VkDescriptorBufferInfo_sizeof + let writes = bytes(ww * (nb + 1)) + Vk.zero(writes, ww * (nb + 1)) + let infos = bytes(bw * (nb + 1)) + Vk.zero(infos, bw * (nb + 1)) + for k in 0 .. nb + 1 { + var buf = gvk_ring_buf + var off: long = 0 + var range: long = 0 + var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER + if k == 0 { off = at; range = n_params; kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER } + else { buf = bufs[k - 1]; range = gvk_buf_size[buf]; gvk_buf_used[buf] = gvk_frame_no } + Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_buffer, gvk_buf[buf]) + Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_offset, off) + Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_range, range) + Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET) + Vk.put_i64(writes, k * ww + VkWriteDescriptorSet_dstSet, set) + Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_dstBinding, k) + Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorCount, 1) + Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorType, kind) + Vk.put_ptr(writes, k * ww + VkWriteDescriptorSet_pBufferInfo, mem_off(infos, k * bw)) + } + Vk.update_descriptor_sets(gvk_dev, nb + 1, writes, 0, null) + Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_COMPUTE, gvk_cp_pipe[c]) + Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_COMPUTE, gvk_cp_layout[c], 0, 1, sets, 0, null) + Vk.cmd_dispatch(cb, gx, gy, gz) +} + +# R3D_VK_PROBE=1: one dispatch over a GPU-owned buffer, read back - the compute path end to end +function gvk_compute_probe() -> void { + let c = gvk_compute_new("probe", 1) + if c == 0 { print("r3d: vulkan compute probe: FAILED (no program)"); return } + let n = 1000 + let b = gvk_buf_new() + let data = bytes(n * 4) + for i in 0 .. n { Vk.put_i32(data, i * 4, fi(i)) } + gvk_buf_upload(b, n * 4, data) + gvk_buf_gpu_owned(b) + let pr = bytes(8) + Vk.put_i32(pr, 0, n) + Vk.put_i32(pr, 4, fi(3)) + let bufs = words(1) + bufs[0] = b + gvk_dispatch(c, pr, 8, bufs, (n + 63) / 64, 1, 1) + gvk_flush() + var bad = 0 + let mp = gvk_buf_map[b] + for i in 0 .. n { if Vk.get_i32(mp, i * 4) != fi(i * 3) { bad += 1 } } + if bad == 0 { print(`r3d: vulkan compute probe OK ({n} values)`) } else { print(`r3d: vulkan compute probe: FAILED ({bad} of {n} wrong)`) } +} + # ---- pipelines ---------------------------------------------------------------------------- # A pipeline is everything OpenGL decides at the draw: the program, the vertex layout the mesh # recorded, the render state the renderer set, and the formats and sample count of the pass it @@ -480,16 +615,18 @@ function gvk_frame_init() -> bool { if gvk_ring_align < 16 { gvk_ring_align = 16 } gvk_ring_buf = gvk_buf_new() if not gvk_buf_reserve(gvk_ring_buf, GVK_RING_BYTES) { return false } - let sizes = bytes(VkDescriptorPoolSize_sizeof * 2) + let sizes = bytes(VkDescriptorPoolSize_sizeof * 3) Vk.put_i32(sizes, VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER) Vk.put_i32(sizes, VkDescriptorPoolSize_descriptorCount, 32768) Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER) Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_descriptorCount, 131072) + Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof * 2 + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_STORAGE_BUFFER) + Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof * 2 + VkDescriptorPoolSize_descriptorCount, 16384) let dpci = bytes(VkDescriptorPoolCreateInfo_sizeof) Vk.zero(dpci, VkDescriptorPoolCreateInfo_sizeof) Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO) Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_maxSets, 16384) - Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_poolSizeCount, 2) + Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_poolSizeCount, 3) Vk.put_ptr(dpci, VkDescriptorPoolCreateInfo_pPoolSizes, sizes) let out = bytes(8) let r = Vk.create_descriptor_pool(gvk_dev, dpci, null, out) @@ -507,7 +644,7 @@ function gvk_frame_reset() -> void { } # a block into the ring; its offset, or -1 when the frame has used the whole ring -function gvk_ring_put(blk: bytes, n: int) -> int { +function gvk_ring_put(blk: pointer, n: int) -> int { let at = (gvk_ring_off + gvk_ring_align - 1) / gvk_ring_align * gvk_ring_align if at + n > GVK_RING_BYTES { return -1 } mem_copy(mem_off(gvk_buf_map[gvk_ring_buf], at), blk, n) @@ -914,7 +1051,20 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc if m.itype == GL_UNSIGNED_SHORT { itype = VK_INDEX_TYPE_UINT16 } Vk.cmd_bind_index_buffer(cb, gvk_buf[m.ebo], zero, itype) gvk_buf_used[m.ebo] = gvk_frame_no - Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0) + if gvk_ind_buf > 0 { + # the draws are records in a buffer (gvk_draw_indirect_now); the GPU may have written them + let ioff: long = gvk_ind_off + gvk_buf_used[gvk_ind_buf] = gvk_frame_no + if gvk_ind_cbuf > 0 and gvk_has_dic { + let coff: long = gvk_ind_coff + gvk_buf_used[gvk_ind_cbuf] = gvk_frame_no + Vk.cmd_draw_indexed_indirect_count(cb, gvk_buf[gvk_ind_buf], ioff, gvk_buf[gvk_ind_cbuf], coff, gvk_ind_n, VkDrawIndexedIndirectCommand_sizeof) + } else { + Vk.cmd_draw_indexed_indirect(cb, gvk_buf[gvk_ind_buf], ioff, gvk_ind_n, VkDrawIndexedIndirectCommand_sizeof) + } + } else { + Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0) + } } else { Vk.cmd_draw(cb, n, instances, first, 0) } @@ -1002,6 +1152,7 @@ function gvk_open(w: int, h: int, title: string) -> bool { } if not gvk_frame_init() { return false } if not gvk_screen_make(gl_w, gl_h) { return false } + if Os.has_env("R3D_VK_PROBE") { gvk_compute_probe() } if is_windowed() { return gvk_swap_make(gl_w, gl_h) } return true } @@ -1073,6 +1224,20 @@ function gvk_state_now() -> GvkState { st.wireframe = gvk_wireframe return st } +# An indirect draw goes through gvk_draw like any other - the same pipeline, set and buffers - +# and only its last call differs, so the record is handed over in these for that one draw. +var gvk_ind_buf: int = 0 +var gvk_ind_off: int = 0 +var gvk_ind_n: int = 0 +var gvk_ind_cbuf: int = 0 +var gvk_ind_coff: int = 0 +function gvk_draw_indirect_now(m: Mesh, cmds: int, offset: int, n: int, count_buf: int, count_off: int) -> void { + if m == null or m.ebo == 0 or cmds <= 0 or n <= 0 { return } + gvk_ind_buf = cmds; gvk_ind_off = offset; gvk_ind_n = n; gvk_ind_cbuf = count_buf; gvk_ind_coff = count_off + gvk_draw_now(m, 0, 0, 1) + gvk_ind_buf = 0; gvk_ind_cbuf = 0 +} + function gvk_draw_now(m: Mesh, first: int, count: int, instances: int) -> void { gvk_draw(gpu_prog_cur, m, gvk_state_now(), first, count, instances, gpu_tx, GPU_TX_W, gpu_tx_cap, gpu_fb, gpu_fb_at(gvk_fb_cur)) } diff --git a/packages/ludic.render3d/gpu_vk_res.ludic b/packages/ludic.render3d/gpu_vk_res.ludic index 44422113..870e9a8f 100644 --- a/packages/ludic.render3d/gpu_vk_res.ludic +++ b/packages/ludic.render3d/gpu_vk_res.ludic @@ -561,12 +561,22 @@ function gvk_retire_flush() -> void { gvk_retired_mem = new []long } +# A buffer a compute pass writes: a draw later in the same frame reads what the GPU put there, +# so it is never swapped for fresh storage because it was used this frame (gvk_buf_reserve). +var gvk_buf_gpu: []int = null +function gvk_buf_gpu_owned(b: int) -> void { + if gvk_buf_gpu == null { gvk_buf_gpu = new []int } + while len(gvk_buf_gpu) <= b { push(gvk_buf_gpu, 0) } + gvk_buf_gpu[b] = 1 +} +function gvk_buf_is_gpu(b: int) -> bool { return gvk_buf_gpu != null and b < len(gvk_buf_gpu) and gvk_buf_gpu[b] == 1 } + # Room for at least n bytes behind handle b; a buffer that is already big enough is kept, so a # stream re-filled every frame allocates once. function gvk_buf_reserve(b: int, n: int) -> bool { if b <= 0 or b >= len(gvk_buf) { return false } # already big enough, and no draw this frame reads what is there: fill it in place - if gvk_buf[b] != 0 and gvk_buf_size[b] >= n and gvk_buf_used[b] != gvk_frame_no { return true } + if gvk_buf[b] != 0 and gvk_buf_size[b] >= n and (gvk_buf_used[b] != gvk_frame_no or gvk_buf_is_gpu(b)) { return true } gvk_buf_release(b) var size = n if size < 64 { size = 64 } @@ -575,7 +585,7 @@ function gvk_buf_reserve(b: int, n: int) -> bool { Vk.zero(bci, VkBufferCreateInfo_sizeof) Vk.put_i32(bci, VkBufferCreateInfo_sType, VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO) Vk.put_i64(bci, VkBufferCreateInfo_size, size_l) - Vk.put_i32(bci, VkBufferCreateInfo_usage, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT) + Vk.put_i32(bci, VkBufferCreateInfo_usage, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT) Vk.put_i32(bci, VkBufferCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE) let out = bytes(8) var r = Vk.create_buffer(gvk_dev, bci, null, out) diff --git a/packages/ludic.render3d/scatter.ludic b/packages/ludic.render3d/scatter.ludic index 4bf79fad..81dc205a 100644 --- a/packages/ludic.render3d/scatter.ludic +++ b/packages/ludic.render3d/scatter.ludic @@ -61,6 +61,15 @@ property Layer { # crossed card carrying its atlas (cover keeps its baked card as the far level). lods: []Model, n_lods: int = 0, + # The GPU-culled path (Vulkan, phase 38): every instance in g_src, a compute pass packs the + # visible ones per bucket into g_dst and writes the instance counts of the draw records in + # g_cmds (see scatter_cull.comp for the record layout). g_on once it is set up for g_n instances. + g_on: bool = false, + g_n: int = 0, + g_src: int = 0, + g_dst: int = 0, + g_cmds: int = 0, + g_counts: int = 0, lod_dist: words, lod_card: words, lod_buf: words, @@ -525,6 +534,88 @@ function scatter_begin_frame() -> void { v3_copy(sc_view_pos, cam_pos) v3_copy(sc_view_fwd, cam_fwd) } + # GPU-culled layers dispatch before the first pass of the frame, so no pass is split for it + if sc_layers != null { + for i in 0 .. len(sc_layers) { if sc_layers[i].n_lods > 1 and sc_layers[i].imp != null { layer_update(sc_layers[i]) } } + } +} + +# ---- the GPU-culled path ------------------------------------------------------------------ +# R3D_GPU_CULL=1 on Vulkan: a tree layer (a LOD chain of up to four levels of up to four prims, +# with an impostor, not streamed) is culled and split into its buckets by scatter_cull.comp, and +# its lit, prepass, impostor and shadow-LOD draws read the records that pass wrote. Nothing is +# partitioned or uploaded on the CPU when the view moves. +const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand +const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp) +var sc_cull_prog: int = 0 +var sc_cull_tried: bool = false +var sc_ind_base: int = -1 # >= 0: layer_draw_model / layer_draw_depth draw records from here + +function layer_gpu_eligible(l: Layer) -> bool { + if l.n_lods < 2 or l.n_lods > 4 or l.imp == null or l.streamed or l.flower or l.blade or l.count == 0 { return false } + for k in 0 .. l.n_lods { if len(l.lods[k].prims) > 4 { return false } } + return true +} + +function layer_gpu_prepare(l: Layer) -> bool { + if not sc_cull_tried { + sc_cull_tried = true + # Os.env is null when the variable is unset, and a compare reads through it: ask first + if gpu_has_compute() and Os.has_env("R3D_GPU_CULL") and Os.env("R3D_GPU_CULL") == "1" { + sc_cull_prog = gpu_compute("scatter_cull", 4) + if sc_cull_prog > 0 { print("r3d: scatter: tree layers are culled on the GPU") } + } + } + if sc_cull_prog == 0 or not layer_gpu_eligible(l) { return false } + if l.g_on and l.g_n == l.count { return true } + if l.g_src == 0 { + l.g_src = gpu_buffer_new(); l.g_dst = gpu_buffer_new(); l.g_cmds = gpu_buffer_new(); l.g_counts = gpu_buffer_new() + gpu_buffer_gpu_owned(l.g_dst); gpu_buffer_gpu_owned(l.g_cmds); gpu_buffer_gpu_owned(l.g_counts) + } + let n = l.n_lods + let cap = l.count + gpu_buffer_upload(l.g_src, cap * INST_FLOATS * 4, l.inst, GPU_STATIC) + gpu_buffer_upload(l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC) + let rec = words(SC_RECS * 5) + for i in 0 .. SC_RECS * 5 { rec[i] = 0 } + for k in 0 .. n { + let m = l.lods[k] + for j in 0 .. len(m.prims) { rec[(k * 4 + j) * 5] = m.prims[j].mesh.count; rec[(k * 4 + j) * 5 + 4] = k * cap } + } + rec[16 * 5] = sc_card.count; rec[16 * 5 + 4] = n * cap + if n > 2 { + let m2 = l.lods[2] + for b in 0 .. 3 { + for j in 0 .. len(m2.prims) { rec[(17 + b * 4 + j) * 5] = m2.prims[j].mesh.count; rec[(17 + b * 4 + j) * 5 + 4] = b * cap } + } + } + gpu_buffer_upload(l.g_cmds, SC_RECS * SC_REC_W, rec, GPU_DYNAMIC) + let zeros = words(5) + for i in 0 .. 5 { zeros[i] = 0 } + gpu_buffer_upload(l.g_counts, 20, zeros, GPU_DYNAMIC) + free(rec); free(zeros) + # the card casts every instance, as on the CPU path (layer_grid_build uploads this there) + l.n_sh = l.count + gpu_buffer_upload(l.sh_buf, l.count * INST_FLOATS * 4, l.inst, GPU_STATIC) + l.g_n = l.count + l.g_on = true + return true +} + +# the dispatch for the view as it stands: frustum, camera, distances, the layer's shape +function layer_gpu_cull(l: Layer) -> void { + let pr = words(36) + for i in 0 .. 36 { pr[i] = 0 } + if cam_planes != null { for i in 0 .. 16 { pr[i] = cam_planes[i] } } + pr[16] = cam_pos[0]; pr[17] = cam_pos[1]; pr[18] = cam_pos[2]; pr[19] = l.cull + for k in 0 .. l.n_lods { pr[20 + k] = l.lod_dist[k]; pr[24 + k] = len(l.lods[k].prims) } + pr[28] = l.count; pr[29] = l.count; pr[30] = l.n_lods; pr[31] = 1 + # as layer_grid_gather pads a cell: the tallest instance, plus a margin + pr[32] = f_mul(l.lods[0].height, F_TWO); pr[33] = fi(4) + let bufs = words(4) + bufs[0] = l.g_src; bufs[1] = l.g_dst; bufs[2] = l.g_cmds; bufs[3] = l.g_counts + gpu_dispatch(sc_cull_prog, pr, 144, bufs, 1) + free(pr); free(bufs) } # Sort a static layer's instances into square cells (call once, after placement; a @@ -667,6 +758,7 @@ function layer_update(l: Layer) -> void { if sc_freeze { return } if l.view_gen == sc_view_gen { return } l.view_gen = sc_view_gen + if layer_gpu_prepare(l) { layer_gpu_cull(l); return } let t_lu = gl_now_us() let n_lu = l.count # A streamed layer's instances were already gathered per visible chunk: no split, no @@ -764,6 +856,11 @@ var sc_dbg_tint: words = null # no cheap stand-in to cast from, so without this its shadow simply began at the near # distance — which is the crown shadow that appeared as you walked up to a tree. function layer_draw_near(l: Layer, shadow: bool, light_vp: words, full: bool) -> void { + if l.n_lods > 1 and l.g_on and not shadow { + for k in 0 .. l.n_lods { sc_dbg_level = k; sc_ind_base = k * 4; layer_draw_model(l, l.lods[k], l.g_dst, 1, l.lod_card[k] == 1, shadow, light_vp) } + sc_ind_base = -1; sc_dbg_level = -1 + return + } if l.n_lods > 1 { # a LOD chain: every level from its own bucket (casters are what is drawn) for k in 0 .. l.n_lods { sc_dbg_level = k; layer_draw_model(l, l.lods[k], l.lod_buf[k], l.n_lod[k], l.lod_card[k] == 1, shadow, light_vp) } @@ -829,14 +926,15 @@ function layer_draw_model(l: Layer, model: Model, vb: int, cnt: int, card: bool, r3d_bind_2d(p, "u_diff", 0, pr.diff) if not shadow { r3d_bind_2d(p, "u_nrm", 1, pr.nrm); r3d_bind_2d(p, "u_arm", 2, pr.arm) } } - mesh_draw_instanced(pr.mesh, cnt) + if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i) * SC_REC_W, 1, 0, 0) } + else { mesh_draw_instanced(pr.mesh, cnt) } } gpu_alpha_to_coverage(false) if p == sc_prog_fol_eq { gpu_depth_func(GL_LESS) } } function layer_draw_far(l: Layer, shadow: bool, light_vp: words) -> void { - if l.n_far == 0 or l.imp == null { return } + if l.imp == null or (l.n_far == 0 and not l.g_on) { return } var p = sc_imp_prog if shadow { p = sc_imp_prog_shadow } gpu_use_program(p) @@ -863,8 +961,13 @@ function layer_draw_far(l: Layer, shadow: bool, light_vp: words) -> void { } gpu_cull(false) if not shadow and sc_a2c { gpu_alpha_to_coverage(true) } - scatter_attach(sc_card, l.imp_buf) - mesh_draw_instanced(sc_card, l.n_far) + if l.g_on { + scatter_attach(sc_card, l.g_dst) + gpu_draw_mesh_indirect(sc_card, l.g_cmds, 16 * SC_REC_W, 1, 0, 0) + } else { + scatter_attach(sc_card, l.imp_buf) + mesh_draw_instanced(sc_card, l.n_far) + } gpu_alpha_to_coverage(false) } @@ -915,7 +1018,8 @@ function layer_draw_depth(l: Layer, model: Model, vb: int, cnt: int) -> void { let pr = model.prims[i] scatter_attach(pr.mesh, vb) r3d_bind_2d(p, "u_diff", 0, pr.diff) - mesh_draw_instanced(pr.mesh, cnt) + if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i) * SC_REC_W, 1, 0, 0) } + else { mesh_draw_instanced(pr.mesh, cnt) } } } @@ -928,7 +1032,10 @@ function scatter_draw_depth() -> void { if not l.foliage or l.blade or l.flower { continue } if r3d_no_trees and l.imp != null { continue } layer_update(l) - if l.n_lods > 1 { + if l.n_lods > 1 and l.g_on { + for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { sc_ind_base = k * 4; layer_draw_depth(l, l.lods[k], l.g_dst, 1) } } + sc_ind_base = -1 + } else if l.n_lods > 1 { # (a tree layer is flagged `card` for its distant level; its mesh levels still count) for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { layer_draw_depth(l, l.lods[k], l.lod_buf[k], l.n_lod[k]) } } } else if not l.card { @@ -969,7 +1076,10 @@ function scatter_draw_casters(light_vp: words) -> void { # crown of drooping needle cards is mostly slivers from the side, so the sun, which # sees the crown from above, cast a trunk line with a few blobs. The near levels # now also cast their LOD2 mesh, alpha-tested, on top of the card. - if l.n_lods > 2 { + if l.n_lods > 2 and l.g_on { + for k in 0 .. 3 { sc_ind_base = 17 + k * 4; layer_draw_model(l, l.lods[2], l.g_dst, 1, false, true, light_vp) } + sc_ind_base = -1 + } else if l.n_lods > 2 { for k in 0 .. 3 { layer_draw_model(l, l.lods[2], l.lod_buf[k], l.n_lod[k], false, true, light_vp) } } } diff --git a/packages/ludic.render3d/shaders/compute.list b/packages/ludic.render3d/shaders/compute.list new file mode 100644 index 00000000..10bd7bd4 --- /dev/null +++ b/packages/ludic.render3d/shaders/compute.list @@ -0,0 +1,4 @@ +# compute.list - render3d's compute programs: name|file.comp. Built by `ludic-dev shaders` into +# spv/.comp.spv. Binding 0 is the uniform block of parameters, 1.. storage buffers. +probe|probe.comp +scatter_cull|scatter_cull.comp diff --git a/packages/ludic.render3d/shaders/probe.comp b/packages/ludic.render3d/shaders/probe.comp new file mode 100644 index 00000000..d78066cc --- /dev/null +++ b/packages/ludic.render3d/shaders/probe.comp @@ -0,0 +1,10 @@ +// probe.comp - the compute path's own check (R3D_VK_PROBE=1): every float in the buffer times +// the parameter, so a readback proves the dispatch, the bindings and the GPU-owned buffer. +layout(local_size_x = 64) in; +layout(set = 0, binding = 0) uniform Params { uint count; float mul; } pr; +layout(set = 0, binding = 1) buffer Data { float v[]; } data; +void main() { + uint i = gl_GlobalInvocationID.x; + if (i >= pr.count) { return; } + data.v[i] = data.v[i] * pr.mul; +} diff --git a/packages/ludic.render3d/shaders/scatter_cull.comp b/packages/ludic.render3d/shaders/scatter_cull.comp new file mode 100644 index 00000000..2d12cfa0 --- /dev/null +++ b/packages/ludic.render3d/shaders/scatter_cull.comp @@ -0,0 +1,82 @@ +// scatter_cull.comp - a scatter layer's per-frame culling on the GPU (phase 38, stage 2). +// +// Every instance of the layer (8 floats: x, y, z, scale, sin yaw, cos yaw, seed, wind) is tested +// against the view frustum and sorted into the layer's LOD buckets by distance - the same rule as +// layer_partition_lods on the CPU. Bucket k's instances are packed from out[k * cap], and the +// instanceCount of each indirect draw command for that bucket is written here, so the draws that +// follow read exactly what survived. Bucket n_lods is the impostor bucket. +// +// One invocation walks the list in order: packing needs a running count per bucket, and a walk of +// tens of thousands of instances is a few hundred microseconds of GPU time - far below the CPU +// partition and upload it replaces. A parallel prefix-sum version can come later. +layout(local_size_x = 1) in; + +layout(set = 0, binding = 0) uniform Params { + vec4 planes[4]; // the side frustum planes (cam_planes): xyz in, w distance; inside when dot >= -r + vec4 cam; // xyz the camera, w the cull distance (0 = none) + vec4 lod_dist; // outer distance of each level (0 = open: runs out to the cull distance) + uvec4 prims; // the prims each level draws (at most 4) + uint count; // instances in the layer + uint cap; // instances each bucket can hold + uint n_lods; // levels (1 .. 4) + uint has_imp; // 1 when the layer has an impostor bucket + float pad; // sphere radius per unit of instance scale (the model's bound) + float pad_abs; // added to every radius (m) +} pr; + +layout(set = 0, binding = 1) readonly buffer Src { float src[]; }; +layout(set = 0, binding = 2) buffer Dst { float dst[]; }; +// VkDrawIndexedIndirectCommand records, 5 uints each; the CPU fills indexCount / firstIndex / +// vertexOffset / firstInstance once, this writes instanceCount: +// 0 .. 15 level * 4 + prim the lit pass and the depth prepass +// 16 the impostor card bucket n_lods +// 17 .. 28 17 + bucket * 4 + prim level 2's mesh cast from buckets 0 .. 2 (the shadow LOD) +layout(set = 0, binding = 3) buffer Cmds { uint cmds[]; }; +// how many instances landed in each bucket (levels, then the impostor), for the CPU to read +layout(set = 0, binding = 4) buffer Counts { uint counts[5]; }; + +void main() { + uint n = pr.n_lods; + uint cnt[5] = uint[5](0u, 0u, 0u, 0u, 0u); + float cull2 = pr.cam.w * pr.cam.w; + bool open = pr.lod_dist[n - 1u] == 0.0; + for (uint i = 0u; i < pr.count; i++) { + uint o = i * 8u; + vec3 p = vec3(src[o], src[o + 1u], src[o + 2u]); + float dx = p.x - pr.cam.x; + float dz = p.z - pr.cam.z; + float d2 = dx * dx + dz * dz; + if (pr.cam.w != 0.0 && d2 > cull2) { continue; } + float r = pr.pad * src[o + 3u] + pr.pad_abs; + bool inside = true; + for (int k = 0; k < 4; k++) { + if (dot(pr.planes[k].xyz, p) + pr.planes[k].w < -r) { inside = false; break; } + } + if (!inside) { continue; } + float d = sqrt(d2); + uint lv = n; + for (uint k = 0u; k < n; k++) { + if (pr.lod_dist[k] != 0.0 && d < pr.lod_dist[k]) { lv = k; break; } + } + if (lv == n && open) { lv = n - 1u; } + if (lv == n && pr.has_imp == 0u) { continue; } + if (cnt[lv] >= pr.cap) { continue; } + uint q = (lv * pr.cap + cnt[lv]) * 8u; + for (uint k = 0u; k < 8u; k++) { dst[q + k] = src[o + k]; } + cnt[lv] += 1u; + } + for (uint k = 0u; k < 4u; k++) { + for (uint j = 0u; j < 4u; j++) { + uint c = (k * 4u + j) * 5u; + cmds[c + 1u] = (k < n && j < pr.prims[k]) ? cnt[k] : 0u; + } + } + cmds[16u * 5u + 1u] = (pr.has_imp != 0u) ? cnt[n] : 0u; + for (uint b = 0u; b < 3u; b++) { + for (uint j = 0u; j < 4u; j++) { + uint c = (17u + b * 4u + j) * 5u; + cmds[c + 1u] = (n > 2u && b < n && j < pr.prims[2]) ? cnt[b] : 0u; + } + } + for (uint k = 0u; k < 5u; k++) { counts[k] = cnt[k]; } +} diff --git a/packages/ludic.render3d/shaders/spv/probe.comp.spv b/packages/ludic.render3d/shaders/spv/probe.comp.spv new file mode 100644 index 0000000000000000000000000000000000000000..511ceb3fa7047f572d4e96e36f8d2c79a8411c31 GIT binary patch literal 1268 zcmYk5>q;A85Qe{G6SdWPs)wG`jc2Q-{n3^}sVE4k;EzHtKvo3_*7evmJ`%*;FQH#0lCNvdn3AymS67z-amab`jlCcus4wsrdJw0S$YY<~KjOD02- z3)M`asbg!Po3(o;#=$ha+zdDbYDL1YihU%gqZqzdCpSi}4eEH-x$2y?I{ovkbJDx- zUu1)Jzjtz+=WAgW`!l=Ay0;;p!q{`szv~UKP9SIB?p=qPNtnaGzA1XA_{Z5G!{l0H z&pmLqRqSP%ZxK6>cn4b4vbN9i19j}ZM(l~4z2$sPYuLX5JTbP9$9M%h0t$T+Qhc{< zJ$+B?q29aJtAH2}*ZTnP%&);qrgh#16|Jd1PYwH}z>^a5LFABMBz}g~_m27JSo_I; zD`R`e?Gme<03P)(u;%0!sNy$~f@N~SvEMd+rHt*f4-SXoCcgPSrf^5T z12yvE3-8kUzGIJj{R`v&$Qj(f{{!t2toxe;?%2J4M`Pb9Ab&sw^~NzbjW6eYck&9j zM|Tx9eiQP)(YU9#zazk4}n)&I{^>dVENew+4<@p>5_F|oVc1n$>4 t%&Bv?Tfm(2s@nn1yA7h|eaSzdDc(nMPd?V)!;kfy#ofs50_Xh-o&ZLRJ~aRU literal 0 HcmV?d00001 diff --git a/packages/ludic.render3d/shaders/spv/scatter_cull.comp.spv b/packages/ludic.render3d/shaders/spv/scatter_cull.comp.spv new file mode 100644 index 0000000000000000000000000000000000000000..3d742bc005f484c9c23d3e6408b515cf3027584d GIT binary patch literal 9540 zcmZvh2ar|u6~^DP%K~CU6dS@KN)bdsup%x+78M1>-j>HM>guwrEZECZ?7b&(l9-GM zwir!JG^S2^G0oITVq!9;C#G5xll*?~-4E~V0K+RkY+XYFsmww1o!PHa!t(oyZ*64Q*`(@w?G9Bf$SylP*2 z*DB-cdCuNj?c9<#tZuhgSM*V{kv*}$^&2a@f;;HyI`+YU`)}Dl3-Mg~3kH${N z?yC73E3@?#cpkd9zzfjJ)^yZn5&QIEo4>Gb*6>N-&P(buoFTrfj;mq@Fy8%P%-Ssd z=&qi=uFjQp-$v{UYQDzGgXqozZ$@`tRF8QWoS2uu7Zx!um-yGf3pgJ4PVCc(ZT~mw zW({x6;ilG(;PderGaB4c!%da(;Q6aOo%*?I`*akX@p~7X_&D$i{Nj_rsd+ZIUbC^X z0bK9fRJjQ}-#b#AzqU{9-oRPsmV#?zJzL}R&UoH!_`Flz$wfqpea6?-7S43JRwQxe z$So~#jRm*7#5EP1cOkjnVROCL;hag1 z-K^R3!S-`D_h13K5m^Ky_Spq-w%FG9G2T0?-`Zk(xks_fyd-744fZduP5>uHzk-H2VT=lw~&+@*3o@6tGG)%UKc zcDKs)?!`YGzu%CNtipASqoOs%JCSL>7R09o+jCarM-l0Hwjev>+X}61zbQqo^|eRy zU+=Rqa>j3i&U(h+^-d?oxcH5USij!0iN6>1In6tadfMKn%`H5)jXs6C(u(6qI3OIGm-w*Lg{K4SzInBZLef3;g(dJ~1xnTRXB74Bj zN9!L0UVzqit|QUTAm@2#FD$t9ZNv7-oPLusx98(`#yJx2_rzY4>E(LYBlg;WcBc9n zH&t#1TgQ8|PZ4`RICY)}8z1`xu;=W0eDAX!-+66&ypFb}oa@o{9hWmN@3;Hl`>bu= zUT9*zRxax!A`Sb~Cyi+qHQQ^WJ+8?d7{JXRn*l)|IoDw(q^1XL%3U zdiI(M()PXgenbvrCye(#_!#>-jQe9v_sug=Yx^Gf&4}&2_P0muiQq{kTYmZgdv?j5 z3!Yc9t>^EH)YJAiM%k_ou>GSQ|3q+q$=(3oShB76V97Rb^8owd0rpD+?3V}F{=P_m zwZAdSw!bsV_RInHkppahTa@GdeNndkjZwDO4Y2*)p?@x?=yz^M#BZbDzr#60@5D}s z{2Wf)^VJr6wtmOv64inphnVZP%I}l;zAt+s`Vu!D{Qu(ig4cHqvzWIxVy?XXygPE4 zYag)lY#`n_ry_FR*Yw&KY_G|P{>(HDtnVpuC!qI7vY>+|~#)w%k4F2a_}9vuZJMT~@wlVUFSqZjI>~jh`Ip<;2r=MS}yD9 z0%tvA!Kr;dSl`psnSj0kk@G!EOgGqiu~&h8*PktNdcelWZ!WmiU~B1njd?Ca`}nTt z>qX>zSH!-XzNgyqo>dEa9pc<;k@UZ);I*5<7bE4ndP%{#&aD4Z@Iu6QNPlWw2G*Co zy&PU&(b?<=@xihX$pwt4d2zjvX1+#`MOMC9Bfan9*}uzR2VGe+BUF;<;5y&LSB z{N8#t??L3oAlB0-S3Ya&nm+;kUZfv+sIVUZ%Q^4+(LT=m0Q!B1oHK~6nfFCL`|<&B z71^3IcAq|oG$8J;KIeK6EpJXU`XOW}qHl9yiysE>hS=*7w7u2d)yL2uLVWZ;j@GYE z-4BDU5&H?SeeyT-Nw6{Uu2cI_#M;&pC-);@Yvq0ZDA@Zm8PR9lF!WQ1zI@L^`OcC&1?Uj8rN89r zA>NJWk@Wh0!Kd~Qu;uc8rH(rH;)h`SIJ5EgaE>3NeVoIXA0cwiA$Gm93R^xi`~>WJ zopT8Kr^v2|d}4l9@L9{xv0cjxrI=rUi2fxa=X~PweDdl2 zYjEasAAf@k?ETw<&u{79VVftPI==^}&Jge)5Z5W6m_HVL?){&zttFq@e+J9XsYzqy zCA6G(>t%G#?k`~1ruN(TH}u~TAJ_U1w0?E^{}XJDd<$Oz%h}6))BY=BEpx@meG}{~ zvHu0OR@U-wurcz9c?&$dWd8?T&iOC4G4lDw`iEm7b{nGKTEox{{y{+!HwgR&lyUZU zf8>mF{qBWpP-mV-eC7U4*m7fve0_4}XMf#`Mse!w3wHk4`+@D_`{8$Nf3V*% z|E*A8;--W3rT+}Dwd6gE^it<8j2Z8Ie)&B+@ zjP3oGPs|};dG~oHTHAT-VLj`{mwOeTI&}^Mo1cA{4c;5^JK_0QL(Vwsdk?LzPL0FC z<-0LQACmV?pPc!*+ulQM-zV#-Q|}0{>(4!D1{VwrSPfkdm8$5#K*p86n=I3o(Z-_`kn=r%ePDWRK!~L6(@H& z*qQSUs)FUr&HY({-Hzx>UI*BoiRlCzC!bm?!SeZ@oeh>tjdQ?qsc|k?uKYba58M0l zSaI*Vu;rG*rOx@-t|9gX;PN-G8`~K9{2pHgww6Bc?`pJ<=cKO(k#oPqsoe{9-MLd2 zg7e)g&$I?!U(TctY_7cLnA+-m_tt{#?fDpQUuU}*o!N}J2)P__HgRUV1nfTK-MSR) zyCt8P%fRy9yLD)7=dp+NtQ%kMWPIw>xdLo{_Tfr!-Yw738gj;2-#cu5b!uD%F5jW6 zvE}le(kEyBDQNS(!`i-!+UnHn2fO~y9~6@GPkUk|oM&ie+ioOj int { } made += 1 } + # Compute programs: compute.list names each one (name|file.comp). The renderer loads + # .comp.spv straight from the directory; its bindings are fixed by convention - binding 0 + # a uniform block of parameters, 1.. storage buffers - so there is nothing to reflect. + var made_c = 0 + let clist = read_file("packages/ludic.render3d/shaders/compute.list") + if clist != null { + let clines = Text.split(clist, "\n") + var ci = 0 + while ci < len(clines) { + let cl = Text.trim(clines[ci]) + ci += 1 + if slen(cl) == 0 or cl[0] == '#' { continue } + let cp = Text.split(cl, "|") + if len(cp) < 2 { err(`shaders: bad compute line: {cl}\n`); return 1 } + let glsl = `{tmp}/{cp[0]}.comp.glsl` + if not write_file(glsl, "#version 460\n" + shd_file(cp[1])) { err(`shaders: cannot write {glsl}\n`); return 1 } + let spv = `{outdir}/{cp[0]}.comp.spv` + let logf = `{tmp}/{cp[0]}.comp.log` + if not shq(`{shd_q(shd_bin + "/glslangValidator")} -V -S comp {shd_q(glsl)} -o {shd_q(spv)} > {shd_q(logf)} 2>&1`) { + err(`shaders: {cp[1]} (compute) did not compile:\n{capture("grep ERROR " + shd_q(logf) + " | head -5")}\n`) + return 1 + } + if not shq(`{shd_q(shd_bin + "/spirv-val")} --target-env vulkan1.3 {shd_q(spv)} > /dev/null 2>&1`) { err(`shaders: {spv} fails spirv-val\n`); return 1 } + made_c += 1 + } + } if not write_file(`{outdir}/manifest.txt`, sb_str(mf)) { err("shaders: cannot write manifest.txt\n"); return 1 } if check { let real = "packages/ludic.render3d/shaders/spv" @@ -287,6 +313,6 @@ function cmd_shaders() -> int { } var note = "" if check { note = " (unchanged)" } - print(`OK {string(made)} programs, {string(made * 2)} SPIR-V stages{note}`) + print(`OK {string(made)} programs, {string(made * 2 + made_c)} SPIR-V stages ({string(made_c)} compute){note}`) return 0 }