diff --git a/packages/ludic.render3d/gpu.ludic b/packages/ludic.render3d/gpu.ludic index 224c32bb..aef9ed0c 100644 --- a/packages/ludic.render3d/gpu.ludic +++ b/packages/ludic.render3d/gpu.ludic @@ -394,7 +394,7 @@ function gpu_caps_probe() -> void { # DLSS super resolution and Reflex run through NVIDIA Streamline (streamline.ludic), and HDR output is an # HDR10 swapchain (gpu_vk_draw.ludic); whether this # machine can use one is the caps' question, not this one. -function gpu_feature_implemented(f: int) -> bool { return f == GF_VULKAN or f == GF_DLSS or f == GF_REFLEX or f == GF_HDR } +function gpu_feature_implemented(f: int) -> bool { return f == GF_VULKAN or f == GF_DLSS or f == GF_REFLEX or f == GF_HDR or f == GF_MESH_GRASS } # ---- vertex data -------------------------------------------------------------------- # A Mesh is built through these and records what it is made of - which buffer feeds which @@ -524,6 +524,9 @@ function gpu_buffer_free(buf: int) -> void { function gpu_has_compute() -> bool { return gpu_kind == GPU_VK } # several records in one indirect draw, each with its own firstInstance function gpu_has_mdi() -> bool { return gpu_kind == GPU_VK and gvk_has_mdi } +# mesh shaders (VK_EXT_mesh_shader): a *.mesh program drawn with gpu_draw_mesh_tasks +function gpu_has_mesh() -> bool { return gpu_kind == GPU_VK and gvk_has_mesh } +function gpu_draw_mesh_tasks(x: int, y: int, z: int) -> void { if gpu_kind == GPU_VK { gvk_draw_mesh_tasks_now(x, y, z) } } function gpu_compute(name: string, n_bufs: int) -> int { if gpu_kind != GPU_VK { return 0 } return gvk_compute_new(name, n_bufs) diff --git a/packages/ludic.render3d/gpu_vk.ludic b/packages/ludic.render3d/gpu_vk.ludic index e7dd3788..30a56421 100644 --- a/packages/ludic.render3d/gpu_vk.ludic +++ b/packages/ludic.render3d/gpu_vk.ludic @@ -58,6 +58,8 @@ function gvk_ext_in(props: bytes, n: int, want: string) -> bool { # the caller stays on OpenGL. var gvk_has_mdi: bool = false # multiDrawIndirect + drawIndirectFirstInstance var gvk_has_dic: bool = false # drawIndirectCount +var gvk_has_mesh: bool = false # VK_EXT_mesh_shader with its meshShader feature on +var gvk_mesh_fn: pointer = null # vkCmdDrawMeshTasksEXT: the device's own, the interposer exports none function gvk_init() -> bool { if gvk_ready { return true } @@ -161,6 +163,8 @@ function gvk_init() -> bool { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES) Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_dynamicRendering, 1) Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_synchronization2, 1) + # maintenance4: glslang's mesh stages declare their work group size with LocalSizeId, which needs it + if Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_maintenance4) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_maintenance4, 1) } # Streamline's hooks keep their own data on our objects (private data slots), and ask the device # to have the feature rather than turning it on themselves if gsl_on and Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_privateData) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_privateData, 1) } @@ -198,6 +202,27 @@ function gvk_init() -> bool { let nde = Vk.get_i32(cnt, 0) let dexts = bytes(nde * VkExtensionProperties_sizeof + 8) Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, dexts) + # mesh-shader grass: the extension, and its meshShader feature asked for where the device has it. + # R3D_NO_MESH=1 leaves it off. + gvk_has_mesh = false + if gvk_ext_in(dexts, nde, VK_EXT_MESH_SHADER_EXTENSION_NAME) and not Os.has_env("R3D_NO_MESH") { + let fm = bytes(VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof) + Vk.zero(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof) + Vk.put_i32(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MESH_SHADER_FEATURES_EXT) + let qm = bytes(VkPhysicalDeviceFeatures2_sizeof) + Vk.zero(qm, VkPhysicalDeviceFeatures2_sizeof) + Vk.put_i32(qm, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2) + Vk.put_ptr(qm, VkPhysicalDeviceFeatures2_pNext, fm) + Vk.get_physical_device_features2(gvk_pd, qm) + if Vk.get_i32(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_meshShader) == 1 { + let wm = bytes(VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof) + Vk.zero(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof) + Vk.put_i32(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MESH_SHADER_FEATURES_EXT) + Vk.put_i32(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_meshShader, 1) + Vk.put_ptr(want13, VkPhysicalDeviceVulkan13Features_pNext, wm) + gvk_has_mesh = true + } + } let prio = bytes(4) Vk.put_i32(prio, 0, 0x3F800000) let qci = bytes(VkDeviceQueueCreateInfo_sizeof) @@ -226,6 +251,7 @@ function gvk_init() -> bool { gvk_has_hdr_meta = gvk_ext_in(dexts, nde, VK_EXT_HDR_METADATA_EXTENSION_NAME) and Vk.has("vkSetHdrMetadataEXT") == 1 if gvk_has_hdr_meta { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_HDR_METADATA_EXTENSION_NAME); n_dext += 1 } } + if gvk_has_mesh { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_MESH_SHADER_EXTENSION_NAME); n_dext += 1 } if n_dext > 0 { Vk.put_i32(dci, VkDeviceCreateInfo_enabledExtensionCount, n_dext) Vk.put_ptr(dci, VkDeviceCreateInfo_ppEnabledExtensionNames, dext_names) @@ -236,12 +262,16 @@ function gvk_init() -> bool { Vk.get_device_queue(gvk_dev, gvk_family, 0, out) gvk_queue = Vk.get_ptr(out, 0) gsl_probe_device(gvk_pd) + if gvk_has_mesh { + gvk_mesh_fn = Vk.get_device_proc_addr(gvk_dev, "vkCmdDrawMeshTasksEXT") + if gvk_mesh_fn == null { gvk_has_mesh = false } + } gvk_mp = bytes(VkPhysicalDeviceMemoryProperties_sizeof) Vk.get_physical_device_memory_properties(gvk_pd, gvk_mp) if not gvk_cmd_init() { return false } gvk_ready = true - print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x, multi-draw indirect {gvk_has_mdi}, indirect count {gvk_has_dic}`) + print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x, multi-draw indirect {gvk_has_mdi}, indirect count {gvk_has_dic}, mesh shaders {gvk_has_mesh}`) return true } diff --git a/packages/ludic.render3d/gpu_vk_draw.ludic b/packages/ludic.render3d/gpu_vk_draw.ludic index f668d565..fdbe74dc 100644 --- a/packages/ludic.render3d/gpu_vk_draw.ludic +++ b/packages/ludic.render3d/gpu_vk_draw.ludic @@ -50,14 +50,21 @@ function gvk_module(path: string) -> long { # The Vulkan side of a program handle the renderer already made (gpu_program): the handle's # manifest key finds the variant. Returns false when there is no such variant. +# a program whose first stage is a mesh shader (its first file is *.mesh): its pipeline has no vertex +# input, and its first stage's bindings are the mesh stage's +var gvk_prog_mesh: []int = null +function gvk_stage_first(p: int) -> int { + if gvk_prog_mesh != null and p < len(gvk_prog_mesh) and gvk_prog_mesh[p] == 1 { return VK_SHADER_STAGE_MESH_BIT_EXT } + return VK_SHADER_STAGE_VERTEX_BIT +} function gvk_program(p: int, key: string, spv_dir: string) -> bool { let zero: long = 0 if gvk_prog_var == null { gvk_prog_var = new []GpuVariant; gvk_prog_vs = new []long; gvk_prog_fs = new []long - gvk_prog_dsl = new []long; gvk_prog_layout = new []long + gvk_prog_dsl = new []long; gvk_prog_layout = new []long; gvk_prog_mesh = new []int } while len(gvk_prog_var) <= p { - push(gvk_prog_var, null); push(gvk_prog_vs, zero); push(gvk_prog_fs, zero); push(gvk_prog_dsl, zero); push(gvk_prog_layout, zero) + push(gvk_prog_var, null); push(gvk_prog_vs, zero); push(gvk_prog_fs, zero); push(gvk_prog_dsl, zero); push(gvk_prog_layout, zero); push(gvk_prog_mesh, 0) } let parts = Text.split(key, "|") var defs = "" @@ -65,6 +72,7 @@ function gvk_program(p: int, key: string, spv_dir: string) -> bool { let v = gpu_variant_find_key(parts[0], parts[1], defs) if v == null { print(`r3d: vulkan: no SPIR-V variant for {key}`); return false } let vs = gvk_module(`{spv_dir}/{v.id}.vert.spv`) + if Text.ends_with(v.vs, ".mesh") { gvk_prog_mesh[p] = 1 } else { gvk_prog_mesh[p] = 0 } let fs = gvk_module(`{spv_dir}/{v.id}.frag.spv`) if vs == 0 or fs == 0 { return false } @@ -80,7 +88,7 @@ function gvk_program(p: int, key: string, spv_dir: string) -> bool { Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, 0) Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC) Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1) - Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_VERTEX_BIT) + Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, gvk_stage_first(p)) k += 1 } if v.fblock >= 0 { @@ -94,7 +102,7 @@ function gvk_program(p: int, key: string, spv_dir: string) -> bool { Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, v.t_bind[t]) Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER) Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1) - Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT) + Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, gvk_stage_first(p) | VK_SHADER_STAGE_FRAGMENT_BIT) k += 1 } let dslci = bytes(VkDescriptorSetLayoutCreateInfo_sizeof) @@ -371,7 +379,7 @@ function gvk_pipeline(p: int, m: Mesh, st: GvkState, n_color: int, color_fmt: in let stages = bytes(ss * 2) Vk.zero(stages, ss * 2) Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO) - Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_stage, VK_SHADER_STAGE_VERTEX_BIT) + Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_stage, gvk_stage_first(p)) Vk.put_i64(stages, VkPipelineShaderStageCreateInfo_module, gvk_prog_vs[p]) Vk.put_ptr(stages, VkPipelineShaderStageCreateInfo_pName, "main") Vk.put_i32(stages, ss + VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO) @@ -524,8 +532,11 @@ function gvk_pipeline(p: int, m: Mesh, st: GvkState, n_color: int, color_fmt: in Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pNext, prci) Vk.put_i32(gpci, VkGraphicsPipelineCreateInfo_stageCount, 2) Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pStages, stages) - Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pVertexInputState, vin) - Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pInputAssemblyState, ias) + # a mesh-shader pipeline has neither: the mesh stage makes its own vertices + if gvk_stage_first(p) == VK_SHADER_STAGE_VERTEX_BIT { + Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pVertexInputState, vin) + Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pInputAssemblyState, ias) + } Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pViewportState, vps) Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pRasterizationState, rs) Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pMultisampleState, ms) @@ -1273,6 +1284,12 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc } var n = count if n == 0 and m != null { n = m.count } + if gvk_mesh_x > 0 { + # a mesh-shader dispatch (gpu_draw_mesh_tasks): the device's command, through a pointer + Vk.sl_call_piii(gvk_mesh_fn, cb, gvk_mesh_x, gvk_mesh_y, gvk_mesh_z) + if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) } + return + } if m != null and m.ebo != 0 { let zero: long = 0 var itype = VK_INDEX_TYPE_UINT32 @@ -1461,6 +1478,16 @@ var gvk_ind_off: int = 0 var gvk_ind_n: int = 0 var gvk_ind_cbuf: int = 0 var gvk_ind_coff: int = 0 +# x * y * z mesh-shader invocations with the current program (a *.mesh one) and state +var gvk_mesh_x: int = 0 +var gvk_mesh_y: int = 0 +var gvk_mesh_z: int = 0 +function gvk_draw_mesh_tasks_now(x: int, y: int, z: int) -> void { + if not gvk_has_mesh or gvk_mesh_fn == null or x <= 0 or y <= 0 or z <= 0 { return } + gvk_mesh_x = x; gvk_mesh_y = y; gvk_mesh_z = z + gvk_draw_now(null, 0, 0, 1) + gvk_mesh_x = 0 +} function gvk_draw_indirect_now(m: Mesh, cmds: int, offset: int, n: int, count_buf: int, count_off: int) -> void { if m == null or m.ebo == 0 or cmds <= 0 or n <= 0 { return } gvk_ind_buf = cmds; gvk_ind_off = offset; gvk_ind_n = n; gvk_ind_cbuf = count_buf; gvk_ind_coff = count_off diff --git a/packages/ludic.render3d/grass.ludic b/packages/ludic.render3d/grass.ludic index d4104447..fb5952f0 100644 --- a/packages/ludic.render3d/grass.ludic +++ b/packages/ludic.render3d/grass.ludic @@ -35,6 +35,17 @@ var grass_cmds: int = 0 var grass_band_start: words = null # per band: its first record var grass_band_cells: words = null # ... and its 16 m cells per tile side var grass_band_n: int = 0 +# Mesh-shader grass (Settings, Video, Advanced): the chunked path's records, but each chunk is one +# mesh dispatch - work group y a tile, x a batch of GRASS_MESH_BLADES of its blades - so a blade the +# tests reject emits nothing instead of eight degenerate vertices. R3D_MESH_GRASS=1 / 0 overrides. +const GRASS_MESH_BLADES: int = 16 +var grass_mesh_on: bool = false +var grass_mesh_prog: int = 0 +function r3d_mesh_grass(on: bool) -> void { + grass_mesh_on = on + if Os.has_env("R3D_MESH_GRASS") { grass_mesh_on = Text.to_int(Os.env("R3D_MESH_GRASS")) != 0 } +} +function grass_mesh_live() -> bool { return grass_mesh_on and grass_merge and grass_mesh_prog != 0 and gpu_has_mesh() } # a blade: `rows` rows of 2 vertices (x across, y along, z bend), attribute 2 = uv function grass_blade_mesh(rows: int) -> Mesh { @@ -81,6 +92,7 @@ function grass_init() -> void { grass_cmds = gpu_buffer_new() } grass_prog = r3d_program("grass.vert", "model.frag", defs) + if grass_merge and gpu_has_mesh() { grass_mesh_prog = r3d_program("grass.mesh", "model.frag", "#define FOLIAGE\n#define BLADE\n#define MESH\n") } grass_mesh = grass_blade_mesh(4) grass_wind = fl(2.4) grass_s0 = fl(0.11) @@ -155,7 +167,8 @@ function grass_tiles(size: int, d_min: int, d_max: int) -> void { function grass_draw() -> void { if not grass_on or ter_reflect or grass_prog == 0 { return } - let p = grass_prog + var p = grass_prog + if grass_mesh_live() { p = grass_mesh_prog } gpu_use_program(p) u_mat4(gpu_uniform(p, "u_view"), cam_view) u_mat4(gpu_uniform(p, "u_proj"), cam_proj) @@ -203,8 +216,9 @@ function grass_draw() -> void { # the records gathered this frame: uploaded once, then each band GRASS_CHUNK records a draw function grass_flush() -> void { if grass_n == 0 { return } - let p = grass_prog - gpu_buffer_upload(grass_cmds, grass_n * 20, grass_rec, GPU_DYNAMIC) + var p = grass_prog + let mesh = grass_mesh_live() + if mesh { p = grass_mesh_prog } else { gpu_buffer_upload(grass_cmds, grass_n * 20, grass_rec, GPU_DYNAMIC) } for b in 0 .. grass_band_n { let s = grass_band_start[b] var e = grass_n @@ -216,7 +230,16 @@ function grass_flush() -> void { if m > GRASS_CHUNK { m = GRASS_CHUNK } for q in 0 .. m * 4 { grass_chunk_tv[q] = grass_tv[k * 4 + q] } u_f4v(gpu_uniform(p, "u_tiles"), m, grass_chunk_tv) - gpu_draw_mesh_indirect(grass_mesh, grass_cmds, k * 20, m, 0, 0) + if mesh { + # enough blade batches for the chunk's largest tile, capped as the instanced path caps a tile + let cells = grass_band_cells[b] + var most = 0 + for q in 0 .. m { let t = f_to_int(grass_tv[(k + q) * 4 + 2]) * cells * cells; if t > most { most = t } } + if most > 65535 { most = 65535 } + gpu_draw_mesh_tasks((most + GRASS_MESH_BLADES - 1) / GRASS_MESH_BLADES, m, 1) + } else { + gpu_draw_mesh_indirect(grass_mesh, grass_cmds, k * 20, m, 0, 0) + } grass_draws += 1 k += m } diff --git a/packages/ludic.render3d/shaders/grass.mesh b/packages/ludic.render3d/shaders/grass.mesh new file mode 100644 index 00000000..f7c3de5f --- /dev/null +++ b/packages/ludic.render3d/shaders/grass.mesh @@ -0,0 +1,167 @@ +// Procedural ground-cover blades as a MESH shader (Vulkan, VK_EXT_mesh_shader): the same blades +// grass.vert draws - the same places, density, thinning, culling, size, sway and lighting normal - +// but a blade the tests reject emits nothing, where the instanced path still runs all eight of its +// vertices to a degenerate position. +// +// One dispatch covers a chunk of tiles (grass.ludic): work group y is the tile's place in u_tiles, +// work group x a batch of BLADES blade indices within that tile. Everything past the tile's own +// count is skipped. +layout(local_size_x = 1, local_size_y = 1, local_size_z = 1) in; +const int BLADES = 16; // blades an invocation may emit +const int ROWS = 4; // grass_blade_mesh(4): two vertices a row, three quads +layout(triangles, max_vertices = 128, max_primitives = 96) out; + +uniform mat4 u_view; +uniform mat4 u_proj; +uniform mat4 u_vp; +uniform vec3 u_cam_pos; +uniform float u_time; +uniform sampler2D u_ts_height; +uniform vec2 u_ts_origin; +uniform float u_ts_half; +uniform sampler2D u_ortho; +uniform float u_ortho_on; +uniform float u_lake_level; +uniform float u_sea_level; +uniform vec4 u_lake; +uniform float u_snow_line; +uniform float u_wind; +uniform int u_tile_cells; // 16 m cells per tile side +uniform vec4 u_tiles[256]; // per tile of the dispatch: corner x, corner z, indices per cell +uniform float u_s0; +uniform float u_d0; +uniform float u_radius; +uniform int u_dbg; +out vec3 v_wpos[]; +out vec3 v_nrm[]; +out vec2 v_uv[]; +out float v_seed[]; +out vec2 v_rot[]; +out float v_hull[]; + +const float CELL = 16.0; +float hash1(vec2 p) { return fract(sin(dot(p, vec2(127.1, 311.7))) * 43758.5453123); } +uint pcg(uint v) { uint s = v * 747796405u + 2891336453u; uint w = ((s >> ((s >> 28u) + 4u)) ^ s) * 277803737u; return (w >> 22u) ^ w; } +float bladeHash(ivec2 cell, int j, int k) { + uint h = pcg(uint(cell.x + 32768) * 73856093u ^ uint(cell.y + 32768) * 19349663u ^ uint(j) * 83492791u ^ uint(k) * 2654435761u); + return float(h) * (1.0 / 4294967295.0); +} +// no implicit level of detail outside a fragment shader: every lookup names level 0 +float heightSmooth(sampler2D tex, vec2 uv) { + vec2 res = vec2(textureSize(tex, 0)); + vec2 t = uv * res - 0.5; + vec2 f = fract(t); + vec2 i = floor(t); + vec2 w0 = (1.0 - f) * (1.0 - f) * (1.0 - f) / 6.0; + vec2 w1 = (4.0 - 6.0 * f * f + 3.0 * f * f * f) / 6.0; + vec2 w3 = f * f * f / 6.0; + vec2 w2 = 1.0 - w0 - w1 - w3; + vec2 s0 = w0 + w1, s1 = w2 + w3; + vec2 o0 = (i - 1.0 + w1 / s0 + 0.5) / res; + vec2 o1 = (i + 1.0 + w3 / s1 + 0.5) / res; + return (textureLod(tex, vec2(o0.x, o0.y), 0.0).r * s0.x + textureLod(tex, vec2(o1.x, o0.y), 0.0).r * s1.x) * s0.y + + (textureLod(tex, vec2(o0.x, o1.y), 0.0).r * s0.x + textureLod(tex, vec2(o1.x, o1.y), 0.0).r * s1.x) * s1.y; +} + +void main() { + int ti = int(gl_WorkGroupID.y); + vec2 tile = u_tiles[ti].xy; + int per_cell = int(u_tiles[ti].z + 0.5); + int total = min(per_cell * u_tile_cells * u_tile_cells, 65535); // the instanced path's cap on a tile + int first = int(gl_WorkGroupID.x) * BLADES; + int nv = 0; + int np = 0; + for (int b = 0; b < BLADES; b++) { + int i = first + b; + if (per_cell <= 0 || i >= total) break; + int c = i / per_cell; + int j = i - c * per_cell; + vec2 cell = tile + vec2(float(c % u_tile_cells), float(c / u_tile_cells)) * CELL; + vec2 cid = floor(cell / CELL + 0.5); + ivec2 ci = ivec2(cid); + float fj = float(j); + vec2 hv = vec2(bladeHash(ci, j, 0), bladeHash(ci, j, 1)); + vec2 xz = cell + hv * CELL; + float dist = length(xz - u_cam_pos.xz); + if (dist >= u_radius) continue; + float spacing = u_s0 * (1.0 + dist / u_d0); + float count = CELL * CELL / (spacing * spacing) * (1.0 - smoothstep(u_radius * 0.7, u_radius, dist)); + if (fj >= count) continue; + float life = 1.0 - smoothstep(0.8, 1.0, fj / max(count, 1.0)); + vec2 huv = (xz - u_ts_origin) / (2.0 * u_ts_half) + 0.5; + if (huv.x < 0.0 || huv.x > 1.0 || huv.y < 0.0 || huv.y > 1.0) continue; + vec4 ht = textureLod(u_ts_height, huv, 0.0); + vec4 croot = u_vp * vec4(xz.x, ht.r, xz.y, 1.0); + if (croot.w < -1.0 || abs(croot.x) > croot.w * 1.25 + 1.5 || abs(croot.y) > croot.w * 1.4 + 1.5) continue; + vec3 gn = normalize(ht.gba); + float h3 = bladeHash(ci, j, 2), h4 = bladeHash(ci, j, 3); + float wl = u_sea_level; + if (u_lake.z > 0.0) { vec2 q = (xz - u_lake.xy) / u_lake.zw; if (dot(q, q) < 1.0) wl = max(wl, u_lake_level); } + float ok = (1.0 - smoothstep(0.30, 0.55, 1.0 - gn.y)) * smoothstep(0.0, 0.6, ht.r - wl - 0.15) * smoothstep(u_snow_line - 80.0, u_snow_line - 200.0, ht.r); + if (u_ortho_on > 0.5) { + vec3 oc = textureLod(u_ortho, huv, 1.5).rgb; + ok *= 0.25 + 0.75 * smoothstep(0.0, 0.02, oc.g - max(oc.r, oc.b)); + } + if (h4 > ok) continue; + bool far = dist > 300.0; + float h = far ? ht.r : heightSmooth(u_ts_height, huv); + if (far) h += 0.03; + if ((u_dbg & 1) != 0) h += 0.3; + float seed = hv.x * 0.7 + hv.y * 0.3; + float ang = hv.y * 6.2831853; + float s = sin(ang), c_ = cos(ang); + float grow = spacing / u_s0; + float tall = mix(0.18, 0.42, h3) * mix(0.8, 1.2, hash1(cid * 0.1)) * (1.0 + 0.35 * smoothstep(1.0, 12.0, grow)) * life; + float bw = 0.028 * mix(1.0, 0.45 * grow, smoothstep(1.0, 4.0, grow)); + if (far) { bw = max(bw, spacing * 0.35); tall = min(tall, spacing * 0.3); } + float gust = sin(xz.x * 0.09 + u_time * 1.1) * 0.5 + sin(xz.y * 0.13 - u_time * 0.8 + xz.x * 0.05) * 0.5; + float ph = u_time * 1.7 + seed * 6.2831 + xz.x * 0.05 + xz.y * 0.07; + float sway = (sin(ph) * 0.6 + sin(ph * 2.3 + 1.0) * 0.4 + gust) * u_wind; + // the ground's frame, shared by the blade's vertices + vec3 up = vec3(0.0, 1.0, 0.0); + vec3 k = cross(up, gn); + float sk = length(k), ck = gn.y; + bool tilt = sk > 1e-4; + if (tilt) k /= sk; + vec3 n = vec3(0.0, 0.3, 1.0); + n = normalize(vec3(c_ * n.x + s * n.z, n.y, -s * n.x + c_ * n.z)); + if (tilt) n = normalize(n * ck + cross(k, n) * sk + k * dot(k, n) * (1.0 - ck)); + n = normalize(mix(n, gn, smoothstep(2.0, 12.0, dist))); + float hull = (dist > 2.0 || far || (u_dbg & 2) != 0) ? -1.0 : 1.0; + // the blade: grass_blade_mesh(4)'s vertices, placed as grass.vert places them + for (int r = 0; r < ROWS; r++) { + float t = float(r) / float(ROWS - 1); + float taper = max(1.0 - t * t * sqrt(t), 0.12); + float bend = t * t * 0.28; + for (int sd = 0; sd < 2; sd++) { + vec3 a_pos = vec3((float(sd) - 0.5) * taper, t, bend); + vec2 a_uv = vec2(float(sd), t); + vec3 p = vec3(a_pos.x * bw, a_pos.y * tall, a_pos.z * tall * (0.6 + 0.8 * h4)); + float hgt = max(p.y, 0.0); + p.x += sway * hgt * hgt * 0.35; + p.z += sway * hgt * hgt * 0.15 * cos(ph * 0.7); + p = vec3(c_ * p.x + s * p.z, p.y, -s * p.x + c_ * p.z); + if (tilt) p = p * ck + cross(k, p) * sk + k * dot(k, p) * (1.0 - ck); + vec3 w = vec3(xz.x, h - 0.02, xz.y) + p; + vec4 clip = u_proj * u_view * vec4(w, 1.0); + clip.z = (clip.z + clip.w) * 0.5; // OpenGL's depth range to Vulkan's, as the vertex wrapper does + int o = nv + r * 2 + sd; + gl_MeshVerticesEXT[o].gl_Position = clip; + v_wpos[o] = w; + v_nrm[o] = n; + v_uv[o] = far ? vec2(a_uv.x, 0.45 + 0.2 * a_uv.y) : a_uv; + v_seed[o] = seed; + v_rot[o] = vec2(s, c_); + v_hull[o] = hull; + } + } + for (int q = 0; q < ROWS - 1; q++) { + uint bb = uint(nv + q * 2); + gl_PrimitiveTriangleIndicesEXT[np] = uvec3(bb, bb + 1u, bb + 2u); + gl_PrimitiveTriangleIndicesEXT[np + 1] = uvec3(bb + 1u, bb + 3u, bb + 2u); + np += 2; + } + nv += ROWS * 2; + } + SetMeshOutputsEXT(uint(nv), uint(np)); +} diff --git a/packages/ludic.render3d/shaders/spv/d48d6a48.frag.spv b/packages/ludic.render3d/shaders/spv/d48d6a48.frag.spv new file mode 100644 index 00000000..f83e77c0 Binary files /dev/null and b/packages/ludic.render3d/shaders/spv/d48d6a48.frag.spv differ diff --git a/packages/ludic.render3d/shaders/spv/d48d6a48.vert.spv b/packages/ludic.render3d/shaders/spv/d48d6a48.vert.spv new file mode 100644 index 00000000..5d50274f Binary files /dev/null and b/packages/ludic.render3d/shaders/spv/d48d6a48.vert.spv differ diff --git a/packages/ludic.render3d/shaders/spv/manifest.txt b/packages/ludic.render3d/shaders/spv/manifest.txt index 1738f4c7..fcc70708 100644 --- a/packages/ludic.render3d/shaders/spv/manifest.txt +++ b/packages/ludic.render3d/shaders/spv/manifest.txt @@ -628,6 +628,73 @@ T a4683767 u_prefilter 8 T a4683767 u_shadow 9 T a4683767 u_tershadow 10 T a4683767 u_ts_height 11 +P d48d6a48 grass.mesh model.frag #define FOLIAGE;#define BLADE;#define MESH; +B d48d6a48 vert 0 4384 +U d48d6a48 vert u_view 0 mat4 1 0 +U d48d6a48 vert u_proj 64 mat4 1 0 +U d48d6a48 vert u_vp 128 mat4 1 0 +U d48d6a48 vert u_cam_pos 192 vec3 1 0 +U d48d6a48 vert u_time 204 float 1 0 +U d48d6a48 vert u_ts_origin 208 vec2 1 0 +U d48d6a48 vert u_ts_half 216 float 1 0 +U d48d6a48 vert u_ortho_on 220 float 1 0 +U d48d6a48 vert u_lake_level 224 float 1 0 +U d48d6a48 vert u_sea_level 228 float 1 0 +U d48d6a48 vert u_lake 240 vec4 1 0 +U d48d6a48 vert u_snow_line 256 float 1 0 +U d48d6a48 vert u_wind 260 float 1 0 +U d48d6a48 vert u_tile_cells 264 int 1 0 +U d48d6a48 vert u_tiles 272 vec4 256 16 +U d48d6a48 vert u_s0 4368 float 1 0 +U d48d6a48 vert u_d0 4372 float 1 0 +U d48d6a48 vert u_radius 4376 float 1 0 +U d48d6a48 vert u_dbg 4380 int 1 0 +B d48d6a48 frag 1 892 +U d48d6a48 frag u_cascade_vp 0 mat4 5 64 +U d48d6a48 frag u_cascade_split 320 float 5 16 +U d48d6a48 frag u_cascade_range 400 float 5 16 +U d48d6a48 frag u_cascade_texel 480 float 5 16 +U d48d6a48 frag u_sun_dir 560 vec3 1 0 +U d48d6a48 frag u_sun_color 576 vec3 1 0 +U d48d6a48 frag u_cam_pos 592 vec3 1 0 +U d48d6a48 frag u_prefilter_levels 604 float 1 0 +U d48d6a48 frag u_fog_density 608 float 1 0 +U d48d6a48 frag u_fog_height_falloff 612 float 1 0 +U d48d6a48 frag u_fog_base 616 float 1 0 +U d48d6a48 frag u_clip_y 620 float 1 0 +U d48d6a48 frag u_spec_scale 624 float 1 0 +U d48d6a48 frag u_sky_rot 632 vec2 1 0 +U d48d6a48 frag u_ibl_scale 640 vec3 1 0 +U d48d6a48 frag u_daylight 652 float 1 0 +U d48d6a48 frag u_fire_pos 656 vec3 1 0 +U d48d6a48 frag u_fire_color 672 vec3 1 0 +U d48d6a48 frag u_hand_pos 688 vec3 1 0 +U d48d6a48 frag u_hand_color 704 vec3 1 0 +U d48d6a48 frag u_hand_dir 720 vec3 1 0 +U d48d6a48 frag u_hand_cone 732 float 1 0 +U d48d6a48 frag u_ts_origin 736 vec2 1 0 +U d48d6a48 frag u_ts_half 744 float 1 0 +U d48d6a48 frag u_ts_on 748 float 1 0 +U d48d6a48 frag u_force_cascade 752 int 1 0 +U d48d6a48 frag u_cloud_shadow 756 float 1 0 +U d48d6a48 frag u_time 760 float 1 0 +U d48d6a48 frag u_model_h 764 float 1 0 +U d48d6a48 frag u_view 768 mat4 1 0 +U d48d6a48 frag u_tint 832 vec3 1 0 +U d48d6a48 frag u_rough_scale 844 float 1 0 +U d48d6a48 frag u_emissive 848 float 1 0 +U d48d6a48 frag u_blade_base 864 vec3 1 0 +U d48d6a48 frag u_blade_tip 880 vec3 1 0 +T d48d6a48 u_arm 2 +T d48d6a48 u_brdf 3 +T d48d6a48 u_diff 4 +T d48d6a48 u_irradiance 5 +T d48d6a48 u_nrm 6 +T d48d6a48 u_ortho 7 +T d48d6a48 u_prefilter 8 +T d48d6a48 u_shadow 9 +T d48d6a48 u_tershadow 10 +T d48d6a48 u_ts_height 11 P 7ed7de52 grass.vert model.frag #define FOLIAGE;#define BLADE; B 7ed7de52 vert 0 296 U 7ed7de52 vert u_view 0 mat4 1 0 diff --git a/packages/ludic.render3d/shaders/variants.list b/packages/ludic.render3d/shaders/variants.list index 85bcff7e..8dc8e83e 100644 --- a/packages/ludic.render3d/shaders/variants.list +++ b/packages/ludic.render3d/shaders/variants.list @@ -17,6 +17,7 @@ fullscreen.vert|ternormal.frag| fullscreen.vert|tershadow.frag|#define NOISE_ONLY; fullscreen.vert|tonemap.frag| fullscreen.vert|tonemap.frag|#define HDR10; +grass.mesh|model.frag|#define FOLIAGE;#define BLADE;#define MESH; grass.vert|model.frag|#define FOLIAGE;#define BLADE; grass.vert|model.frag|#define FOLIAGE;#define BLADE;#define TILES; impostor.vert|impostor.frag| diff --git a/runtime/native/vk.ludic b/runtime/native/vk.ludic index cd352a15..7ad3fdfd 100644 --- a/runtime/native/vk.ludic +++ b/runtime/native/vk.ludic @@ -43,3 +43,6 @@ extern function vk_sl_call_p(fn: pointer, a: pointer) -> int = "lsl_call_p" extern function vk_sl_call_pp(fn: pointer, a: pointer, b: pointer) -> int = "lsl_call_pp" extern function vk_sl_call_ppp(fn: pointer, a: pointer, b: pointer, c: pointer) -> int = "lsl_call_ppp" extern function vk_sl_call_ip(fn: pointer, a: int, b: pointer) -> int = "lsl_call_ip" +# a command-buffer command with three counts through a pointer (vkCmdDrawMeshTasksEXT, which the +# Streamline interposer does not export; looked up with vkGetDeviceProcAddr) +extern function vk_sl_call_piii(fn: pointer, a: pointer, b: int, c: int, d: int) = "lsl_call_piii"