feat(render3d): mesh-shader grass

The chunked grass path draws each chunk as one mesh-shader dispatch when the
setting asks and the card has VK_EXT_mesh_shader: work group y is a tile, x a
batch of 16 of its blades, and a blade the placement, density, frustum, water
or slope tests reject emits nothing - the instanced path still runs its eight
vertices to a degenerate position. grass.mesh generates the same blades as
grass.vert (grass_blade_mesh(4)'s rows, the same hashes, sway and lighting
normal), capped at the instanced path's 65535 a tile.

Vulkan: VK_EXT_mesh_shader with meshShader, and maintenance4 (glslang's mesh
stages declare LocalSizeId); vkCmdDrawMeshTasksEXT looked up per device, as the
Streamline interposer exports none; a *.mesh program's pipeline takes the mesh
stage and no vertex input, its bindings the mesh stage bit. gpu_has_mesh,
gpu_draw_mesh_tasks; r3d_mesh_grass and R3D_MESH_GRASS / R3D_NO_MESH.
bin/ludic-dev rebuilt: the committed binary predated the shader tool's mesh
support and compiled grass.mesh as a vertex stage.

PC (RTX 3070 Ti): the camp matches the chunked path; validation only the
no-window present-id message.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-15 20:20:42 +03:00
parent 6525c11b3c
commit 750d13779d
10 changed files with 334 additions and 13 deletions

View file

@ -394,7 +394,7 @@ function gpu_caps_probe() -> void {
# DLSS super resolution and Reflex run through NVIDIA Streamline (streamline.ludic), and HDR output is an
# HDR10 swapchain (gpu_vk_draw.ludic); whether this
# machine can use one is the caps' question, not this one.
function gpu_feature_implemented(f: int) -> bool { return f == GF_VULKAN or f == GF_DLSS or f == GF_REFLEX or f == GF_HDR }
function gpu_feature_implemented(f: int) -> bool { return f == GF_VULKAN or f == GF_DLSS or f == GF_REFLEX or f == GF_HDR or f == GF_MESH_GRASS }
# ---- vertex data --------------------------------------------------------------------
# A Mesh is built through these and records what it is made of - which buffer feeds which
@ -524,6 +524,9 @@ function gpu_buffer_free(buf: int) -> void {
function gpu_has_compute() -> bool { return gpu_kind == GPU_VK }
# several records in one indirect draw, each with its own firstInstance
function gpu_has_mdi() -> bool { return gpu_kind == GPU_VK and gvk_has_mdi }
# mesh shaders (VK_EXT_mesh_shader): a *.mesh program drawn with gpu_draw_mesh_tasks
function gpu_has_mesh() -> bool { return gpu_kind == GPU_VK and gvk_has_mesh }
function gpu_draw_mesh_tasks(x: int, y: int, z: int) -> void { if gpu_kind == GPU_VK { gvk_draw_mesh_tasks_now(x, y, z) } }
function gpu_compute(name: string, n_bufs: int) -> int {
if gpu_kind != GPU_VK { return 0 }
return gvk_compute_new(name, n_bufs)

View file

@ -58,6 +58,8 @@ function gvk_ext_in(props: bytes, n: int, want: string) -> bool {
# the caller stays on OpenGL.
var gvk_has_mdi: bool = false # multiDrawIndirect + drawIndirectFirstInstance
var gvk_has_dic: bool = false # drawIndirectCount
var gvk_has_mesh: bool = false # VK_EXT_mesh_shader with its meshShader feature on
var gvk_mesh_fn: pointer = null # vkCmdDrawMeshTasksEXT: the device's own, the interposer exports none
function gvk_init() -> bool {
if gvk_ready { return true }
@ -161,6 +163,8 @@ function gvk_init() -> bool {
Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES)
Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_dynamicRendering, 1)
Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_synchronization2, 1)
# maintenance4: glslang's mesh stages declare their work group size with LocalSizeId, which needs it
if Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_maintenance4) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_maintenance4, 1) }
# Streamline's hooks keep their own data on our objects (private data slots), and ask the device
# to have the feature rather than turning it on themselves
if gsl_on and Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_privateData) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_privateData, 1) }
@ -198,6 +202,27 @@ function gvk_init() -> bool {
let nde = Vk.get_i32(cnt, 0)
let dexts = bytes(nde * VkExtensionProperties_sizeof + 8)
Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, dexts)
# mesh-shader grass: the extension, and its meshShader feature asked for where the device has it.
# R3D_NO_MESH=1 leaves it off.
gvk_has_mesh = false
if gvk_ext_in(dexts, nde, VK_EXT_MESH_SHADER_EXTENSION_NAME) and not Os.has_env("R3D_NO_MESH") {
let fm = bytes(VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
Vk.zero(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
Vk.put_i32(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MESH_SHADER_FEATURES_EXT)
let qm = bytes(VkPhysicalDeviceFeatures2_sizeof)
Vk.zero(qm, VkPhysicalDeviceFeatures2_sizeof)
Vk.put_i32(qm, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2)
Vk.put_ptr(qm, VkPhysicalDeviceFeatures2_pNext, fm)
Vk.get_physical_device_features2(gvk_pd, qm)
if Vk.get_i32(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_meshShader) == 1 {
let wm = bytes(VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
Vk.zero(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
Vk.put_i32(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MESH_SHADER_FEATURES_EXT)
Vk.put_i32(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_meshShader, 1)
Vk.put_ptr(want13, VkPhysicalDeviceVulkan13Features_pNext, wm)
gvk_has_mesh = true
}
}
let prio = bytes(4)
Vk.put_i32(prio, 0, 0x3F800000)
let qci = bytes(VkDeviceQueueCreateInfo_sizeof)
@ -226,6 +251,7 @@ function gvk_init() -> bool {
gvk_has_hdr_meta = gvk_ext_in(dexts, nde, VK_EXT_HDR_METADATA_EXTENSION_NAME) and Vk.has("vkSetHdrMetadataEXT") == 1
if gvk_has_hdr_meta { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_HDR_METADATA_EXTENSION_NAME); n_dext += 1 }
}
if gvk_has_mesh { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_MESH_SHADER_EXTENSION_NAME); n_dext += 1 }
if n_dext > 0 {
Vk.put_i32(dci, VkDeviceCreateInfo_enabledExtensionCount, n_dext)
Vk.put_ptr(dci, VkDeviceCreateInfo_ppEnabledExtensionNames, dext_names)
@ -236,12 +262,16 @@ function gvk_init() -> bool {
Vk.get_device_queue(gvk_dev, gvk_family, 0, out)
gvk_queue = Vk.get_ptr(out, 0)
gsl_probe_device(gvk_pd)
if gvk_has_mesh {
gvk_mesh_fn = Vk.get_device_proc_addr(gvk_dev, "vkCmdDrawMeshTasksEXT")
if gvk_mesh_fn == null { gvk_has_mesh = false }
}
gvk_mp = bytes(VkPhysicalDeviceMemoryProperties_sizeof)
Vk.get_physical_device_memory_properties(gvk_pd, gvk_mp)
if not gvk_cmd_init() { return false }
gvk_ready = true
print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x, multi-draw indirect {gvk_has_mdi}, indirect count {gvk_has_dic}`)
print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x, multi-draw indirect {gvk_has_mdi}, indirect count {gvk_has_dic}, mesh shaders {gvk_has_mesh}`)
return true
}

View file

@ -50,14 +50,21 @@ function gvk_module(path: string) -> long {
# The Vulkan side of a program handle the renderer already made (gpu_program): the handle's
# manifest key finds the variant. Returns false when there is no such variant.
# a program whose first stage is a mesh shader (its first file is *.mesh): its pipeline has no vertex
# input, and its first stage's bindings are the mesh stage's
var gvk_prog_mesh: []int = null
function gvk_stage_first(p: int) -> int {
if gvk_prog_mesh != null and p < len(gvk_prog_mesh) and gvk_prog_mesh[p] == 1 { return VK_SHADER_STAGE_MESH_BIT_EXT }
return VK_SHADER_STAGE_VERTEX_BIT
}
function gvk_program(p: int, key: string, spv_dir: string) -> bool {
let zero: long = 0
if gvk_prog_var == null {
gvk_prog_var = new []GpuVariant; gvk_prog_vs = new []long; gvk_prog_fs = new []long
gvk_prog_dsl = new []long; gvk_prog_layout = new []long
gvk_prog_dsl = new []long; gvk_prog_layout = new []long; gvk_prog_mesh = new []int
}
while len(gvk_prog_var) <= p {
push(gvk_prog_var, null); push(gvk_prog_vs, zero); push(gvk_prog_fs, zero); push(gvk_prog_dsl, zero); push(gvk_prog_layout, zero)
push(gvk_prog_var, null); push(gvk_prog_vs, zero); push(gvk_prog_fs, zero); push(gvk_prog_dsl, zero); push(gvk_prog_layout, zero); push(gvk_prog_mesh, 0)
}
let parts = Text.split(key, "|")
var defs = ""
@ -65,6 +72,7 @@ function gvk_program(p: int, key: string, spv_dir: string) -> bool {
let v = gpu_variant_find_key(parts[0], parts[1], defs)
if v == null { print(`r3d: vulkan: no SPIR-V variant for {key}`); return false }
let vs = gvk_module(`{spv_dir}/{v.id}.vert.spv`)
if Text.ends_with(v.vs, ".mesh") { gvk_prog_mesh[p] = 1 } else { gvk_prog_mesh[p] = 0 }
let fs = gvk_module(`{spv_dir}/{v.id}.frag.spv`)
if vs == 0 or fs == 0 { return false }
@ -80,7 +88,7 @@ function gvk_program(p: int, key: string, spv_dir: string) -> bool {
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, 0)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_VERTEX_BIT)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, gvk_stage_first(p))
k += 1
}
if v.fblock >= 0 {
@ -94,7 +102,7 @@ function gvk_program(p: int, key: string, spv_dir: string) -> bool {
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, v.t_bind[t])
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT)
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, gvk_stage_first(p) | VK_SHADER_STAGE_FRAGMENT_BIT)
k += 1
}
let dslci = bytes(VkDescriptorSetLayoutCreateInfo_sizeof)
@ -371,7 +379,7 @@ function gvk_pipeline(p: int, m: Mesh, st: GvkState, n_color: int, color_fmt: in
let stages = bytes(ss * 2)
Vk.zero(stages, ss * 2)
Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_stage, VK_SHADER_STAGE_VERTEX_BIT)
Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_stage, gvk_stage_first(p))
Vk.put_i64(stages, VkPipelineShaderStageCreateInfo_module, gvk_prog_vs[p])
Vk.put_ptr(stages, VkPipelineShaderStageCreateInfo_pName, "main")
Vk.put_i32(stages, ss + VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
@ -524,8 +532,11 @@ function gvk_pipeline(p: int, m: Mesh, st: GvkState, n_color: int, color_fmt: in
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pNext, prci)
Vk.put_i32(gpci, VkGraphicsPipelineCreateInfo_stageCount, 2)
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pStages, stages)
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pVertexInputState, vin)
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pInputAssemblyState, ias)
# a mesh-shader pipeline has neither: the mesh stage makes its own vertices
if gvk_stage_first(p) == VK_SHADER_STAGE_VERTEX_BIT {
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pVertexInputState, vin)
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pInputAssemblyState, ias)
}
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pViewportState, vps)
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pRasterizationState, rs)
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pMultisampleState, ms)
@ -1273,6 +1284,12 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc
}
var n = count
if n == 0 and m != null { n = m.count }
if gvk_mesh_x > 0 {
# a mesh-shader dispatch (gpu_draw_mesh_tasks): the device's command, through a pointer
Vk.sl_call_piii(gvk_mesh_fn, cb, gvk_mesh_x, gvk_mesh_y, gvk_mesh_z)
if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) }
return
}
if m != null and m.ebo != 0 {
let zero: long = 0
var itype = VK_INDEX_TYPE_UINT32
@ -1461,6 +1478,16 @@ var gvk_ind_off: int = 0
var gvk_ind_n: int = 0
var gvk_ind_cbuf: int = 0
var gvk_ind_coff: int = 0
# x * y * z mesh-shader invocations with the current program (a *.mesh one) and state
var gvk_mesh_x: int = 0
var gvk_mesh_y: int = 0
var gvk_mesh_z: int = 0
function gvk_draw_mesh_tasks_now(x: int, y: int, z: int) -> void {
if not gvk_has_mesh or gvk_mesh_fn == null or x <= 0 or y <= 0 or z <= 0 { return }
gvk_mesh_x = x; gvk_mesh_y = y; gvk_mesh_z = z
gvk_draw_now(null, 0, 0, 1)
gvk_mesh_x = 0
}
function gvk_draw_indirect_now(m: Mesh, cmds: int, offset: int, n: int, count_buf: int, count_off: int) -> void {
if m == null or m.ebo == 0 or cmds <= 0 or n <= 0 { return }
gvk_ind_buf = cmds; gvk_ind_off = offset; gvk_ind_n = n; gvk_ind_cbuf = count_buf; gvk_ind_coff = count_off

View file

@ -35,6 +35,17 @@ var grass_cmds: int = 0
var grass_band_start: words = null # per band: its first record
var grass_band_cells: words = null # ... and its 16 m cells per tile side
var grass_band_n: int = 0
# Mesh-shader grass (Settings, Video, Advanced): the chunked path's records, but each chunk is one
# mesh dispatch - work group y a tile, x a batch of GRASS_MESH_BLADES of its blades - so a blade the
# tests reject emits nothing instead of eight degenerate vertices. R3D_MESH_GRASS=1 / 0 overrides.
const GRASS_MESH_BLADES: int = 16
var grass_mesh_on: bool = false
var grass_mesh_prog: int = 0
function r3d_mesh_grass(on: bool) -> void {
grass_mesh_on = on
if Os.has_env("R3D_MESH_GRASS") { grass_mesh_on = Text.to_int(Os.env("R3D_MESH_GRASS")) != 0 }
}
function grass_mesh_live() -> bool { return grass_mesh_on and grass_merge and grass_mesh_prog != 0 and gpu_has_mesh() }
# a blade: `rows` rows of 2 vertices (x across, y along, z bend), attribute 2 = uv
function grass_blade_mesh(rows: int) -> Mesh {
@ -81,6 +92,7 @@ function grass_init() -> void {
grass_cmds = gpu_buffer_new()
}
grass_prog = r3d_program("grass.vert", "model.frag", defs)
if grass_merge and gpu_has_mesh() { grass_mesh_prog = r3d_program("grass.mesh", "model.frag", "#define FOLIAGE\n#define BLADE\n#define MESH\n") }
grass_mesh = grass_blade_mesh(4)
grass_wind = fl(2.4)
grass_s0 = fl(0.11)
@ -155,7 +167,8 @@ function grass_tiles(size: int, d_min: int, d_max: int) -> void {
function grass_draw() -> void {
if not grass_on or ter_reflect or grass_prog == 0 { return }
let p = grass_prog
var p = grass_prog
if grass_mesh_live() { p = grass_mesh_prog }
gpu_use_program(p)
u_mat4(gpu_uniform(p, "u_view"), cam_view)
u_mat4(gpu_uniform(p, "u_proj"), cam_proj)
@ -203,8 +216,9 @@ function grass_draw() -> void {
# the records gathered this frame: uploaded once, then each band GRASS_CHUNK records a draw
function grass_flush() -> void {
if grass_n == 0 { return }
let p = grass_prog
gpu_buffer_upload(grass_cmds, grass_n * 20, grass_rec, GPU_DYNAMIC)
var p = grass_prog
let mesh = grass_mesh_live()
if mesh { p = grass_mesh_prog } else { gpu_buffer_upload(grass_cmds, grass_n * 20, grass_rec, GPU_DYNAMIC) }
for b in 0 .. grass_band_n {
let s = grass_band_start[b]
var e = grass_n
@ -216,7 +230,16 @@ function grass_flush() -> void {
if m > GRASS_CHUNK { m = GRASS_CHUNK }
for q in 0 .. m * 4 { grass_chunk_tv[q] = grass_tv[k * 4 + q] }
u_f4v(gpu_uniform(p, "u_tiles"), m, grass_chunk_tv)
gpu_draw_mesh_indirect(grass_mesh, grass_cmds, k * 20, m, 0, 0)
if mesh {
# enough blade batches for the chunk's largest tile, capped as the instanced path caps a tile
let cells = grass_band_cells[b]
var most = 0
for q in 0 .. m { let t = f_to_int(grass_tv[(k + q) * 4 + 2]) * cells * cells; if t > most { most = t } }
if most > 65535 { most = 65535 }
gpu_draw_mesh_tasks((most + GRASS_MESH_BLADES - 1) / GRASS_MESH_BLADES, m, 1)
} else {
gpu_draw_mesh_indirect(grass_mesh, grass_cmds, k * 20, m, 0, 0)
}
grass_draws += 1
k += m
}

View file

@ -0,0 +1,167 @@
// Procedural ground-cover blades as a MESH shader (Vulkan, VK_EXT_mesh_shader): the same blades
// grass.vert draws - the same places, density, thinning, culling, size, sway and lighting normal -
// but a blade the tests reject emits nothing, where the instanced path still runs all eight of its
// vertices to a degenerate position.
//
// One dispatch covers a chunk of tiles (grass.ludic): work group y is the tile's place in u_tiles,
// work group x a batch of BLADES blade indices within that tile. Everything past the tile's own
// count is skipped.
layout(local_size_x = 1, local_size_y = 1, local_size_z = 1) in;
const int BLADES = 16; // blades an invocation may emit
const int ROWS = 4; // grass_blade_mesh(4): two vertices a row, three quads
layout(triangles, max_vertices = 128, max_primitives = 96) out;
uniform mat4 u_view;
uniform mat4 u_proj;
uniform mat4 u_vp;
uniform vec3 u_cam_pos;
uniform float u_time;
uniform sampler2D u_ts_height;
uniform vec2 u_ts_origin;
uniform float u_ts_half;
uniform sampler2D u_ortho;
uniform float u_ortho_on;
uniform float u_lake_level;
uniform float u_sea_level;
uniform vec4 u_lake;
uniform float u_snow_line;
uniform float u_wind;
uniform int u_tile_cells; // 16 m cells per tile side
uniform vec4 u_tiles[256]; // per tile of the dispatch: corner x, corner z, indices per cell
uniform float u_s0;
uniform float u_d0;
uniform float u_radius;
uniform int u_dbg;
out vec3 v_wpos[];
out vec3 v_nrm[];
out vec2 v_uv[];
out float v_seed[];
out vec2 v_rot[];
out float v_hull[];
const float CELL = 16.0;
float hash1(vec2 p) { return fract(sin(dot(p, vec2(127.1, 311.7))) * 43758.5453123); }
uint pcg(uint v) { uint s = v * 747796405u + 2891336453u; uint w = ((s >> ((s >> 28u) + 4u)) ^ s) * 277803737u; return (w >> 22u) ^ w; }
float bladeHash(ivec2 cell, int j, int k) {
uint h = pcg(uint(cell.x + 32768) * 73856093u ^ uint(cell.y + 32768) * 19349663u ^ uint(j) * 83492791u ^ uint(k) * 2654435761u);
return float(h) * (1.0 / 4294967295.0);
}
// no implicit level of detail outside a fragment shader: every lookup names level 0
float heightSmooth(sampler2D tex, vec2 uv) {
vec2 res = vec2(textureSize(tex, 0));
vec2 t = uv * res - 0.5;
vec2 f = fract(t);
vec2 i = floor(t);
vec2 w0 = (1.0 - f) * (1.0 - f) * (1.0 - f) / 6.0;
vec2 w1 = (4.0 - 6.0 * f * f + 3.0 * f * f * f) / 6.0;
vec2 w3 = f * f * f / 6.0;
vec2 w2 = 1.0 - w0 - w1 - w3;
vec2 s0 = w0 + w1, s1 = w2 + w3;
vec2 o0 = (i - 1.0 + w1 / s0 + 0.5) / res;
vec2 o1 = (i + 1.0 + w3 / s1 + 0.5) / res;
return (textureLod(tex, vec2(o0.x, o0.y), 0.0).r * s0.x + textureLod(tex, vec2(o1.x, o0.y), 0.0).r * s1.x) * s0.y
+ (textureLod(tex, vec2(o0.x, o1.y), 0.0).r * s0.x + textureLod(tex, vec2(o1.x, o1.y), 0.0).r * s1.x) * s1.y;
}
void main() {
int ti = int(gl_WorkGroupID.y);
vec2 tile = u_tiles[ti].xy;
int per_cell = int(u_tiles[ti].z + 0.5);
int total = min(per_cell * u_tile_cells * u_tile_cells, 65535); // the instanced path's cap on a tile
int first = int(gl_WorkGroupID.x) * BLADES;
int nv = 0;
int np = 0;
for (int b = 0; b < BLADES; b++) {
int i = first + b;
if (per_cell <= 0 || i >= total) break;
int c = i / per_cell;
int j = i - c * per_cell;
vec2 cell = tile + vec2(float(c % u_tile_cells), float(c / u_tile_cells)) * CELL;
vec2 cid = floor(cell / CELL + 0.5);
ivec2 ci = ivec2(cid);
float fj = float(j);
vec2 hv = vec2(bladeHash(ci, j, 0), bladeHash(ci, j, 1));
vec2 xz = cell + hv * CELL;
float dist = length(xz - u_cam_pos.xz);
if (dist >= u_radius) continue;
float spacing = u_s0 * (1.0 + dist / u_d0);
float count = CELL * CELL / (spacing * spacing) * (1.0 - smoothstep(u_radius * 0.7, u_radius, dist));
if (fj >= count) continue;
float life = 1.0 - smoothstep(0.8, 1.0, fj / max(count, 1.0));
vec2 huv = (xz - u_ts_origin) / (2.0 * u_ts_half) + 0.5;
if (huv.x < 0.0 || huv.x > 1.0 || huv.y < 0.0 || huv.y > 1.0) continue;
vec4 ht = textureLod(u_ts_height, huv, 0.0);
vec4 croot = u_vp * vec4(xz.x, ht.r, xz.y, 1.0);
if (croot.w < -1.0 || abs(croot.x) > croot.w * 1.25 + 1.5 || abs(croot.y) > croot.w * 1.4 + 1.5) continue;
vec3 gn = normalize(ht.gba);
float h3 = bladeHash(ci, j, 2), h4 = bladeHash(ci, j, 3);
float wl = u_sea_level;
if (u_lake.z > 0.0) { vec2 q = (xz - u_lake.xy) / u_lake.zw; if (dot(q, q) < 1.0) wl = max(wl, u_lake_level); }
float ok = (1.0 - smoothstep(0.30, 0.55, 1.0 - gn.y)) * smoothstep(0.0, 0.6, ht.r - wl - 0.15) * smoothstep(u_snow_line - 80.0, u_snow_line - 200.0, ht.r);
if (u_ortho_on > 0.5) {
vec3 oc = textureLod(u_ortho, huv, 1.5).rgb;
ok *= 0.25 + 0.75 * smoothstep(0.0, 0.02, oc.g - max(oc.r, oc.b));
}
if (h4 > ok) continue;
bool far = dist > 300.0;
float h = far ? ht.r : heightSmooth(u_ts_height, huv);
if (far) h += 0.03;
if ((u_dbg & 1) != 0) h += 0.3;
float seed = hv.x * 0.7 + hv.y * 0.3;
float ang = hv.y * 6.2831853;
float s = sin(ang), c_ = cos(ang);
float grow = spacing / u_s0;
float tall = mix(0.18, 0.42, h3) * mix(0.8, 1.2, hash1(cid * 0.1)) * (1.0 + 0.35 * smoothstep(1.0, 12.0, grow)) * life;
float bw = 0.028 * mix(1.0, 0.45 * grow, smoothstep(1.0, 4.0, grow));
if (far) { bw = max(bw, spacing * 0.35); tall = min(tall, spacing * 0.3); }
float gust = sin(xz.x * 0.09 + u_time * 1.1) * 0.5 + sin(xz.y * 0.13 - u_time * 0.8 + xz.x * 0.05) * 0.5;
float ph = u_time * 1.7 + seed * 6.2831 + xz.x * 0.05 + xz.y * 0.07;
float sway = (sin(ph) * 0.6 + sin(ph * 2.3 + 1.0) * 0.4 + gust) * u_wind;
// the ground's frame, shared by the blade's vertices
vec3 up = vec3(0.0, 1.0, 0.0);
vec3 k = cross(up, gn);
float sk = length(k), ck = gn.y;
bool tilt = sk > 1e-4;
if (tilt) k /= sk;
vec3 n = vec3(0.0, 0.3, 1.0);
n = normalize(vec3(c_ * n.x + s * n.z, n.y, -s * n.x + c_ * n.z));
if (tilt) n = normalize(n * ck + cross(k, n) * sk + k * dot(k, n) * (1.0 - ck));
n = normalize(mix(n, gn, smoothstep(2.0, 12.0, dist)));
float hull = (dist > 2.0 || far || (u_dbg & 2) != 0) ? -1.0 : 1.0;
// the blade: grass_blade_mesh(4)'s vertices, placed as grass.vert places them
for (int r = 0; r < ROWS; r++) {
float t = float(r) / float(ROWS - 1);
float taper = max(1.0 - t * t * sqrt(t), 0.12);
float bend = t * t * 0.28;
for (int sd = 0; sd < 2; sd++) {
vec3 a_pos = vec3((float(sd) - 0.5) * taper, t, bend);
vec2 a_uv = vec2(float(sd), t);
vec3 p = vec3(a_pos.x * bw, a_pos.y * tall, a_pos.z * tall * (0.6 + 0.8 * h4));
float hgt = max(p.y, 0.0);
p.x += sway * hgt * hgt * 0.35;
p.z += sway * hgt * hgt * 0.15 * cos(ph * 0.7);
p = vec3(c_ * p.x + s * p.z, p.y, -s * p.x + c_ * p.z);
if (tilt) p = p * ck + cross(k, p) * sk + k * dot(k, p) * (1.0 - ck);
vec3 w = vec3(xz.x, h - 0.02, xz.y) + p;
vec4 clip = u_proj * u_view * vec4(w, 1.0);
clip.z = (clip.z + clip.w) * 0.5; // OpenGL's depth range to Vulkan's, as the vertex wrapper does
int o = nv + r * 2 + sd;
gl_MeshVerticesEXT[o].gl_Position = clip;
v_wpos[o] = w;
v_nrm[o] = n;
v_uv[o] = far ? vec2(a_uv.x, 0.45 + 0.2 * a_uv.y) : a_uv;
v_seed[o] = seed;
v_rot[o] = vec2(s, c_);
v_hull[o] = hull;
}
}
for (int q = 0; q < ROWS - 1; q++) {
uint bb = uint(nv + q * 2);
gl_PrimitiveTriangleIndicesEXT[np] = uvec3(bb, bb + 1u, bb + 2u);
gl_PrimitiveTriangleIndicesEXT[np + 1] = uvec3(bb + 1u, bb + 3u, bb + 2u);
np += 2;
}
nv += ROWS * 2;
}
SetMeshOutputsEXT(uint(nv), uint(np));
}

Binary file not shown.

Binary file not shown.

View file

@ -628,6 +628,73 @@ T a4683767 u_prefilter 8
T a4683767 u_shadow 9
T a4683767 u_tershadow 10
T a4683767 u_ts_height 11
P d48d6a48 grass.mesh model.frag #define FOLIAGE;#define BLADE;#define MESH;
B d48d6a48 vert 0 4384
U d48d6a48 vert u_view 0 mat4 1 0
U d48d6a48 vert u_proj 64 mat4 1 0
U d48d6a48 vert u_vp 128 mat4 1 0
U d48d6a48 vert u_cam_pos 192 vec3 1 0
U d48d6a48 vert u_time 204 float 1 0
U d48d6a48 vert u_ts_origin 208 vec2 1 0
U d48d6a48 vert u_ts_half 216 float 1 0
U d48d6a48 vert u_ortho_on 220 float 1 0
U d48d6a48 vert u_lake_level 224 float 1 0
U d48d6a48 vert u_sea_level 228 float 1 0
U d48d6a48 vert u_lake 240 vec4 1 0
U d48d6a48 vert u_snow_line 256 float 1 0
U d48d6a48 vert u_wind 260 float 1 0
U d48d6a48 vert u_tile_cells 264 int 1 0
U d48d6a48 vert u_tiles 272 vec4 256 16
U d48d6a48 vert u_s0 4368 float 1 0
U d48d6a48 vert u_d0 4372 float 1 0
U d48d6a48 vert u_radius 4376 float 1 0
U d48d6a48 vert u_dbg 4380 int 1 0
B d48d6a48 frag 1 892
U d48d6a48 frag u_cascade_vp 0 mat4 5 64
U d48d6a48 frag u_cascade_split 320 float 5 16
U d48d6a48 frag u_cascade_range 400 float 5 16
U d48d6a48 frag u_cascade_texel 480 float 5 16
U d48d6a48 frag u_sun_dir 560 vec3 1 0
U d48d6a48 frag u_sun_color 576 vec3 1 0
U d48d6a48 frag u_cam_pos 592 vec3 1 0
U d48d6a48 frag u_prefilter_levels 604 float 1 0
U d48d6a48 frag u_fog_density 608 float 1 0
U d48d6a48 frag u_fog_height_falloff 612 float 1 0
U d48d6a48 frag u_fog_base 616 float 1 0
U d48d6a48 frag u_clip_y 620 float 1 0
U d48d6a48 frag u_spec_scale 624 float 1 0
U d48d6a48 frag u_sky_rot 632 vec2 1 0
U d48d6a48 frag u_ibl_scale 640 vec3 1 0
U d48d6a48 frag u_daylight 652 float 1 0
U d48d6a48 frag u_fire_pos 656 vec3 1 0
U d48d6a48 frag u_fire_color 672 vec3 1 0
U d48d6a48 frag u_hand_pos 688 vec3 1 0
U d48d6a48 frag u_hand_color 704 vec3 1 0
U d48d6a48 frag u_hand_dir 720 vec3 1 0
U d48d6a48 frag u_hand_cone 732 float 1 0
U d48d6a48 frag u_ts_origin 736 vec2 1 0
U d48d6a48 frag u_ts_half 744 float 1 0
U d48d6a48 frag u_ts_on 748 float 1 0
U d48d6a48 frag u_force_cascade 752 int 1 0
U d48d6a48 frag u_cloud_shadow 756 float 1 0
U d48d6a48 frag u_time 760 float 1 0
U d48d6a48 frag u_model_h 764 float 1 0
U d48d6a48 frag u_view 768 mat4 1 0
U d48d6a48 frag u_tint 832 vec3 1 0
U d48d6a48 frag u_rough_scale 844 float 1 0
U d48d6a48 frag u_emissive 848 float 1 0
U d48d6a48 frag u_blade_base 864 vec3 1 0
U d48d6a48 frag u_blade_tip 880 vec3 1 0
T d48d6a48 u_arm 2
T d48d6a48 u_brdf 3
T d48d6a48 u_diff 4
T d48d6a48 u_irradiance 5
T d48d6a48 u_nrm 6
T d48d6a48 u_ortho 7
T d48d6a48 u_prefilter 8
T d48d6a48 u_shadow 9
T d48d6a48 u_tershadow 10
T d48d6a48 u_ts_height 11
P 7ed7de52 grass.vert model.frag #define FOLIAGE;#define BLADE;
B 7ed7de52 vert 0 296
U 7ed7de52 vert u_view 0 mat4 1 0

View file

@ -17,6 +17,7 @@ fullscreen.vert|ternormal.frag|
fullscreen.vert|tershadow.frag|#define NOISE_ONLY;
fullscreen.vert|tonemap.frag|
fullscreen.vert|tonemap.frag|#define HDR10;
grass.mesh|model.frag|#define FOLIAGE;#define BLADE;#define MESH;
grass.vert|model.frag|#define FOLIAGE;#define BLADE;
grass.vert|model.frag|#define FOLIAGE;#define BLADE;#define TILES;
impostor.vert|impostor.frag|

View file

@ -43,3 +43,6 @@ extern function vk_sl_call_p(fn: pointer, a: pointer) -> int = "lsl_call_p"
extern function vk_sl_call_pp(fn: pointer, a: pointer, b: pointer) -> int = "lsl_call_pp"
extern function vk_sl_call_ppp(fn: pointer, a: pointer, b: pointer, c: pointer) -> int = "lsl_call_ppp"
extern function vk_sl_call_ip(fn: pointer, a: int, b: pointer) -> int = "lsl_call_ip"
# a command-buffer command with three counts through a pointer (vkCmdDrawMeshTasksEXT, which the
# Streamline interposer does not export; looked up with vkGetDeviceProcAddr)
extern function vk_sl_call_piii(fn: pointer, a: pointer, b: int, c: int, d: int) = "lsl_call_piii"