perf(render3d): mesh-shader grass dispatches each tile with its own count; not yet counted as implemented
A chunk's dispatch was sized for its largest tile, so the far tiles beside a near one ran thousands of empty invocations: 9.8 ms of grass at 4K on an RTX 3070 Ti. One dispatch per tile, sized to that tile, brings it to 5.0 ms - still three times the chunked path's 1.7 (36 fps against 41), so GF_MESH_GRASS is not implemented yet and the Advanced row says a coming update. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
a868378b99
commit
d80786e99a
3 changed files with 23 additions and 11 deletions
|
|
@ -1,9 +1,12 @@
|
||||||
bump: minor
|
bump: minor
|
||||||
type: feat
|
type: feat
|
||||||
**Mesh-shader grass** — on a card with `VK_EXT_mesh_shader`, `r3d_mesh_grass(on)` draws each chunk of grass
|
**Mesh-shader grass** — on a card with `VK_EXT_mesh_shader`, `r3d_mesh_grass(on)` draws each grass tile as one
|
||||||
tiles as one mesh-shader dispatch, and a blade that is culled emits no vertices at all.
|
mesh-shader dispatch sized to that tile's blades, and a blade that is culled emits no vertices at all.
|
||||||
|
|
||||||
- **`ludic-dev shaders`** — a variant whose first file is `*.mesh` compiles that stage as a Vulkan 1.3 mesh
|
- **`ludic-dev shaders`** — a variant whose first file is `*.mesh` compiles that stage as a Vulkan 1.3 mesh
|
||||||
shader; its SPIR-V keeps the `.vert` name, so the manifest is unchanged.
|
shader; its SPIR-V keeps the `.vert` name, so the manifest is unchanged.
|
||||||
- **The seam** — `gpu_has_mesh()` and `gpu_draw_mesh_tasks(x, y, z)`; a mesh program's pipeline has no
|
- **The seam** — `gpu_has_mesh()` and `gpu_draw_mesh_tasks(x, y, z)`; a mesh program's pipeline has no
|
||||||
vertex input and its bindings the mesh stage. The device enables `meshShader` and `maintenance4`.
|
vertex input and its bindings the mesh stage. The device enables `meshShader` and `maintenance4`.
|
||||||
|
|
||||||
|
Not yet faster: at 4K on an RTX 3070 Ti the grass pass takes 5.0 ms against the chunked path's 1.7, so
|
||||||
|
`gpu_feature_implemented(GF_MESH_GRASS)` stays false and a game should not offer it as a finished setting.
|
||||||
|
|
|
||||||
|
|
@ -392,9 +392,10 @@ function gpu_caps_probe() -> void {
|
||||||
# Whether the renderer actually draws a feature yet. Until a feature lands, choosing it is saved and
|
# Whether the renderer actually draws a feature yet. Until a feature lands, choosing it is saved and
|
||||||
# shown, and says it takes effect later. The Vulkan renderer draws the whole game (phase 37-38), and
|
# shown, and says it takes effect later. The Vulkan renderer draws the whole game (phase 37-38), and
|
||||||
# DLSS super resolution and Reflex run through NVIDIA Streamline (streamline.ludic), and HDR output is an
|
# DLSS super resolution and Reflex run through NVIDIA Streamline (streamline.ludic), and HDR output is an
|
||||||
# HDR10 swapchain (gpu_vk_draw.ludic); whether this
|
# HDR10 swapchain (gpu_vk_draw.ludic). Mesh-shader grass draws but is not yet counted: at 4K on an
|
||||||
|
# RTX 3070 Ti its grass pass took 5.0 ms against the chunked path's 1.7. Whether this
|
||||||
# machine can use one is the caps' question, not this one.
|
# machine can use one is the caps' question, not this one.
|
||||||
function gpu_feature_implemented(f: int) -> bool { return f == GF_VULKAN or f == GF_DLSS or f == GF_REFLEX or f == GF_HDR or f == GF_MESH_GRASS }
|
function gpu_feature_implemented(f: int) -> bool { return f == GF_VULKAN or f == GF_DLSS or f == GF_REFLEX or f == GF_HDR }
|
||||||
|
|
||||||
# ---- vertex data --------------------------------------------------------------------
|
# ---- vertex data --------------------------------------------------------------------
|
||||||
# A Mesh is built through these and records what it is made of - which buffer feeds which
|
# A Mesh is built through these and records what it is made of - which buffer feeds which
|
||||||
|
|
|
||||||
|
|
@ -229,18 +229,26 @@ function grass_flush() -> void {
|
||||||
var m = e - k
|
var m = e - k
|
||||||
if m > GRASS_CHUNK { m = GRASS_CHUNK }
|
if m > GRASS_CHUNK { m = GRASS_CHUNK }
|
||||||
for q in 0 .. m * 4 { grass_chunk_tv[q] = grass_tv[k * 4 + q] }
|
for q in 0 .. m * 4 { grass_chunk_tv[q] = grass_tv[k * 4 + q] }
|
||||||
u_f4v(gpu_uniform(p, "u_tiles"), m, grass_chunk_tv)
|
|
||||||
if mesh {
|
if mesh {
|
||||||
# enough blade batches for the chunk's largest tile, capped as the instanced path caps a tile
|
# one dispatch per tile, as many blade batches as that tile has: sized for a chunk's largest
|
||||||
|
# tile, the far tiles beside a near one ran thousands of empty invocations (9.8 ms against the
|
||||||
|
# chunked path's 2 at 4K on an RTX 3070 Ti)
|
||||||
let cells = grass_band_cells[b]
|
let cells = grass_band_cells[b]
|
||||||
var most = 0
|
for q in 0 .. m {
|
||||||
for q in 0 .. m { let t = f_to_int(grass_tv[(k + q) * 4 + 2]) * cells * cells; if t > most { most = t } }
|
var total = f_to_int(grass_tv[(k + q) * 4 + 2]) * cells * cells
|
||||||
if most > 65535 { most = 65535 }
|
if total > 65535 { total = 65535 }
|
||||||
gpu_draw_mesh_tasks((most + GRASS_MESH_BLADES - 1) / GRASS_MESH_BLADES, m, 1)
|
if total > 0 {
|
||||||
|
for c in 0 .. 4 { grass_chunk_tv[c] = grass_tv[(k + q) * 4 + c] }
|
||||||
|
u_f4v(gpu_uniform(p, "u_tiles"), 1, grass_chunk_tv)
|
||||||
|
gpu_draw_mesh_tasks((total + GRASS_MESH_BLADES - 1) / GRASS_MESH_BLADES, 1, 1)
|
||||||
|
grass_draws += 1
|
||||||
|
}
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
|
u_f4v(gpu_uniform(p, "u_tiles"), m, grass_chunk_tv)
|
||||||
gpu_draw_mesh_indirect(grass_mesh, grass_cmds, k * 20, m, 0, 0)
|
gpu_draw_mesh_indirect(grass_mesh, grass_cmds, k * 20, m, 0, 0)
|
||||||
}
|
}
|
||||||
grass_draws += 1
|
if not mesh { grass_draws += 1 }
|
||||||
k += m
|
k += m
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue