cam_sphere_visible refuses a sphere wholly past the wall (terrain patches, scatter cells, stream
chunks, grass tiles, water), actors past it are skipped for draws, casters and outlines, the grass
reach and the streams' generate-and-gather reach end at it, and the card shadows cast only from
the wall's share of a layer (fog_casters.ludic, rebuilt every 4 m). r3d_beyond_fog is exported for
the game to skip animating what will not be drawn. Render-only: no query of the ground or the world
changes, and off is the old frame (0 pixels over 8 against 160ce96, twice per side).
Draws per frame at the overlook: 2757 off, 1104 at 175 m, 484 at 30 m.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1249 lines
64 KiB
Text
1249 lines
64 KiB
Text
# ============================================================================
|
|
# scatter.ludic — instanced vegetation and props. A Layer is one model placed
|
|
# many times (position, scale, yaw, seed, wind weight per instance). Each frame
|
|
# the instances are split by distance: the near ones draw as the full scanned
|
|
# mesh, the far ones as impostor cards baked from that mesh at load.
|
|
# ============================================================================
|
|
|
|
const INST_FLOATS: int = 8
|
|
|
|
property Impostor {
|
|
albedo: int = 0,
|
|
normal: int = 0,
|
|
tiles: int = 16,
|
|
radius: float = 0.0,
|
|
height: float = 0.0
|
|
}
|
|
property Layer {
|
|
model: Model,
|
|
imp: Impostor,
|
|
foliage: bool = false,
|
|
wind: float = 0.0, # float bits
|
|
flutter: float = 0.0, # float bits; per-leaf tremble (aspen), 0 = none
|
|
tint: floats,
|
|
inst: floats, # INST_FLOATS per instance
|
|
count: int = 0,
|
|
cap: int = 0,
|
|
have: int = 0, # instances inst and scratch have room for now: grown toward cap as filled
|
|
near: float = 0.0, # float bits; instances beyond it draw as impostors (or not at all)
|
|
cull: float = 0.0, # float bits; instances beyond it are skipped (0 = never)
|
|
buf: int = 0,
|
|
n_near: int = 0,
|
|
imp_buf: int = 0,
|
|
sh_buf: int = 0, # every instance, for shadow casting (no cull, no LOD split)
|
|
n_sh: int = 0,
|
|
fog_buf: int = 0, # the casters inside the fog wall (fog_casters.ludic)
|
|
fog_n: int = 0,
|
|
fog_src: int = -1,
|
|
fog_wall: float = 0.0,
|
|
fog_x: float = 0.0,
|
|
fog_z: float = 0.0,
|
|
n_far: int = 0,
|
|
scratch: floats,
|
|
last_cam: floats,
|
|
rough: float = 0.0,
|
|
blade: bool = false,
|
|
grass: bool = false, # ground cover the GPU blades replace: skipped while they draw (and under R3D_NOGRASS)
|
|
flower: bool = false,
|
|
card: bool = false,
|
|
cheap: bool = false, # distant cover: no shadows, no wind, flat lighting
|
|
atlas: Impostor,
|
|
streamed: bool = false, # fed by a Stream: already frustum-culled per chunk, no split needed
|
|
grounded: bool = false, # the vertex shader stands each instance on the drawn terrain
|
|
view_gen: int = -1, # sc_view_gen this layer's partition was built for
|
|
# static layers with many instances are sorted into a cell grid once, and only the
|
|
# cells inside the view frustum (and within cull) are partitioned each frame
|
|
gcell: float = 0.0, # cell size (float bits); 0 = no grid
|
|
gx0: float = 0.0,
|
|
gz0: float = 0.0,
|
|
gnx: int = 0,
|
|
gnz: int = 0,
|
|
gstart: words, # per cell: first index into gsorted (ncell + 1 entries)
|
|
gsorted: floats, # the instances, grouped by cell
|
|
gymin: floats, # per cell height range (float bits)
|
|
gymax: floats,
|
|
vis: floats, # the instances gathered from visible cells this frame
|
|
n_vis: int = 0,
|
|
# A LOD chain: lods[k] is drawn for instances within lod_dist[k] (and beyond lod_dist[k-1]);
|
|
# past the last level the impostor takes over (or, if the last distance is 0, the last
|
|
# level runs out to the cull distance). lod_card[k] = 1 marks a level that is the layer's
|
|
# crossed card carrying its atlas (cover keeps its baked card as the far level).
|
|
lods: []Model,
|
|
n_lods: int = 0,
|
|
# The GPU-culled path (Vulkan, phase 38): every instance in g_src, a compute pass packs the
|
|
# visible ones per bucket into g_dst and writes the instance counts of the draw records in
|
|
# g_cmds (see scatter_cull.comp for the record layout). g_on once it is set up for g_n instances.
|
|
g_on: bool = false,
|
|
g_n: int = 0,
|
|
g_src: int = 0,
|
|
g_dst: int = 0,
|
|
g_cmds: int = 0,
|
|
g_counts: int = 0,
|
|
g_arena: []Prim, # one merged mesh per material, holding every level's copy (layer_arena_build)
|
|
g_first: words, # material * 4 + level: that level's first index in the merged mesh
|
|
g_base: words, # ... and its first vertex
|
|
g_model: Model, # the merged meshes as a model the draws take (level 0's height)
|
|
g_model_sh: Model, # the same with level 2's height, for the shadow LOD
|
|
lod_dist: floats,
|
|
lod_card: words,
|
|
lod_buf: words,
|
|
n_lod: words,
|
|
lvl: words # scratch: the level chosen per gathered instance
|
|
}
|
|
|
|
# the procedural models' one layout: position, normal, uv, interleaved at 32 bytes
|
|
function sc_model_layout(render3d_st: Render3dState, m: Mesh) -> void {
|
|
gpu_mesh_attr(render3d_st, m, 0, 3, GPU_F32, 32, 0, false)
|
|
gpu_mesh_attr(render3d_st, m, 1, 3, GPU_F32, 32, 12, false)
|
|
gpu_mesh_attr(render3d_st, m, 2, 2, GPU_F32, 32, 24, false)
|
|
}
|
|
|
|
# Two crossed unit quads (x in [-0.5, 0.5], y in [0, 1]), attribute 0 = pos,
|
|
# 1 = the quad's facing normal, 2 = uv. Scaled per layer to the atlas card size.
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function model_cross_card(render3d_st: mut Render3dState) -> Model {
|
|
let model = new Model
|
|
model.prims = new []Prim
|
|
let pr = new Prim
|
|
let m = gpu_mesh_new(render3d_st)
|
|
let v = gl_floats(8 * 8)
|
|
var k = 0
|
|
for q in 0 .. 2 {
|
|
for c in 0 .. 4 {
|
|
var sx = -0.5; var sy = 0.0; var u = 0.0; var vv = 0.0
|
|
if c == 1 or c == 2 { sx = 0.5; u = 1.0 }
|
|
if c == 2 or c == 3 { sy = 1.0; vv = 1.0 }
|
|
if q == 0 { gl_put_bits(v, k, float_bits(sx)); gl_put_bits(v, k + 1, float_bits(sy)); gl_put_bits(v, k + 2, float_bits(0.0)); gl_put_bits(v, k + 3, float_bits(0.0)); gl_put_bits(v, k + 4, float_bits(0.0)); gl_put_bits(v, k + 5, float_bits(-1.0)) }
|
|
else { gl_put_bits(v, k, float_bits(0.0)); gl_put_bits(v, k + 1, float_bits(sy)); gl_put_bits(v, k + 2, float_bits(sx)); gl_put_bits(v, k + 3, float_bits(-1.0)); gl_put_bits(v, k + 4, float_bits(0.0)); gl_put_bits(v, k + 5, float_bits(0.0)) }
|
|
gl_put_bits(v, k + 6, float_bits(u)); gl_put_bits(v, k + 7, float_bits(vv))
|
|
k += 8
|
|
}
|
|
}
|
|
gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(64), GPU_STATIC)
|
|
sc_model_layout(render3d_st, m)
|
|
free(v)
|
|
let idx = words(12)
|
|
idx[0] = 0; idx[1] = 1; idx[2] = 2; idx[3] = 0; idx[4] = 2; idx[5] = 3
|
|
idx[6] = 4; idx[7] = 5; idx[8] = 6; idx[9] = 4; idx[10] = 6; idx[11] = 7
|
|
gpu_mesh_indices(render3d_st, m, data_of(idx), 48, 4)
|
|
free(idx)
|
|
m.count = 12
|
|
gpu_mesh_done(render3d_st, m)
|
|
pr.mesh = m
|
|
if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) }
|
|
pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white
|
|
push(model.prims, pr)
|
|
model.radius = 0.5; model.height = 1.0; model.tris = 4
|
|
return model
|
|
}
|
|
|
|
# A card layer: crossed cards carrying a single-tile atlas baked from `scan`.
|
|
# An aspen leaf hangs on a flattened stalk and turns in air nothing else feels. Give the
|
|
# layer a flutter and its leaves tremble and flash their pale undersides; everything else
|
|
# leaves it at zero. It is per layer rather than per instance because a species quakes or
|
|
# it does not.
|
|
function layer_flutter(l: Layer, v: float) -> void { l.flutter = v }
|
|
function layer_cards(render3d_st: mut Render3dState, scan: Model, cap: int, wind: float, cull: float) -> Layer {
|
|
let l = layer_new(render3d_st, model_cross_card(render3d_st), cap, true, wind, 0.0, cull)
|
|
l.card = true
|
|
l.atlas = impostor_bake(render3d_st, scan, 1, 512, 512)
|
|
return l
|
|
}
|
|
|
|
# A procedural grass blade (1 m tall, 5 cm wide, curved): 5 rows of 2 vertices.
|
|
@alloc_ok("a model: made once by the program that asks for it, and kept with its layer")
|
|
function model_blade(render3d_st: mut Render3dState) -> Model {
|
|
let model = new Model
|
|
model.prims = new []Prim
|
|
let pr = new Prim
|
|
let m = gpu_mesh_new(render3d_st)
|
|
let rows = 5
|
|
let v = gl_floats(rows * 2 * 8)
|
|
var k = 0
|
|
for r in 0 .. rows {
|
|
let t = float(r) / float(rows - 1)
|
|
# never a zero-width tip: a sliver triangle extrapolates its attributes wildly
|
|
let taper = Math.max(1.0 - t * (t * Math.sqrt(t)), 0.12)
|
|
let hw = 0.05 * taper
|
|
let bend = t * t * 0.28
|
|
for sd in 0 .. 2 {
|
|
var x = -hw
|
|
if sd == 1 { x = hw }
|
|
gl_put_bits(v, k, float_bits(x)); gl_put_bits(v, k + 1, float_bits(t)); gl_put_bits(v, k + 2, float_bits(bend))
|
|
gl_put_bits(v, k + 3, float_bits(0.0)); gl_put_bits(v, k + 4, float_bits(0.3)); gl_put_bits(v, k + 5, float_bits(1.0))
|
|
gl_put_bits(v, k + 6, float_bits(float(sd))); gl_put_bits(v, k + 7, float_bits(t))
|
|
k += 8
|
|
}
|
|
}
|
|
gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(rows * 2 * 8), GPU_STATIC)
|
|
sc_model_layout(render3d_st, m)
|
|
free(v)
|
|
let ni = (rows - 1) * 6
|
|
let idx = words(ni)
|
|
k = 0
|
|
for r in 0 .. rows - 1 {
|
|
let a = r * 2
|
|
idx[k] = a; idx[k + 1] = a + 1; idx[k + 2] = a + 2
|
|
idx[k + 3] = a + 1; idx[k + 4] = a + 3; idx[k + 5] = a + 2
|
|
k += 6
|
|
}
|
|
gpu_mesh_indices(render3d_st, m, data_of(idx), ni * 4, 4)
|
|
free(idx)
|
|
m.count = ni
|
|
gpu_mesh_done(render3d_st, m)
|
|
pr.mesh = m
|
|
if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) }
|
|
pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white
|
|
push(model.prims, pr)
|
|
model.radius = 0.05; model.height = 1.0; model.tris = ni / 3
|
|
return model
|
|
}
|
|
|
|
# The foliage depth prepass (render.ludic): the near tree LODs write depth first with a
|
|
# shader that only runs the alpha test, then the lit pass shades them with no discard and
|
|
# an equal depth test, so a pixel of needles is lit once rather than once for every card
|
|
# stacked behind it. In a dense stand at 4K that overdraw was the largest pass in the frame.
|
|
|
|
function scatter_init(render3d_st: mut Render3dState) -> void {
|
|
# R3D_DUMP_ATLAS: every impostor and card atlas the run bakes, to build/atlas_<n>_{color,alpha}.ppm
|
|
render3d_st.sc_debug_dump = r3d_env_has(render3d_st, "R3D_DUMP_ATLAS")
|
|
render3d_st.sc_prog = r3d_program(render3d_st, "model.vert", "model.frag", "")
|
|
render3d_st.sc_prog_fol = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define ALPHA_TEST\n#define NEAR_FADE\n")
|
|
render3d_st.sc_prog_fol_depth = r3d_program(render3d_st, "model.vert", "depth.frag", "#define FOLIAGE\n#define WIND\n#define ALPHA_TEST\n#define NEAR_FADE\n")
|
|
render3d_st.sc_prog_fol_eq = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define EQ_PASS\n#define NEAR_FADE\n")
|
|
render3d_st.sc_prog_wind = r3d_program(render3d_st, "model.vert", "model.frag", "#define WIND\n")
|
|
render3d_st.sc_prog_blade = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define BLADE\n")
|
|
render3d_st.sc_prog_flower = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define FLOWER\n")
|
|
render3d_st.sc_prog_card = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define CARD\n")
|
|
render3d_st.sc_prog_card_shadow = r3d_program(render3d_st, "model.vert", "model.frag", "#define SHADOW_PASS\n#define WIND\n#define CARD\n")
|
|
render3d_st.sc_prog_card_cheap = r3d_program(render3d_st, "model.vert", "model.frag", "#define CARD\n#define CHEAP\n")
|
|
# a dry alpine meadow: brown-olive roots, straw with a little green at the tips
|
|
render3d_st.sc_blade_base = v3_new(0.045, 0.06, 0.025)
|
|
render3d_st.sc_blade_tip = v3_new(0.22, 0.27, 0.13)
|
|
render3d_st.sc_blade_tint = v3_new(1.0, 1.0, 1.0)
|
|
render3d_st.sc_prog_shadow = r3d_program(render3d_st, "model.vert", "shadow.frag", "#define SHADOW_PASS\n")
|
|
render3d_st.sc_prog_shadow_wind = r3d_program(render3d_st, "model.vert", "shadow.frag", "#define SHADOW_PASS\n#define WIND\n")
|
|
render3d_st.sc_prog_shadow_fol = r3d_program(render3d_st, "model.vert", "shadow.frag", "#define SHADOW_PASS\n#define WIND\n#define ALPHA_TEST\n")
|
|
render3d_st.sc_imp_prog = r3d_program(render3d_st, "impostor.vert", "impostor.frag", "")
|
|
render3d_st.sc_imp_prog_shadow = r3d_program(render3d_st, "impostor.vert", "impostor.frag", "#define SHADOW_PASS\n")
|
|
render3d_st.sc_bake_prog = r3d_program(render3d_st, "model.vert", "bake.frag", "")
|
|
render3d_st.sc_bake_flower_prog = r3d_program(render3d_st, "model.vert", "bake.frag", "#define FLOWER\n")
|
|
render3d_st.sc_bake_card_prog = r3d_program(render3d_st, "model.vert", "bake.frag", "#define CARD\n")
|
|
render3d_st.sc_card = mesh_card(render3d_st)
|
|
# a single identity instance, for baking
|
|
let one = gl_floats(INST_FLOATS)
|
|
for i in 0 .. INST_FLOATS { gl_put_bits(one, i, float_bits(0.0)) }
|
|
gl_put_bits(one, 3, float_bits(1.0)); gl_put_bits(one, 5, float_bits(1.0))
|
|
render3d_st.sc_ident_buf = gpu_buffer_new(render3d_st)
|
|
gpu_buffer_upload(render3d_st, render3d_st.sc_ident_buf, INST_FLOATS * 4, one, GPU_STATIC)
|
|
free(one)
|
|
render3d_st.sc_layers = new []Layer
|
|
}
|
|
|
|
# feed a mesh its instances from `buf`: attribute 3 = position + scale, 4 = sin, cos, seed, wind
|
|
function scatter_attach(render3d_st: Render3dState, m: Mesh, buf: int) -> void {
|
|
gpu_mesh_bind_instances(render3d_st, m, buf)
|
|
gpu_mesh_attr_inst(render3d_st, m, 3, 4, GPU_F32, INST_FLOATS * 4, 0)
|
|
gpu_mesh_attr_inst(render3d_st, m, 4, 4, GPU_F32, INST_FLOATS * 4, 16)
|
|
gpu_mesh_done(render3d_st, m)
|
|
}
|
|
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function layer_new(render3d_st: mut Render3dState, model: Model, cap: int, foliage: bool, wind: float, near: float, cull: float) -> Layer {
|
|
let l = new Layer
|
|
l.model = model
|
|
l.cap = cap
|
|
l.foliage = foliage
|
|
l.wind = wind
|
|
l.near = near
|
|
l.cull = cull
|
|
l.tint = v3_new(1.0, 1.0, 1.0)
|
|
# the whole capacity up front: a game writes l.inst directly (Maroon Lake's prints, trees and
|
|
# rocks do), so a layer cannot start smaller than it may be written to
|
|
l.have = cap
|
|
l.inst = floats(l.have * INST_FLOATS)
|
|
l.scratch = floats(l.have * INST_FLOATS)
|
|
l.last_cam = v3_new(100000.0, 0.0, 0.0)
|
|
l.buf = gpu_buffer_new(render3d_st)
|
|
l.imp_buf = gpu_buffer_new(render3d_st)
|
|
l.sh_buf = gpu_buffer_new(render3d_st)
|
|
l.fog_buf = gpu_buffer_new(render3d_st)
|
|
l.rough = 1.0
|
|
for i in 0 .. len(model.prims) { scatter_attach(render3d_st, model.prims[i].mesh, l.buf) }
|
|
push(render3d_st.sc_layers, l)
|
|
return l
|
|
}
|
|
|
|
# room for n instances (at most the layer's cap), doubling what there is so a fill costs a few copies;
|
|
# a caller writing l.inst itself reserves first
|
|
function layer_reserve(l: Layer, n: int) -> void { layer_room(l, n) }
|
|
@alloc_ok("a layer outgrowing its capacity, grown once to what it now holds (rare, and bounded by the scene)")
|
|
function layer_room(l: Layer, n: int) -> void {
|
|
if n <= l.have { return }
|
|
var want = max(l.have * 2, n)
|
|
if want > l.cap { want = l.cap }
|
|
let inst = floats(want * INST_FLOATS)
|
|
if l.count > 0 { mem_copy(data_of(inst), data_of(l.inst), l.count * INST_FLOATS * 4) }
|
|
free(l.inst)
|
|
free(l.scratch)
|
|
l.inst = inst
|
|
l.scratch = floats(want * INST_FLOATS)
|
|
l.have = want
|
|
}
|
|
|
|
function layer_add(l: Layer, x: float, y: float, z: float, scale: float, yaw: float, seed: float, wind: float) -> void {
|
|
if l.count >= l.cap { return }
|
|
layer_room(l, l.count + 1)
|
|
let o = l.count * INST_FLOATS
|
|
l.inst[o] = x; l.inst[o + 1] = y; l.inst[o + 2] = z; l.inst[o + 3] = scale
|
|
l.inst[o + 4] = Math.sin(yaw); l.inst[o + 5] = Math.cos(yaw); l.inst[o + 6] = seed; l.inst[o + 7] = wind
|
|
l.count += 1
|
|
}
|
|
|
|
# ---- impostors ---------------------------------------------------------------------
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function impostor_bake(render3d_st: mut Render3dState, model: Model, tiles: int, tw: int, th: int) -> Impostor {
|
|
let im = new Impostor
|
|
im.tiles = tiles
|
|
im.radius = model.radius * 1.02
|
|
im.height = model.height
|
|
let aw = tiles * tw
|
|
im.albedo = tex_target(render3d_st, aw, th, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
im.normal = tex_target(render3d_st, aw, th, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
let fbo = gpu_fb_new(render3d_st)
|
|
gpu_fb_bind(render3d_st, fbo)
|
|
gpu_fb_color(render3d_st, 0, im.albedo)
|
|
gpu_fb_color(render3d_st, 1, im.normal)
|
|
let rb = gpu_rb_new(render3d_st)
|
|
gpu_rb_storage(render3d_st, rb, GL_DEPTH_COMPONENT24, aw, th, 0)
|
|
gpu_fb_depth_rb(render3d_st, rb)
|
|
gpu_fb_draw_buffers(render3d_st, 2)
|
|
gpu_viewport(render3d_st, 0, 0, aw, th)
|
|
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 0.0)
|
|
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
|
gpu_depth_test(render3d_st, true)
|
|
gpu_depth_func(render3d_st, GL_LESS)
|
|
gpu_cull(render3d_st, false)
|
|
gpu_blend(render3d_st, false)
|
|
# the model's prims temporarily take the identity instance
|
|
for i in 0 .. len(model.prims) { scatter_attach(render3d_st, model.prims[i].mesh, render3d_st.sc_ident_buf) }
|
|
let view = m4_new(); let proj = m4_new()
|
|
let eye = floats(3); let at = floats(3); let up = v3_new(0.0, 1.0, 0.0)
|
|
let cy = model.ymin + model.height * 0.5
|
|
let r = im.radius
|
|
let hh = model.height * 0.5
|
|
var bake = render3d_st.sc_bake_prog
|
|
if render3d_st.sc_bake_flower { bake = render3d_st.sc_bake_flower_prog }
|
|
gpu_use_program(render3d_st, bake)
|
|
for t in 0 .. tiles {
|
|
let a = 2.0 * PI * (float(t) / float(tiles))
|
|
v3_set(at, 0.0, cy, 0.0)
|
|
# a touch of elevation (the viewer usually looks slightly down at a tree)
|
|
v3_set(eye, Math.sin(a) * (r * 4.0), cy + r * 0.5, -(Math.cos(a) * (r * 4.0)))
|
|
m4_look_at(view, eye, at, up)
|
|
m4_ortho(proj, -r, r, -hh, hh, 0.1, r * 9.0)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, bake, "u_view"), view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, bake, "u_proj"), proj)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_wind"), 0.0)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_flutter"), 0.0)
|
|
gpu_viewport(render3d_st, t * tw, 0, tw, th)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
r3d_bind_2d(render3d_st, bake, "u_diff", 0, pr.diff)
|
|
r3d_bind_2d(render3d_st, bake, "u_arm", 2, pr.arm)
|
|
mesh_draw_instanced(render3d_st, pr.mesh, 1)
|
|
}
|
|
}
|
|
free(view); free(proj); free(eye); free(at); free(up)
|
|
gpu_fb_bind(render3d_st, 0)
|
|
gpu_fb_free(render3d_st, fbo)
|
|
gpu_rb_free(render3d_st, rb)
|
|
gpu_tex_bind(render3d_st, GPU_TEX2D, im.albedo)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
|
gpu_tex_bind(render3d_st, GPU_TEX2D, im.normal)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
|
gpu_check(render3d_st, "impostor bake")
|
|
# numbered, so every bake of a run survives to be compared (a card layer per species bakes one)
|
|
if render3d_st.sc_debug_dump {
|
|
render3d_st.sc_dump_n += 1
|
|
render3d_st.tex_dump_alpha = true; tex_dump(render3d_st, im.albedo, aw, th, `build/atlas_{render3d_st.sc_dump_n}_alpha.ppm`); render3d_st.tex_dump_alpha = false
|
|
tex_dump(render3d_st, im.albedo, aw, th, `build/atlas_{render3d_st.sc_dump_n}_color.ppm`)
|
|
}
|
|
return im
|
|
}
|
|
|
|
# Give a layer a LOD chain. `dists` (float bits) are the outer distances of each level;
|
|
# the last one becomes the layer's `near` so the impostor (if any) starts there.
|
|
@alloc_ok("start-up: the device, its tables, the programs, the passes and the world's first textures are made once, before play")
|
|
function layer_set_lods(render3d_st: mut Render3dState, l: Layer, models: []Model, dists: floats) -> void {
|
|
l.lods = models
|
|
l.n_lods = len(models)
|
|
l.lod_dist = floats(l.n_lods); l.lod_card = words(l.n_lods); l.lod_buf = words(l.n_lods); l.n_lod = words(l.n_lods)
|
|
for k in 0 .. l.n_lods {
|
|
l.lod_dist[k] = dists[k]; l.lod_card[k] = 0; l.n_lod[k] = 0
|
|
l.lod_buf[k] = gpu_buffer_new(render3d_st)
|
|
let m = models[k]
|
|
for i in 0 .. len(m.prims) { scatter_attach(render3d_st, m.prims[i].mesh, l.lod_buf[k]) }
|
|
}
|
|
l.model = models[0]
|
|
l.near = dists[l.n_lods - 1]
|
|
if l.lvl == null { l.lvl = words(l.cap) }
|
|
}
|
|
# mark level k as the layer's crossed card (drawn with the card program and its atlas)
|
|
function layer_lod_card(l: Layer, k: int) -> void { l.lod_card[k] = 1 }
|
|
|
|
function layer_set_impostor(render3d_st: Render3dState, l: Layer, im: Impostor) -> void {
|
|
l.imp = im
|
|
scatter_attach(render3d_st, render3d_st.sc_card, l.imp_buf)
|
|
}
|
|
|
|
# ---- per frame -----------------------------------------------------------------------
|
|
# Partitions and gathers are redone only when the view changed enough to matter: the
|
|
# camera moved 1.5 m or turned about 2.5 degrees. Everything culled by the frustum keys
|
|
# off this one counter, so a turn re-gathers the streams and the grids together.
|
|
function scatter_begin_frame(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.sc_view_pos == null { render3d_st.sc_view_pos = v3_new(100000.0, 0.0, 0.0); render3d_st.sc_view_fwd = v3_new(0.0, 0.0, -1.0) }
|
|
if v3_dist(render3d_st.sc_view_pos, render3d_st.cam_pos) > 1.5 or v3_dot(render3d_st.sc_view_fwd, render3d_st.cam_fwd) < 0.999 {
|
|
render3d_st.sc_view_gen += 1
|
|
v3_copy(render3d_st.sc_view_pos, render3d_st.cam_pos)
|
|
v3_copy(render3d_st.sc_view_fwd, render3d_st.cam_fwd)
|
|
}
|
|
# GPU-culled layers dispatch before the first pass of the frame, so no pass is split for it
|
|
if render3d_st.sc_layers != null {
|
|
for i in 0 .. len(render3d_st.sc_layers) { if render3d_st.sc_layers[i].n_lods > 1 and render3d_st.sc_layers[i].imp != null { layer_update(render3d_st, render3d_st.sc_layers[i]) } }
|
|
}
|
|
}
|
|
|
|
# ---- the GPU-culled path ------------------------------------------------------------------
|
|
# On Vulkan (R3D_GPU_CULL=0 turns it off): a tree layer (a LOD chain of up to four levels sharing up
|
|
# to four materials, with an impostor, not streamed) is culled and split into its buckets by
|
|
# scatter_cull.comp, and its lit, prepass, impostor and shadow-LOD draws read the records that pass
|
|
# wrote - one draw per material covering every level. Nothing is partitioned or uploaded on the CPU
|
|
# when the view moves. PC camp benchmark: 2791 -> 2657 draws, 4.3 -> 4.1 s for 400 frames.
|
|
const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand
|
|
const SC_LOD_ROOM: int = 64 # a layer's LODs plus near and far, most (sc_lod_* are made this size)
|
|
const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp)
|
|
|
|
function layer_gpu_eligible(render3d_st: Render3dState, l: Layer) -> bool {
|
|
if l.n_lods < 2 or l.n_lods > 4 or l.imp == null or l.streamed or l.flower or l.blade or l.count == 0 { return false }
|
|
return layer_arena_ok(render3d_st, l)
|
|
}
|
|
|
|
# The merged meshes. Every level of a kit tree or rock carries the same materials in the same order
|
|
# (bark then needles; the rock's one), so each material becomes ONE mesh holding all its levels, and
|
|
# one indirect draw of several records draws every level of it: record (material, level) names that
|
|
# level's index and vertex range and its bucket's instances. A conifer's lit pass goes from eight
|
|
# draws to two. Anything that does not fit - a card level, a level with other materials, other
|
|
# attributes or 32-bit indices - keeps the CPU path.
|
|
function layer_arena_ok(render3d_st: Render3dState, l: Layer) -> bool {
|
|
let n_mat = len(l.lods[0].prims)
|
|
if n_mat == 0 or n_mat > 4 { return false }
|
|
for k in 0 .. l.n_lods {
|
|
if l.lod_card[k] == 1 { return false }
|
|
let m = l.lods[k]
|
|
if len(m.prims) != n_mat { return false }
|
|
for j in 0 .. n_mat {
|
|
let pm = m.prims[j].mesh
|
|
let p0 = l.lods[0].prims[j]
|
|
if m.prims[j].diff != p0.diff or m.prims[j].verts == 0 or pm.ebo == 0 or pm.itype != GL_UNSIGNED_SHORT { return false }
|
|
if gpu_buffer_map(render3d_st, pm.ebo) == null { return false }
|
|
for a in 0 .. 3 {
|
|
let o = a * GPU_ATTR_W
|
|
if p0.mesh.attrs[o + 1] == 0 or pm.attrs[o + 1] != p0.mesh.attrs[o + 1] or pm.attrs[o + 3] != pm.attrs[o + 1] * 4 { return false }
|
|
if gpu_buffer_map(render3d_st, pm.attrs[o]) == null { return false }
|
|
}
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function layer_arena_build(render3d_st: mut Render3dState, l: Layer) -> void {
|
|
let n_mat = len(l.lods[0].prims)
|
|
l.g_arena = new []Prim
|
|
l.g_first = words(16); l.g_base = words(16)
|
|
for i in 0 .. 16 { l.g_first[i] = 0; l.g_base[i] = 0 }
|
|
for j in 0 .. n_mat {
|
|
var nv = 0
|
|
var ni = 0
|
|
for k in 0 .. l.n_lods {
|
|
let pr = l.lods[k].prims[j]
|
|
l.g_first[j * 4 + k] = ni; l.g_base[j * 4 + k] = nv
|
|
nv += pr.verts; ni += pr.mesh.count
|
|
}
|
|
let p0 = l.lods[0].prims[j]
|
|
let m = gpu_mesh_new(render3d_st)
|
|
for a in 0 .. 3 {
|
|
let comps = p0.mesh.attrs[a * GPU_ATTR_W + 1]
|
|
let vb = bytes(nv * comps * 4 + 8)
|
|
for k in 0 .. l.n_lods {
|
|
let pr = l.lods[k].prims[j]
|
|
mem_copy(mem_off(vb, l.g_base[j * 4 + k] * comps * 4), gpu_buffer_map(render3d_st, pr.mesh.attrs[a * GPU_ATTR_W]), pr.verts * comps * 4)
|
|
}
|
|
gpu_mesh_vertices(render3d_st, m, vb, nv * comps * 4, GPU_STATIC)
|
|
gpu_mesh_attr(render3d_st, m, a, comps, GPU_F32, 0, 0, false)
|
|
free(vb)
|
|
}
|
|
let ib = bytes(ni * 2 + 8)
|
|
for k in 0 .. l.n_lods {
|
|
let pm = l.lods[k].prims[j].mesh
|
|
mem_copy(mem_off(ib, l.g_first[j * 4 + k] * 2), gpu_buffer_map(render3d_st, pm.ebo), pm.count * 2)
|
|
}
|
|
gpu_mesh_indices(render3d_st, m, ib, ni * 2, 2)
|
|
free(ib)
|
|
m.count = ni
|
|
gpu_mesh_done(render3d_st, m)
|
|
scatter_attach(render3d_st, m, l.g_dst)
|
|
let ap = new Prim
|
|
ap.mesh = m; ap.diff = p0.diff; ap.nrm = p0.nrm; ap.arm = p0.arm; ap.verts = nv; ap.name = p0.name
|
|
push(l.g_arena, ap)
|
|
}
|
|
l.g_model = new Model
|
|
l.g_model.prims = l.g_arena; l.g_model.height = l.lods[0].height; l.g_model.radius = l.lods[0].radius; l.g_model.ymin = l.lods[0].ymin
|
|
l.g_model_sh = new Model
|
|
var sh = l.n_lods - 1
|
|
if sh > 2 { sh = 2 }
|
|
l.g_model_sh.prims = l.g_arena; l.g_model_sh.height = l.lods[sh].height; l.g_model_sh.radius = l.lods[sh].radius; l.g_model_sh.ymin = l.lods[sh].ymin
|
|
}
|
|
|
|
function layer_gpu_prepare(render3d_st: mut Render3dState, l: Layer) -> bool {
|
|
if not render3d_st.sc_cull_tried {
|
|
render3d_st.sc_cull_tried = true
|
|
# Os.env is null when the variable is unset, and a compare reads through it: ask first
|
|
# on by default wherever there is compute; R3D_GPU_CULL=0 keeps the CPU partition (for comparing)
|
|
var off = false
|
|
if r3d_env_has(render3d_st, "R3D_GPU_CULL") { off = r3d_env(render3d_st, "R3D_GPU_CULL") == "0" }
|
|
if gpu_has_compute(render3d_st) and gpu_has_mdi(render3d_st) and not off {
|
|
render3d_st.sc_cull_prog = gpu_compute(render3d_st, "scatter_cull", 4)
|
|
if render3d_st.sc_cull_prog > 0 { print("r3d: scatter: tree layers are culled on the GPU") }
|
|
}
|
|
}
|
|
if render3d_st.sc_cull_prog == 0 or not layer_gpu_eligible(render3d_st, l) { return false }
|
|
if l.g_on and l.g_n == l.count { return true }
|
|
if l.g_src == 0 {
|
|
l.g_src = gpu_buffer_new(render3d_st); l.g_dst = gpu_buffer_new(render3d_st); l.g_cmds = gpu_buffer_new(render3d_st); l.g_counts = gpu_buffer_new(render3d_st)
|
|
gpu_buffer_gpu_owned(render3d_st, l.g_dst); gpu_buffer_gpu_owned(render3d_st, l.g_cmds); gpu_buffer_gpu_owned(render3d_st, l.g_counts)
|
|
}
|
|
let n = l.n_lods
|
|
let cap = l.count
|
|
gpu_buffer_upload(render3d_st, l.g_src, cap * INST_FLOATS * 4, data_of(l.inst), GPU_STATIC)
|
|
gpu_buffer_upload(render3d_st, l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC)
|
|
if l.g_arena == null { layer_arena_build(render3d_st, l) }
|
|
let n_mat = len(l.g_arena)
|
|
let rec = render3d_st.sc_gpu_rec
|
|
for i in 0 .. SC_RECS * 5 { rec[i] = 0 }
|
|
# material j, level k: that level's range of the merged mesh, its instances from bucket k
|
|
for j in 0 .. n_mat {
|
|
for k in 0 .. n {
|
|
let r = (j * 4 + k) * 5
|
|
rec[r] = l.lods[k].prims[j].mesh.count; rec[r + 2] = l.g_first[j * 4 + k]; rec[r + 3] = l.g_base[j * 4 + k]; rec[r + 4] = k * cap
|
|
}
|
|
}
|
|
rec[16 * 5] = render3d_st.sc_card.count; rec[16 * 5 + 4] = n * cap
|
|
# the shadow LOD: level 2's range, from buckets 0 .. 2
|
|
if n > 2 {
|
|
for j in 0 .. n_mat {
|
|
for b in 0 .. 3 {
|
|
let r = (17 + j * 3 + b) * 5
|
|
rec[r] = l.lods[2].prims[j].mesh.count; rec[r + 2] = l.g_first[j * 4 + 2]; rec[r + 3] = l.g_base[j * 4 + 2]; rec[r + 4] = b * cap
|
|
}
|
|
}
|
|
}
|
|
gpu_buffer_upload(render3d_st, l.g_cmds, SC_RECS * SC_REC_W, data_of(rec), GPU_DYNAMIC)
|
|
let zeros = render3d_st.sc_gpu_zeros
|
|
for i in 0 .. 5 { zeros[i] = 0 }
|
|
gpu_buffer_upload(render3d_st, l.g_counts, 20, data_of(zeros), GPU_DYNAMIC)
|
|
# the card casts every instance, as on the CPU path (layer_grid_build uploads this there)
|
|
l.n_sh = l.count
|
|
gpu_buffer_upload(render3d_st, l.sh_buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_STATIC)
|
|
l.g_n = l.count
|
|
l.g_on = true
|
|
return true
|
|
}
|
|
|
|
# the dispatch for the view as it stands: frustum, camera, distances, the layer's shape
|
|
function layer_gpu_cull(render3d_st: mut Render3dState, l: Layer) -> void {
|
|
let pr = render3d_st.sc_gpu_pr
|
|
for i in 0 .. 36 { pr[i] = 0 }
|
|
if render3d_st.cam_planes != null { for i in 0 .. 16 { pr[i] = float_bits(render3d_st.cam_planes[i]) } }
|
|
pr[16] = float_bits(render3d_st.cam_pos[0]); pr[17] = float_bits(render3d_st.cam_pos[1]); pr[18] = float_bits(render3d_st.cam_pos[2]); pr[19] = float_bits(r3d_reach(render3d_st, l.cull))
|
|
for k in 0 .. l.n_lods { pr[20 + k] = float_bits(l.lod_dist[k]); pr[24 + k] = len(l.lods[k].prims) }
|
|
pr[28] = l.count; pr[29] = l.count; pr[30] = l.n_lods; pr[31] = 1
|
|
# as layer_grid_gather pads a cell: the tallest instance, plus a margin
|
|
pr[32] = float_bits(l.lods[0].height * 2.0); pr[33] = float_bits(4.0)
|
|
let bufs = render3d_st.sc_gpu_bufs
|
|
bufs[0] = l.g_src; bufs[1] = l.g_dst; bufs[2] = l.g_cmds; bufs[3] = l.g_counts
|
|
gpu_dispatch(render3d_st, render3d_st.sc_cull_prog, data_of(pr), 144, bufs, 1)
|
|
}
|
|
|
|
# Sort a static layer's instances into square cells (call once, after placement; a
|
|
# large layer that was never gridded gets a 96 m grid on its first update). The
|
|
# shadow buffer is uploaded here once — casters are never culled by the view.
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function layer_grid_build(render3d_st: mut Render3dState, l: Layer, cell: float) -> void {
|
|
if l.count == 0 { return }
|
|
var minx = l.inst[0]; var maxx = minx; var minz = l.inst[2]; var maxz = minz
|
|
for i in 0 .. l.count {
|
|
let o = i * INST_FLOATS
|
|
minx = Math.min(minx, l.inst[o]); maxx = Math.max(maxx, l.inst[o])
|
|
minz = Math.min(minz, l.inst[o + 2]); maxz = Math.max(maxz, l.inst[o + 2])
|
|
}
|
|
l.gcell = cell; l.gx0 = minx; l.gz0 = minz
|
|
l.gnx = int((maxx - minx) / cell) + 1
|
|
l.gnz = int((maxz - minz) / cell) + 1
|
|
let ncell = l.gnx * l.gnz
|
|
l.gstart = words(ncell + 1)
|
|
l.gymin = floats(ncell); l.gymax = floats(ncell)
|
|
let cellof = words(l.count)
|
|
for c in 0 .. ncell + 1 { l.gstart[c] = 0 }
|
|
for i in 0 .. l.count {
|
|
let o = i * INST_FLOATS
|
|
let ix = int((l.inst[o] - minx) / cell)
|
|
let iz = int((l.inst[o + 2] - minz) / cell)
|
|
let c = iz * l.gnx + ix
|
|
cellof[i] = c
|
|
if l.gstart[c + 1] == 0 { l.gymin[c] = l.inst[o + 1]; l.gymax[c] = l.inst[o + 1] }
|
|
else { l.gymin[c] = Math.min(l.gymin[c], l.inst[o + 1]); l.gymax[c] = Math.max(l.gymax[c], l.inst[o + 1]) }
|
|
l.gstart[c + 1] += 1
|
|
}
|
|
for c in 0 .. ncell { l.gstart[c + 1] += l.gstart[c] }
|
|
let fill = words(ncell)
|
|
for c in 0 .. ncell { fill[c] = l.gstart[c] }
|
|
l.gsorted = floats(l.count * INST_FLOATS)
|
|
for i in 0 .. l.count {
|
|
let c = cellof[i]
|
|
let q = fill[c] * INST_FLOATS
|
|
fill[c] += 1
|
|
let o = i * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { l.gsorted[q + k] = l.inst[o + k] }
|
|
}
|
|
free(cellof); free(fill)
|
|
if l.vis == null { l.vis = floats(l.cap * INST_FLOATS) }
|
|
l.n_sh = l.count
|
|
gpu_buffer_upload(render3d_st, l.sh_buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_STATIC)
|
|
}
|
|
|
|
# gather the instances of the cells the camera can see (and that are within cull)
|
|
function layer_grid_gather(render3d_st: Render3dState, l: Layer) -> void {
|
|
let cell = l.gcell
|
|
let half = cell * 0.5
|
|
let reach = r3d_reach(render3d_st, l.cull) + cell * 0.71
|
|
var n = 0
|
|
for iz in 0 .. l.gnz {
|
|
let wz = l.gz0 + float(iz) * cell + half
|
|
for ix in 0 .. l.gnx {
|
|
let c = iz * l.gnx + ix
|
|
let cnt = l.gstart[c + 1] - l.gstart[c]
|
|
if cnt == 0 { continue }
|
|
let wx = l.gx0 + float(ix) * cell + half
|
|
if r3d_reach(render3d_st, l.cull) != 0.0 {
|
|
let dx = wx - render3d_st.cam_pos[0]; let dz = wz - render3d_st.cam_pos[2]
|
|
if Math.sqrt(dx * dx + dz * dz) > reach { continue }
|
|
}
|
|
let hy = (l.gymax[c] - l.gymin[c]) * 0.5
|
|
let cy = l.gymin[c] + hy
|
|
# pad by the tallest instance (scale 2 of the model's height) so crowns at the frame's edge stay
|
|
let r = Math.sqrt(half * half * 2.0 + hy * hy) + (l.model.height * 2.0 + 4.0)
|
|
if not cam_sphere_visible(render3d_st, wx, cy, wz, r) { continue }
|
|
mem_copy(mem_off(l.vis, n * INST_FLOATS * 4), mem_off(l.gsorted, l.gstart[c] * INST_FLOATS * 4), cnt * INST_FLOATS * 4)
|
|
n += cnt
|
|
}
|
|
}
|
|
l.n_vis = n
|
|
}
|
|
|
|
# Sort the gathered instances into their LOD levels (counting sort into the scratch),
|
|
# the impostor bucket last, and upload one buffer per level.
|
|
function layer_partition_lods(render3d_st: mut Render3dState, l: Layer, src: floats, total: int) -> void {
|
|
let n = l.n_lods
|
|
if n + 2 > SC_LOD_ROOM { return } # past the room made with the state
|
|
let counts = render3d_st.sc_lod_counts
|
|
for k in 0 .. n + 2 { counts[k] = 0 }
|
|
let lc = r3d_reach(render3d_st, l.cull)
|
|
let cull2 = lc * lc
|
|
let open = l.lod_dist[n - 1] == 0.0 # the last level runs out to the cull distance
|
|
for i in 0 .. total {
|
|
let o = i * INST_FLOATS
|
|
let dx = src[o] - render3d_st.cam_pos[0]; let dz = src[o + 2] - render3d_st.cam_pos[2]
|
|
let d2 = dx * dx + dz * dz
|
|
var lv = n + 1 # n + 1 = dropped
|
|
if lc == 0.0 or not (d2 > cull2) {
|
|
let d = Math.sqrt(d2)
|
|
lv = n # n = the impostor bucket
|
|
var k = 0
|
|
while k < n { if l.lod_dist[k] != 0.0 and d < l.lod_dist[k] { lv = k; k = n } else { k += 1 } }
|
|
if lv == n and open { lv = n - 1 }
|
|
if lv == n and l.imp == null { lv = n + 1 }
|
|
}
|
|
l.lvl[i] = lv
|
|
counts[lv] += 1
|
|
}
|
|
# prefix offsets (in instances) per bucket
|
|
let start = render3d_st.sc_lod_start
|
|
var acc = 0
|
|
for k in 0 .. n + 2 { start[k] = acc; acc += counts[k] }
|
|
let fill = render3d_st.sc_lod_fill
|
|
for k in 0 .. n + 2 { fill[k] = start[k] }
|
|
let tmp = l.scratch
|
|
for i in 0 .. total {
|
|
let lv = l.lvl[i]
|
|
if lv > n { continue }
|
|
let q = fill[lv] * INST_FLOATS
|
|
fill[lv] += 1
|
|
let o = i * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
}
|
|
for k in 0 .. n {
|
|
l.n_lod[k] = counts[k]
|
|
if counts[k] > 0 {
|
|
gpu_buffer_upload(render3d_st, l.lod_buf[k], counts[k] * INST_FLOATS * 4, mem_off(tmp, start[k] * INST_FLOATS * 4), GPU_DYNAMIC)
|
|
}
|
|
}
|
|
l.n_near = counts[0]
|
|
l.n_far = counts[n]
|
|
if render3d_st.sc_dbg_lod and total > 1000 { sc_debug_lods(render3d_st, l, counts, n, total, src) }
|
|
if l.n_far > 0 {
|
|
gpu_buffer_upload(render3d_st, l.imp_buf, l.n_far * INST_FLOATS * 4, mem_off(tmp, start[n] * INST_FLOATS * 4), GPU_DYNAMIC)
|
|
}
|
|
# casters: the whole (gathered) set from the shadow buffer, unless the impostor casts
|
|
if l.gcell == 0.0 {
|
|
l.n_sh = total
|
|
if total > 0 { gpu_buffer_upload(render3d_st, l.sh_buf, total * INST_FLOATS * 4, data_of(src), GPU_DYNAMIC) }
|
|
}
|
|
}
|
|
|
|
# split the instances by distance to the camera (only when the view changed)
|
|
function layer_update(render3d_st: mut Render3dState, l: Layer) -> void {
|
|
if render3d_st.sc_freeze { return }
|
|
if l.view_gen == render3d_st.sc_view_gen { return }
|
|
l.view_gen = render3d_st.sc_view_gen
|
|
if layer_gpu_prepare(render3d_st, l) { layer_gpu_cull(render3d_st, l); return }
|
|
let t_lu = gl_now_us()
|
|
let n_lu = l.count
|
|
# A streamed layer's instances were already gathered per visible chunk: no split, no
|
|
# per-instance loop — one upload, and the same buffer casts its shadows.
|
|
if l.streamed and l.imp == null and l.near == 0.0 and l.n_lods <= 1 {
|
|
l.n_near = l.count; l.n_far = 0; l.n_sh = l.count
|
|
gpu_buffer_upload(render3d_st, l.buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_DYNAMIC)
|
|
prof_layer_add(render3d_st, gl_now_us() - t_lu, l.count * INST_FLOATS * 4)
|
|
return
|
|
}
|
|
if l.gcell == 0.0 and not l.streamed and l.count > 2000 { layer_grid_build(render3d_st, l, 96.0) }
|
|
var src = l.inst
|
|
var total = l.count
|
|
if l.gcell != 0.0 { layer_grid_gather(render3d_st, l); src = l.vis; total = l.n_vis }
|
|
let near2 = l.near * l.near
|
|
let lc = r3d_reach(render3d_st, l.cull)
|
|
let cull2 = lc * lc
|
|
var nn = 0
|
|
var nf = 0
|
|
let far_off = l.cap * INST_FLOATS # far instances fill the scratch from its end backwards
|
|
let tmp = l.scratch
|
|
if l.n_lods > 1 {
|
|
layer_partition_lods(render3d_st, l, src, total)
|
|
prof_layer_add(render3d_st, gl_now_us() - t_lu, total * INST_FLOATS * 4)
|
|
return
|
|
}
|
|
var i = 0
|
|
while i < total {
|
|
let o = i * INST_FLOATS
|
|
let dx = src[o] - render3d_st.cam_pos[0]
|
|
let dz = src[o + 2] - render3d_st.cam_pos[2]
|
|
let d2 = dx * dx + dz * dz
|
|
if lc != 0.0 and d2 > cull2 { i += 1; continue }
|
|
if l.near == 0.0 or d2 < near2 {
|
|
let q = nn * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
nn += 1
|
|
} else if l.imp != null {
|
|
nf += 1
|
|
let q = far_off - nf * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
}
|
|
i += 1
|
|
}
|
|
l.n_near = nn
|
|
l.n_far = nf
|
|
if render3d_st.sc_debug_dump { sc_debug_layer(l, tmp, nn, nf) }
|
|
# Every instance, unculled and unsplit, for the shadow pass. What the camera draws is
|
|
# allowed to change with distance; what casts must not, or shadows blink in and out as
|
|
# you walk. This is the whole set, drawn one way, into every cascade.
|
|
if l.gcell == 0.0 {
|
|
l.n_sh = l.count
|
|
if l.count > 0 {
|
|
gpu_buffer_upload(render3d_st, l.sh_buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_DYNAMIC)
|
|
}
|
|
}
|
|
gpu_buffer_upload(render3d_st, l.buf, nn * INST_FLOATS * 4, data_of(tmp), GPU_DYNAMIC)
|
|
if nf > 0 {
|
|
gpu_buffer_upload(render3d_st, l.imp_buf, nf * INST_FLOATS * 4, mem_off(tmp, (far_off - nf * INST_FLOATS) * 4), GPU_DYNAMIC)
|
|
}
|
|
prof_layer_add(render3d_st, gl_now_us() - t_lu, (nn + nf + l.n_sh) * INST_FLOATS * 4)
|
|
}
|
|
|
|
function layer_program(render3d_st: Render3dState, l: Layer, shadow: bool, card: bool) -> int {
|
|
if card {
|
|
if shadow { return render3d_st.sc_prog_card_shadow }
|
|
if l.cheap { return render3d_st.sc_prog_card_cheap }
|
|
return render3d_st.sc_prog_card
|
|
}
|
|
if shadow {
|
|
if l.foliage and not l.blade and not l.flower { return render3d_st.sc_prog_shadow_fol }
|
|
if l.wind != 0.0 { return render3d_st.sc_prog_shadow_wind }
|
|
return render3d_st.sc_prog_shadow
|
|
}
|
|
if l.blade { return render3d_st.sc_prog_blade }
|
|
if l.flower { return render3d_st.sc_prog_flower }
|
|
if l.foliage {
|
|
if render3d_st.sc_prepass and render3d_st.sc_prog_fol_eq != 0 { return render3d_st.sc_prog_fol_eq }
|
|
return render3d_st.sc_prog_fol
|
|
}
|
|
if l.wind != 0.0 { return render3d_st.sc_prog_wind }
|
|
return render3d_st.sc_prog
|
|
}
|
|
|
|
# R3D_LODDBG=1 tints each LOD level (red, green, blue, yellow) and impostors magenta
|
|
# Can level k's casters (its instances lie between the previous level's distance and its own) put a
|
|
# shadow on anything the cascade being rendered covers? A receiver in that slice of view depth
|
|
# [near, far] stands between near - dy and far * K metres away on the ground: dy is the camera's height
|
|
# over the ground, K how far the frustum's corners reach past its depth. A prop's shadow falls at most
|
|
# about six times its height past it (the sun near ten degrees). A level outside that range cannot
|
|
# touch a pixel of the cascade, so leaving it out changes no shadow - unlike the old per-class skips
|
|
# at fixed distances, which dropped casters that did cast (see scatter_draw_casters). The flowers'
|
|
# mesh levels (6 - 30 m) stop being drawn into the three outer cascades. R3D_CAST_ALL=1 draws every
|
|
# level into every cascade, for comparing.
|
|
function layer_level_casts_here(render3d_st: mut Render3dState, l: Layer, k: int) -> bool {
|
|
if l.lod_dist == null { return true }
|
|
var dmin = 0.0
|
|
if k > 0 { dmin = l.lod_dist[k - 1] }
|
|
return cast_band_reaches(render3d_st, dmin, l.lod_dist[k], l.lods[k].height)
|
|
}
|
|
# Can something standing between dmin and dmax metres from the camera (dmax 0: no outer limit), this
|
|
# tall, put a shadow on anything the cascade being rendered covers? The flowers' levels ask it
|
|
# (layer_level_casts_here), and so does every actor (actor_draw_casters).
|
|
function cast_band_reaches(render3d_st: mut Render3dState, dmin: float, dmax: float, height: float) -> bool {
|
|
if render3d_st.sc_cast_all < 0 { render3d_st.sc_cast_all = 0; if r3d_env_has(render3d_st, "R3D_CAST_ALL") { render3d_st.sc_cast_all = 1 } }
|
|
if render3d_st.sc_cast_all == 1 or render3d_st.sh_split == null { return true }
|
|
if render3d_st.sc_cast_gen != render3d_st.sc_view_gen {
|
|
render3d_st.sc_cast_gen = render3d_st.sc_view_gen
|
|
render3d_st.sc_cast_dy = Math.abs(render3d_st.cam_pos[1] - terrain_height(render3d_st, render3d_st.cam_pos[0], render3d_st.cam_pos[2])) + 5.0
|
|
let half = render3d_st.cam_fov * 0.5
|
|
let t = Math.sin(half) / Math.cos(half)
|
|
let ta = t * render3d_st.cam_aspect
|
|
render3d_st.sc_cast_k = Math.sqrt(1.0 + (t * t + ta * ta)) * 1.1
|
|
}
|
|
let c = render3d_st.sh_cascade
|
|
var near = render3d_st.cam_near
|
|
# sunShadow cross-fades into this cascade from 0.85 of the previous one's split (lighting.glsl), so
|
|
# its receivers start there, not at the split: starting at the split changed 19 pixels in town
|
|
if c > 0 { near = render3d_st.sh_split[c - 1] * 0.85 }
|
|
let far = render3d_st.sh_split[c]
|
|
let reach = Math.max(Math.min(height * 6.0, 40.0), 8.0)
|
|
if dmax != 0.0 and dmax + reach + render3d_st.sc_cast_dy < near { return false }
|
|
if dmin - reach > far * render3d_st.sc_cast_k { return false }
|
|
return true
|
|
}
|
|
|
|
# `full` casts the layer's entire instance list out of sh_buf instead of the near
|
|
# partition out of l.buf. A layer with no impostor (the tree crowns' branch cards) has
|
|
# no cheap stand-in to cast from, so without this its shadow simply began at the near
|
|
# distance — which is the crown shadow that appeared as you walked up to a tree.
|
|
function layer_draw_near(render3d_st: mut Render3dState, l: Layer, shadow: bool, light_vp: floats, full: bool) -> void {
|
|
if l.n_lods > 1 and l.g_on and not shadow {
|
|
# one draw per material covering all of its levels (the merged meshes)
|
|
render3d_st.sc_ind_base = 0; render3d_st.sc_ind_n = l.n_lods
|
|
layer_draw_model(render3d_st, l, l.g_model, l.g_dst, 1, false, shadow, light_vp)
|
|
render3d_st.sc_ind_base = -1; render3d_st.sc_ind_n = 1
|
|
return
|
|
}
|
|
if l.n_lods > 1 {
|
|
# a LOD chain: every level from its own bucket (casters are what is drawn), and a caster level
|
|
# only into the cascades it can put a shadow in
|
|
for k in 0 .. l.n_lods {
|
|
if shadow and not layer_level_casts_here(render3d_st, l, k) { continue }
|
|
render3d_st.sc_dbg_level = k; layer_draw_model(render3d_st, l, l.lods[k], l.lod_buf[k], l.n_lod[k], l.lod_card[k] == 1, shadow, light_vp)
|
|
}
|
|
render3d_st.sc_dbg_level = -1
|
|
return
|
|
}
|
|
var vb = l.buf
|
|
var cnt = l.n_near
|
|
if full and not l.streamed { vb = l.sh_buf; cnt = l.n_sh }
|
|
layer_draw_model(render3d_st, l, l.model, vb, cnt, l.card, shadow, light_vp)
|
|
}
|
|
|
|
# draw `cnt` instances of `model` out of instance buffer `vb`, as a mesh or as the layer's card
|
|
function layer_draw_model(render3d_st: mut Render3dState, l: Layer, model: Model, vb: int, cnt: int, card: bool, shadow: bool, light_vp: floats) -> void {
|
|
if cnt == 0 { return }
|
|
let p = layer_program(render3d_st, l, shadow, card)
|
|
gpu_use_program(render3d_st, p)
|
|
# over the prepass: only the fragment the prepass kept, at exactly its depth (a texel
|
|
# it cut would otherwise pass LEQUAL over the terrain behind and draw the quad solid)
|
|
if p == render3d_st.sc_prog_fol_eq { gpu_depth_func(render3d_st, GL_EQUAL) }
|
|
var ground = 0.0
|
|
if l.grounded { ground = 1.0 }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_ground"), ground)
|
|
if l.grounded { terrain_bind_height(render3d_st, p) }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_time"), render3d_st.r3d_time)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_wind"), l.wind)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_flutter"), l.flutter)
|
|
var mh = 0.0
|
|
if not card and l.foliage { mh = model.height }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_model_h"), mh)
|
|
if not card and l.foliage and not shadow and render3d_st.sc_a2c { gpu_alpha_to_coverage(render3d_st, true) }
|
|
if card {
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_card_w"), l.atlas.radius * 2.0); u_f(render3d_st, gpu_uniform(render3d_st, p, "u_card_h"), l.atlas.height)
|
|
r3d_bind_2d(render3d_st, p, "u_diff", 0, l.atlas.albedo)
|
|
r3d_bind_2d(render3d_st, p, "u_nrm", 1, l.atlas.normal)
|
|
gpu_alpha_to_coverage(render3d_st, true)
|
|
}
|
|
if shadow { u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_light_vp"), light_vp) }
|
|
else {
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_view"), render3d_st.cam_view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_proj"), render3d_st.cam_proj)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), l.tint)
|
|
if render3d_st.sc_dbg_lod and render3d_st.sc_dbg_level >= 0 {
|
|
if render3d_st.sc_dbg_tint == null { render3d_st.sc_dbg_tint = v3_new(1.0, 1.0, 1.0) }
|
|
let k = render3d_st.sc_dbg_level
|
|
var r = 0.0; var g = 0.0; var b = 0.0
|
|
if k == 0 { r = 3.0 } else if k == 1 { g = 3.0 } else if k == 2 { b = 3.0 } else { r = 3.0; g = 3.0 }
|
|
v3_set(render3d_st.sc_dbg_tint, r, g, b)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), render3d_st.sc_dbg_tint)
|
|
}
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_rough_scale"), l.rough)
|
|
if l.blade { u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_blade_base"), render3d_st.sc_blade_base); u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_blade_tip"), render3d_st.sc_blade_tip) }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_cull"), r3d_reach(render3d_st, l.cull))
|
|
sky_bind_lighting(render3d_st, p)
|
|
shadow_bind(render3d_st, p)
|
|
fog_bind(render3d_st, p)
|
|
if l.foliage { u_f(render3d_st, gpu_uniform(render3d_st, p, "u_spec_scale"), 0.05) }
|
|
}
|
|
gpu_cull(render3d_st, false)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
scatter_attach(render3d_st, pr.mesh, vb)
|
|
if not card {
|
|
r3d_bind_2d(render3d_st, p, "u_diff", 0, pr.diff)
|
|
if not shadow { r3d_bind_2d(render3d_st, p, "u_nrm", 1, pr.nrm); r3d_bind_2d(render3d_st, p, "u_arm", 2, pr.arm) }
|
|
}
|
|
if render3d_st.sc_ind_base >= 0 { gpu_draw_mesh_indirect(render3d_st, pr.mesh, l.g_cmds, (render3d_st.sc_ind_base + i * render3d_st.sc_ind_stride) * SC_REC_W, render3d_st.sc_ind_n, 0, 0) }
|
|
else { mesh_draw_instanced(render3d_st, pr.mesh, cnt) }
|
|
}
|
|
gpu_alpha_to_coverage(render3d_st, false)
|
|
if p == render3d_st.sc_prog_fol_eq { gpu_depth_func(render3d_st, GL_LESS) }
|
|
}
|
|
|
|
function layer_draw_far(render3d_st: mut Render3dState, l: Layer, shadow: bool, light_vp: floats) -> void {
|
|
if l.imp == null or (l.n_far == 0 and not l.g_on) { return }
|
|
var p = render3d_st.sc_imp_prog
|
|
if shadow { p = render3d_st.sc_imp_prog_shadow }
|
|
gpu_use_program(render3d_st, p)
|
|
let im = l.imp
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_radius"), im.radius)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_height"), im.height)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_tiles"), float(im.tiles))
|
|
r3d_bind_2d(render3d_st, p, "u_atlas_albedo", 0, im.albedo)
|
|
if shadow {
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_light_vp"), light_vp)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_face_dir"), render3d_st.sun_dir)
|
|
if render3d_st.r3d_debug_shadow and not render3d_st.sc_printed { render3d_st.sc_printed = true; print(`imp shadow prog {p} face_dir loc {gpu_uniform(render3d_st, p, "u_face_dir")} sun {fixed(render3d_st.sun_dir[0])} {fixed(render3d_st.sun_dir[1])} {fixed(render3d_st.sun_dir[2])} cam {fixed(render3d_st.cam_pos[0])} {fixed(render3d_st.cam_pos[1])} {fixed(render3d_st.cam_pos[2])} n_far {l.n_far}`) }
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_cam_pos"), render3d_st.cam_pos)
|
|
} else {
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_view"), render3d_st.cam_view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_proj"), render3d_st.cam_proj)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), l.tint)
|
|
if render3d_st.sc_dbg_lod { if render3d_st.sc_dbg_tint == null { render3d_st.sc_dbg_tint = v3_new(1.0, 1.0, 1.0) }; v3_set(render3d_st.sc_dbg_tint, 3.0, 0.0, 3.0); u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), render3d_st.sc_dbg_tint) }
|
|
r3d_bind_2d(render3d_st, p, "u_atlas_normal", 1, im.normal)
|
|
sky_bind_lighting(render3d_st, p)
|
|
shadow_bind(render3d_st, p)
|
|
fog_bind(render3d_st, p)
|
|
if l.foliage { u_f(render3d_st, gpu_uniform(render3d_st, p, "u_spec_scale"), 0.05) }
|
|
}
|
|
gpu_cull(render3d_st, false)
|
|
if not shadow and render3d_st.sc_a2c { gpu_alpha_to_coverage(render3d_st, true) }
|
|
if l.g_on {
|
|
scatter_attach(render3d_st, render3d_st.sc_card, l.g_dst)
|
|
gpu_draw_mesh_indirect(render3d_st, render3d_st.sc_card, l.g_cmds, 16 * SC_REC_W, 1, 0, 0)
|
|
} else {
|
|
scatter_attach(render3d_st, render3d_st.sc_card, l.imp_buf)
|
|
mesh_draw_instanced(render3d_st, render3d_st.sc_card, l.n_far)
|
|
}
|
|
gpu_alpha_to_coverage(render3d_st, false)
|
|
}
|
|
|
|
# Cast from the impostor card, always, for every instance in the layer. The lit pass may
|
|
# swap a scanned mesh in up close; the shadow must not, or a tree's shadow changes shape
|
|
# as you approach it. The card is also far the cheaper of the two, which is what pays for
|
|
# casting the whole set into all five cascades.
|
|
function layer_draw_shadow(render3d_st: mut Render3dState, l: Layer, light_vp: floats) -> void {
|
|
if l.n_sh == 0 or l.imp == null { return }
|
|
let p = render3d_st.sc_imp_prog_shadow
|
|
gpu_use_program(render3d_st, p)
|
|
let im = l.imp
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_radius"), im.radius)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_height"), im.height)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_tiles"), float(im.tiles))
|
|
r3d_bind_2d(render3d_st, p, "u_atlas_albedo", 0, im.albedo)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_light_vp"), light_vp)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_face_dir"), render3d_st.sun_dir)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_cam_pos"), render3d_st.cam_pos)
|
|
gpu_cull(render3d_st, false)
|
|
let buf = fog_casters(render3d_st, l)
|
|
let n = fog_casters_n(render3d_st, l)
|
|
if n == 0 { return }
|
|
scatter_attach(render3d_st, render3d_st.sc_card, buf)
|
|
mesh_draw_instanced(render3d_st, render3d_st.sc_card, n)
|
|
}
|
|
|
|
|
|
# ---- the foliage depth prepass -----------------------------------------------------------
|
|
# Exactly the instances and positions layer_draw_model will light (the same LOD buckets,
|
|
# the same vertex shader, the same ground and wind), into depth only.
|
|
function layer_draw_depth(render3d_st: mut Render3dState, l: Layer, model: Model, vb: int, cnt: int) -> void {
|
|
if cnt == 0 or model == null or vb == 0 { return }
|
|
let p = render3d_st.sc_prog_fol_depth
|
|
gpu_use_program(render3d_st, p)
|
|
var ground = 0.0
|
|
if l.grounded { ground = 1.0 }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_ground"), ground)
|
|
if l.grounded { terrain_bind_height(render3d_st, p) }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_time"), render3d_st.r3d_time)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_wind"), l.wind)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_flutter"), l.flutter)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_model_h"), model.height)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_view"), render3d_st.cam_view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_proj"), render3d_st.cam_proj)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_clip_y"), render3d_st.r3d_clip_y)
|
|
gpu_cull(render3d_st, false)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
scatter_attach(render3d_st, pr.mesh, vb)
|
|
r3d_bind_2d(render3d_st, p, "u_diff", 0, pr.diff)
|
|
if render3d_st.sc_ind_base >= 0 { gpu_draw_mesh_indirect(render3d_st, pr.mesh, l.g_cmds, (render3d_st.sc_ind_base + i * render3d_st.sc_ind_stride) * SC_REC_W, render3d_st.sc_ind_n, 0, 0) }
|
|
else { mesh_draw_instanced(render3d_st, pr.mesh, cnt) }
|
|
}
|
|
}
|
|
|
|
# every foliage mesh draw the lit pass will make with sc_prog_fol_eq: not blades, not
|
|
# flowers, not card levels (those keep their own alpha and draw as before)
|
|
function scatter_draw_depth(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.sc_prog_fol_depth == 0 { return }
|
|
for i in 0 .. len(render3d_st.sc_layers) {
|
|
let l = render3d_st.sc_layers[i]
|
|
if not l.foliage or l.blade or l.flower { continue }
|
|
if l.grass and sc_grass_replaced(render3d_st) { continue }
|
|
if render3d_st.r3d_no_trees and l.imp != null { continue }
|
|
layer_update(render3d_st, l)
|
|
if l.n_lods > 1 and l.g_on {
|
|
render3d_st.sc_ind_base = 0; render3d_st.sc_ind_n = l.n_lods
|
|
layer_draw_depth(render3d_st, l, l.g_model, l.g_dst, 1)
|
|
render3d_st.sc_ind_base = -1; render3d_st.sc_ind_n = 1
|
|
} else if l.n_lods > 1 {
|
|
# (a tree layer is flagged `card` for its distant level; its mesh levels still count)
|
|
for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { layer_draw_depth(render3d_st, l, l.lods[k], l.lod_buf[k], l.n_lod[k]) } }
|
|
} else if not l.card {
|
|
layer_draw_depth(render3d_st, l, l.model, l.buf, l.n_near)
|
|
}
|
|
}
|
|
gpu_cull(render3d_st, true)
|
|
}
|
|
# a layer the game flagged `grass` is ground cover the GPU blades stand in for: off while they draw
|
|
function sc_grass_replaced(render3d_st: Render3dState) -> bool { return render3d_st.sc_skip_grass or ((render3d_st.grass_on or render3d_st.grass_force) and not render3d_st.grass_env_off and render3d_st.grass_prog != 0) }
|
|
function scatter_draw(render3d_st: mut Render3dState) -> void {
|
|
for i in 0 .. len(render3d_st.sc_layers) {
|
|
let l = render3d_st.sc_layers[i]
|
|
if render3d_st.sc_skip_blade and l.blade { continue }
|
|
if render3d_st.sc_skip_flower and l.flower { continue }
|
|
if render3d_st.sc_skip_card and l.card { continue }
|
|
if l.grass and sc_grass_replaced(render3d_st) { continue }
|
|
if render3d_st.r3d_no_trees and l.imp != null and not l.card { continue }
|
|
layer_update(render3d_st, l)
|
|
layer_draw_near(render3d_st, l, false, null, false)
|
|
layer_draw_far(render3d_st, l, false, null)
|
|
}
|
|
gpu_cull(render3d_st, true)
|
|
}
|
|
# Nothing here is keyed off the cascade. Every skip that used to be — ground cover past
|
|
# the 250 m cascade, blades past the nearest, the scanned mesh past the second — made a
|
|
# whole class of caster vanish at a fixed distance, which is exactly the popping. A layer
|
|
# with an impostor now casts its entire instance list from the card in every cascade;
|
|
# only layers that have no impostor at all fall back to the mesh.
|
|
const SC_IMP_CAST_CASCADE: int = 2
|
|
function scatter_draw_casters(render3d_st: mut Render3dState, light_vp: floats) -> void {
|
|
for i in 0 .. len(render3d_st.sc_layers) {
|
|
let l = render3d_st.sc_layers[i]
|
|
if render3d_st.sc_skip_blade and l.blade { continue }
|
|
if render3d_st.sc_skip_card and l.card { continue }
|
|
if l.grass and sc_grass_replaced(render3d_st) { continue }
|
|
if render3d_st.r3d_no_trees and l.imp != null and not l.card { continue }
|
|
layer_update(render3d_st, l)
|
|
if l.imp != null {
|
|
# the impostor casts only into the far cascades: every tree on the map was drawn as a card
|
|
# into the near two as well, whose receivers (within 60 m) are shaded by the near trees'
|
|
# own LOD2 meshes, cast just below
|
|
if render3d_st.sh_cascade >= SC_IMP_CAST_CASCADE { layer_draw_shadow(render3d_st, l, light_vp) }
|
|
# Shadow LOD (the practice in every production engine: a caster uses a low mesh LOD,
|
|
# the billboard only far away). The card alone is a side-view silhouette and a
|
|
# crown of drooping needle cards is mostly slivers from the side, so the sun, which
|
|
# sees the crown from above, cast a trunk line with a few blobs. The near levels
|
|
# now also cast their LOD2 mesh, alpha-tested, on top of the card.
|
|
if l.n_lods > 2 and l.g_on {
|
|
render3d_st.sc_ind_base = 17; render3d_st.sc_ind_n = 3; render3d_st.sc_ind_stride = 3
|
|
layer_draw_model(render3d_st, l, l.g_model_sh, l.g_dst, 1, false, true, light_vp)
|
|
render3d_st.sc_ind_base = -1; render3d_st.sc_ind_n = 1; render3d_st.sc_ind_stride = 4
|
|
} else if l.n_lods > 2 {
|
|
for k in 0 .. 3 { layer_draw_model(render3d_st, l, l.lods[2], l.lod_buf[k], l.n_lod[k], false, true, light_vp) }
|
|
}
|
|
}
|
|
else { layer_draw_near(render3d_st, l, true, light_vp, true) }
|
|
}
|
|
prof_cpu_mark(render3d_st, "shadow scatter")
|
|
}
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# The distant-grass carpet: the clump cards rendered straight down into one tiling
|
|
# tile, so the ground beyond the blade rings carries the same clumps, colours and
|
|
# gaps as the near cover instead of a lawn scan (the far-field trick the big open
|
|
# worlds use: geometry up close, an authored ground texture that matches it beyond).
|
|
# Returns an RGBA8 texture (alpha = coverage), repeat-wrapped and mipmapped.
|
|
function cb_rnd(render3d_st: mut Render3dState) -> float {
|
|
render3d_st.cb_state = (render3d_st.cb_state * 1103515245 + 12345) & 0x7FFFFFFF
|
|
return float((render3d_st.cb_state >> 8) & 0xFFFF) / 65536.0
|
|
}
|
|
function carpet_bake(render3d_st: mut Render3dState, layers: []Layer, count: int, tile: float, res: int) -> int {
|
|
let tex = tex_target(render3d_st, res, res, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
let fbo = gpu_fb_new(render3d_st)
|
|
gpu_fb_bind(render3d_st, fbo)
|
|
gpu_fb_color(render3d_st, 0, tex)
|
|
let rb = gpu_rb_new(render3d_st)
|
|
gpu_rb_storage(render3d_st, rb, GL_DEPTH_COMPONENT24, res, res, 0)
|
|
gpu_fb_depth_rb(render3d_st, rb)
|
|
gpu_fb_draw_buffers(render3d_st, 1)
|
|
gpu_viewport(render3d_st, 0, 0, res, res)
|
|
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 0.0)
|
|
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
|
gpu_depth_test(render3d_st, true)
|
|
gpu_depth_func(render3d_st, GL_LESS)
|
|
gpu_cull(render3d_st, false)
|
|
gpu_blend(render3d_st, false)
|
|
# the clumps, and eight wrapped copies so the tile's edges continue
|
|
let half = tile * 0.5
|
|
let n9 = count * 9
|
|
let inst = gl_floats(n9 * INST_FLOATS)
|
|
render3d_st.cb_state = 977
|
|
var k = 0
|
|
for i in 0 .. count {
|
|
let x = cb_rnd(render3d_st) * tile - half
|
|
let z = cb_rnd(render3d_st) * tile - half
|
|
let sc = 1.5 + cb_rnd(render3d_st)
|
|
let yaw = cb_rnd(render3d_st) * (2.0 * PI)
|
|
let sd = cb_rnd(render3d_st)
|
|
for oz in 0 .. 3 {
|
|
for ox in 0 .. 3 {
|
|
let px = x + float(ox - 1) * tile
|
|
let pz = z + float(oz - 1) * tile
|
|
gl_put_bits(inst, k, float_bits(px)); gl_put_bits(inst, k + 1, float_bits(0.0)); gl_put_bits(inst, k + 2, float_bits(pz)); gl_put_bits(inst, k + 3, float_bits(sc))
|
|
gl_put_bits(inst, k + 4, float_bits(Math.sin(yaw))); gl_put_bits(inst, k + 5, float_bits(Math.cos(yaw))); gl_put_bits(inst, k + 6, float_bits(sd)); gl_put_bits(inst, k + 7, float_bits(0.0))
|
|
k += INST_FLOATS
|
|
}
|
|
}
|
|
}
|
|
let buf = gpu_buffer_new(render3d_st)
|
|
gpu_buffer_upload(render3d_st, buf, gl_bytes_of(n9 * INST_FLOATS), inst, GPU_STATIC)
|
|
free(inst)
|
|
# straight down: the window is exactly one tile
|
|
let view = m4_new(); let proj = m4_new()
|
|
let eye = v3_new(0.0, 6.0, 0.0); let at = v3_new(0.0, 0.0, 0.0); let up = v3_new(0.0, 0.0, -1.0)
|
|
m4_look_at(view, eye, at, up)
|
|
m4_ortho(proj, -half, half, -half, half, 0.1, 12.0)
|
|
let bake = render3d_st.sc_bake_card_prog
|
|
gpu_use_program(render3d_st, bake)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, bake, "u_view"), view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, bake, "u_proj"), proj)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_wind"), 0.0)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_flutter"), 0.0)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_time"), 0.0)
|
|
for li in 0 .. len(layers) {
|
|
let l = layers[li]
|
|
if l.atlas == null { continue }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_card_w"), l.atlas.radius * 2.0); u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_card_h"), l.atlas.height)
|
|
r3d_bind_2d(render3d_st, bake, "u_diff", 0, l.atlas.albedo)
|
|
r3d_bind_2d(render3d_st, bake, "u_arm", 2, l.atlas.normal)
|
|
for i in 0 .. len(l.model.prims) {
|
|
let pr = l.model.prims[i]
|
|
scatter_attach(render3d_st, pr.mesh, buf)
|
|
mesh_draw_instanced(render3d_st, pr.mesh, n9)
|
|
}
|
|
}
|
|
free(view); free(proj); free(eye); free(at); free(up)
|
|
gpu_fb_bind(render3d_st, 0)
|
|
gpu_fb_free(render3d_st, fbo)
|
|
gpu_rb_free(render3d_st, rb)
|
|
gpu_buffer_free(render3d_st, buf)
|
|
gpu_tex_bind(render3d_st, GPU_TEX2D, tex)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_REPEAT)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_paramf(render3d_st, GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, render3d_st.tex_anisotropy)
|
|
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
|
gpu_check(render3d_st, "carpet bake")
|
|
return tex
|
|
}
|
|
|
|
# ---- another map at run time ------------------------------------------------------------
|
|
# Every scattered layer released - its instance arrays and GL buffers, a card layer's own
|
|
# crossed-card mesh and baked atlas - and every stream feeding them. The models a layer drew
|
|
# belong to whoever loaded them and are kept, and so are the programs. A game rebuilding its
|
|
# world calls this, then places the new map's layers exactly as it did at boot.
|
|
function scatter_clear_all(render3d_st: mut Render3dState) -> void {
|
|
stream_clear_all(render3d_st)
|
|
if render3d_st.sc_layers == null { return }
|
|
for i in 0 .. len(render3d_st.sc_layers) {
|
|
let l = render3d_st.sc_layers[i]
|
|
if l.buf != 0 { gpu_buffer_free(render3d_st, l.buf) }
|
|
if l.imp_buf != 0 { gpu_buffer_free(render3d_st, l.imp_buf) }
|
|
if l.sh_buf != 0 { gpu_buffer_free(render3d_st, l.sh_buf) }
|
|
if l.fog_buf != 0 { gpu_buffer_free(render3d_st, l.fog_buf) }
|
|
if l.lod_buf != null { for k in 0 .. l.n_lods { gpu_buffer_free(render3d_st, l.lod_buf[k]) } }
|
|
if l.inst != null { free(l.inst) }
|
|
if l.scratch != null { free(l.scratch) }
|
|
if l.tint != null { free(l.tint) }
|
|
if l.last_cam != null { free(l.last_cam) }
|
|
if l.gstart != null { free(l.gstart) }
|
|
if l.gsorted != null { free(l.gsorted) }
|
|
if l.gymin != null { free(l.gymin) }
|
|
if l.gymax != null { free(l.gymax) }
|
|
if l.vis != null { free(l.vis) }
|
|
if l.lod_dist != null { free(l.lod_dist); free(l.lod_card); free(l.lod_buf); free(l.n_lod) }
|
|
if l.lvl != null { free(l.lvl) }
|
|
if l.g_arena != null { for k in 0 .. len(l.g_arena) { mesh_free(render3d_st, l.g_arena[k].mesh) } }
|
|
# layer_cards built this layer's crossed card and its atlas itself
|
|
if l.card and l.n_lods == 0 and l.model != null { for k in 0 .. len(l.model.prims) { mesh_free(render3d_st, l.model.prims[k].mesh) } }
|
|
if l.card and l.atlas != null {
|
|
gpu_tex_free(render3d_st, l.atlas.albedo)
|
|
gpu_tex_free(render3d_st, l.atlas.normal)
|
|
}
|
|
}
|
|
render3d_st.sc_layers = new []Layer
|
|
render3d_st.sc_view_gen += 1
|
|
}
|
|
|
|
# R3D_DUMP_ATLAS / the LOD debug switch: what a layer holds, printed (text made only then)
|
|
@alloc_ok("debug output, only under R3D_DUMP_ATLAS and the LOD debug switch")
|
|
function sc_debug_lods(render3d_st: Render3dState, l: Layer, counts: words, n: int, total: int, src: floats) -> void {
|
|
print(`lod partition: total {total} dropped {counts[n + 1]} far {counts[n]} l0 {counts[0]} l1 {counts[1]} l2 {counts[2]} l3 {counts[3]} dist0 {fixed(l.lod_dist[0])} dist3 {fixed(l.lod_dist[n - 1])} cull {fixed(l.cull)} cam {fixed(render3d_st.cam_pos[0])} {fixed(render3d_st.cam_pos[2])} first {fixed(src[0])} {fixed(src[2])}`)
|
|
}
|
|
@alloc_ok("debug output, only under R3D_DUMP_ATLAS and the LOD debug switch")
|
|
function sc_debug_layer(l: Layer, tmp: floats, nn: int, nf: int) -> void {
|
|
if l.imp != null and nn > 0 { print(`near full-mesh instances: {nn} (first at {fixed(tmp[0])} {fixed(tmp[1])} {fixed(tmp[2])})`) }
|
|
if l.imp != null {
|
|
print(`layer: near {nn} far {nf}`)
|
|
for k in 0 .. nn {
|
|
let q = k * INST_FLOATS
|
|
print(` near {fixed(tmp[q])} {fixed(tmp[q + 1])} {fixed(tmp[q + 2])} s {fixed(tmp[q + 3])}`)
|
|
}
|
|
}
|
|
}
|