ludic.lab: lab_png_write / lab_png_write_from (raw 8-bit, 1-4 channels, stored deflate) in png_write.ludic, importable alone with its own LabPngState; png_convert.ludic's previews of float textures (R32F min..max, RG16F x255, HDR x/(1+x) + sRGB); lab_ppm_to_png on the same encoder. render3d: bake_load.ludic - impostor_from_baked / impostor_source / impostor_refill (a fog re-open reads the bake), sky_baked_in and sky_precompute trying the bake at the start yaw (sky_compute is the convolution, and sky_ibl_bytes always uses it), carpet_from_baked / carpet_bytes / carpet_finish, bake_part_count / _len / _off. Compile-only: nothing run. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1304 lines
66 KiB
Text
1304 lines
66 KiB
Text
# ============================================================================
|
|
# scatter.ludic — instanced vegetation and props. A Layer is one model placed
|
|
# many times (position, scale, yaw, seed, wind weight per instance). Each frame
|
|
# the instances are split by distance: the near ones draw as the full scanned
|
|
# mesh, the far ones as impostor cards baked from that mesh at load.
|
|
# ============================================================================
|
|
|
|
const INST_FLOATS: int = 8
|
|
|
|
property Impostor {
|
|
albedo: int = 0,
|
|
normal: int = 0,
|
|
tiles: int = 16,
|
|
radius: float = 0.0,
|
|
height: float = 0.0,
|
|
model: Model, # what it was baked from, to bake it again (fog_impostors.ludic)
|
|
tw: int = 0,
|
|
th: int = 0,
|
|
flower: bool = false,
|
|
src_path: string = "", # its bake file, where the game names one (bake_load.ludic): what a fog opening reads
|
|
src_key: string = "",
|
|
src_version: int = 0,
|
|
src_ok: bool = false # its atlases came from that bake, so a re-open reads it rather than painting
|
|
}
|
|
property Layer {
|
|
model: Model,
|
|
imp: Impostor,
|
|
foliage: bool = false,
|
|
wind: float = 0.0, # float bits
|
|
flutter: float = 0.0, # float bits; per-leaf tremble (aspen), 0 = none
|
|
tint: floats,
|
|
inst: floats, # INST_FLOATS per instance
|
|
count: int = 0,
|
|
cap: int = 0,
|
|
have: int = 0, # instances inst and scratch have room for now: grown toward cap as filled
|
|
near: float = 0.0, # float bits; instances beyond it draw as impostors (or not at all)
|
|
cull: float = 0.0, # float bits; instances beyond it are skipped (0 = never)
|
|
buf: int = 0,
|
|
n_near: int = 0,
|
|
imp_buf: int = 0,
|
|
sh_buf: int = 0, # every instance, for shadow casting (no cull, no LOD split)
|
|
n_sh: int = 0,
|
|
fog_buf: int = 0, # the casters inside the fog wall (fog_casters.ludic)
|
|
fog_n: int = 0,
|
|
fog_src: int = -1,
|
|
ck_off: words, # per baked chunk: where its slab starts in inst, or -1 (layer_chunks.ludic)
|
|
ck_cnt: words,
|
|
lc_rec: floats, # one decoded record, reused
|
|
fog_all: bool = false, # the wall keeps (nearly) all of it: casters from sh_buf, no rebuild
|
|
fog_wall: float = 0.0,
|
|
fog_x: float = 0.0,
|
|
fog_z: float = 0.0,
|
|
n_far: int = 0,
|
|
scratch: floats,
|
|
last_cam: floats,
|
|
rough: float = 0.0,
|
|
blade: bool = false,
|
|
grass: bool = false, # ground cover the GPU blades replace: skipped while they draw (and under R3D_NOGRASS)
|
|
flower: bool = false,
|
|
card: bool = false,
|
|
cheap: bool = false, # distant cover: no shadows, no wind, flat lighting
|
|
atlas: Impostor,
|
|
streamed: bool = false, # fed by a Stream: already frustum-culled per chunk, no split needed
|
|
grounded: bool = false, # the vertex shader stands each instance on the drawn terrain
|
|
view_gen: int = -1, # sc_view_gen this layer's partition was built for
|
|
# static layers with many instances are sorted into a cell grid once, and only the
|
|
# cells inside the view frustum (and within cull) are partitioned each frame
|
|
gcell: float = 0.0, # cell size (float bits); 0 = no grid
|
|
gx0: float = 0.0,
|
|
gz0: float = 0.0,
|
|
gnx: int = 0,
|
|
gnz: int = 0,
|
|
gstart: words, # per cell: first index into gsorted (ncell + 1 entries)
|
|
gsorted: floats, # the instances, grouped by cell
|
|
gymin: floats, # per cell height range (float bits)
|
|
gymax: floats,
|
|
vis: floats, # the instances gathered from visible cells this frame
|
|
n_vis: int = 0,
|
|
# A LOD chain: lods[k] is drawn for instances within lod_dist[k] (and beyond lod_dist[k-1]);
|
|
# past the last level the impostor takes over (or, if the last distance is 0, the last
|
|
# level runs out to the cull distance). lod_card[k] = 1 marks a level that is the layer's
|
|
# crossed card carrying its atlas (cover keeps its baked card as the far level).
|
|
lods: []Model,
|
|
n_lods: int = 0,
|
|
# The GPU-culled path (Vulkan, phase 38): every instance in g_src, a compute pass packs the
|
|
# visible ones per bucket into g_dst and writes the instance counts of the draw records in
|
|
# g_cmds (see scatter_cull.comp for the record layout). g_on once it is set up for g_n instances.
|
|
g_on: bool = false,
|
|
g_n: int = 0,
|
|
g_src: int = 0,
|
|
g_dst: int = 0,
|
|
g_cmds: int = 0,
|
|
g_counts: int = 0,
|
|
g_arena: []Prim, # one merged mesh per material, holding every level's copy (layer_arena_build)
|
|
g_first: words, # material * 4 + level: that level's first index in the merged mesh
|
|
g_base: words, # ... and its first vertex
|
|
g_model: Model, # the merged meshes as a model the draws take (level 0's height)
|
|
g_model_sh: Model, # the same with level 2's height, for the shadow LOD
|
|
lod_dist: floats,
|
|
lod_card: words,
|
|
lod_buf: words,
|
|
n_lod: words,
|
|
lvl: words # scratch: the level chosen per gathered instance
|
|
}
|
|
|
|
# the procedural models' one layout: position, normal, uv, interleaved at 32 bytes
|
|
function sc_model_layout(render3d_st: Render3dState, m: Mesh) -> void {
|
|
gpu_mesh_attr(render3d_st, m, 0, 3, GPU_F32, 32, 0, false)
|
|
gpu_mesh_attr(render3d_st, m, 1, 3, GPU_F32, 32, 12, false)
|
|
gpu_mesh_attr(render3d_st, m, 2, 2, GPU_F32, 32, 24, false)
|
|
}
|
|
|
|
# Two crossed unit quads (x in [-0.5, 0.5], y in [0, 1]), attribute 0 = pos,
|
|
# 1 = the quad's facing normal, 2 = uv. Scaled per layer to the atlas card size.
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function model_cross_card(render3d_st: mut Render3dState) -> Model {
|
|
let model = new Model
|
|
model.prims = new []Prim
|
|
let pr = new Prim
|
|
let m = gpu_mesh_new(render3d_st)
|
|
let v = gl_floats(8 * 8)
|
|
var k = 0
|
|
for q in 0 .. 2 {
|
|
for c in 0 .. 4 {
|
|
var sx = -0.5; var sy = 0.0; var u = 0.0; var vv = 0.0
|
|
if c == 1 or c == 2 { sx = 0.5; u = 1.0 }
|
|
if c == 2 or c == 3 { sy = 1.0; vv = 1.0 }
|
|
if q == 0 { gl_put_bits(v, k, float_bits(sx)); gl_put_bits(v, k + 1, float_bits(sy)); gl_put_bits(v, k + 2, float_bits(0.0)); gl_put_bits(v, k + 3, float_bits(0.0)); gl_put_bits(v, k + 4, float_bits(0.0)); gl_put_bits(v, k + 5, float_bits(-1.0)) }
|
|
else { gl_put_bits(v, k, float_bits(0.0)); gl_put_bits(v, k + 1, float_bits(sy)); gl_put_bits(v, k + 2, float_bits(sx)); gl_put_bits(v, k + 3, float_bits(-1.0)); gl_put_bits(v, k + 4, float_bits(0.0)); gl_put_bits(v, k + 5, float_bits(0.0)) }
|
|
gl_put_bits(v, k + 6, float_bits(u)); gl_put_bits(v, k + 7, float_bits(vv))
|
|
k += 8
|
|
}
|
|
}
|
|
gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(64), GPU_STATIC)
|
|
sc_model_layout(render3d_st, m)
|
|
free(v)
|
|
let idx = words(12)
|
|
idx[0] = 0; idx[1] = 1; idx[2] = 2; idx[3] = 0; idx[4] = 2; idx[5] = 3
|
|
idx[6] = 4; idx[7] = 5; idx[8] = 6; idx[9] = 4; idx[10] = 6; idx[11] = 7
|
|
gpu_mesh_indices(render3d_st, m, data_of(idx), 48, 4)
|
|
free(idx)
|
|
m.count = 12
|
|
gpu_mesh_done(render3d_st, m)
|
|
pr.mesh = m
|
|
if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) }
|
|
pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white
|
|
push(model.prims, pr)
|
|
model.radius = 0.5; model.height = 1.0; model.tris = 4
|
|
return model
|
|
}
|
|
|
|
# A card layer: crossed cards carrying a single-tile atlas baked from `scan`.
|
|
# An aspen leaf hangs on a flattened stalk and turns in air nothing else feels. Give the
|
|
# layer a flutter and its leaves tremble and flash their pale undersides; everything else
|
|
# leaves it at zero. It is per layer rather than per instance because a species quakes or
|
|
# it does not.
|
|
function layer_flutter(l: Layer, v: float) -> void { l.flutter = v }
|
|
function layer_cards(render3d_st: mut Render3dState, scan: Model, cap: int, wind: float, cull: float) -> Layer {
|
|
let l = layer_new(render3d_st, model_cross_card(render3d_st), cap, true, wind, 0.0, cull)
|
|
l.card = true
|
|
l.atlas = impostor_bake(render3d_st, scan, 1, 512, 512)
|
|
return l
|
|
}
|
|
|
|
# A procedural grass blade (1 m tall, 5 cm wide, curved): 5 rows of 2 vertices.
|
|
@alloc_ok("a model: made once by the program that asks for it, and kept with its layer")
|
|
function model_blade(render3d_st: mut Render3dState) -> Model {
|
|
let model = new Model
|
|
model.prims = new []Prim
|
|
let pr = new Prim
|
|
let m = gpu_mesh_new(render3d_st)
|
|
let rows = 5
|
|
let v = gl_floats(rows * 2 * 8)
|
|
var k = 0
|
|
for r in 0 .. rows {
|
|
let t = float(r) / float(rows - 1)
|
|
# never a zero-width tip: a sliver triangle extrapolates its attributes wildly
|
|
let taper = Math.max(1.0 - t * (t * Math.sqrt(t)), 0.12)
|
|
let hw = 0.05 * taper
|
|
let bend = t * t * 0.28
|
|
for sd in 0 .. 2 {
|
|
var x = -hw
|
|
if sd == 1 { x = hw }
|
|
gl_put_bits(v, k, float_bits(x)); gl_put_bits(v, k + 1, float_bits(t)); gl_put_bits(v, k + 2, float_bits(bend))
|
|
gl_put_bits(v, k + 3, float_bits(0.0)); gl_put_bits(v, k + 4, float_bits(0.3)); gl_put_bits(v, k + 5, float_bits(1.0))
|
|
gl_put_bits(v, k + 6, float_bits(float(sd))); gl_put_bits(v, k + 7, float_bits(t))
|
|
k += 8
|
|
}
|
|
}
|
|
gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(rows * 2 * 8), GPU_STATIC)
|
|
sc_model_layout(render3d_st, m)
|
|
free(v)
|
|
let ni = (rows - 1) * 6
|
|
let idx = words(ni)
|
|
k = 0
|
|
for r in 0 .. rows - 1 {
|
|
let a = r * 2
|
|
idx[k] = a; idx[k + 1] = a + 1; idx[k + 2] = a + 2
|
|
idx[k + 3] = a + 1; idx[k + 4] = a + 3; idx[k + 5] = a + 2
|
|
k += 6
|
|
}
|
|
gpu_mesh_indices(render3d_st, m, data_of(idx), ni * 4, 4)
|
|
free(idx)
|
|
m.count = ni
|
|
gpu_mesh_done(render3d_st, m)
|
|
pr.mesh = m
|
|
if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) }
|
|
pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white
|
|
push(model.prims, pr)
|
|
model.radius = 0.05; model.height = 1.0; model.tris = ni / 3
|
|
return model
|
|
}
|
|
|
|
# The foliage depth prepass (render.ludic): the near tree LODs write depth first with a
|
|
# shader that only runs the alpha test, then the lit pass shades them with no discard and
|
|
# an equal depth test, so a pixel of needles is lit once rather than once for every card
|
|
# stacked behind it. In a dense stand at 4K that overdraw was the largest pass in the frame.
|
|
|
|
function scatter_init(render3d_st: mut Render3dState) -> void {
|
|
# R3D_DUMP_ATLAS: every impostor and card atlas the run bakes, to build/atlas_<n>_{color,alpha}.ppm
|
|
render3d_st.sc_debug_dump = r3d_env_has(render3d_st, "R3D_DUMP_ATLAS")
|
|
render3d_st.sc_prog = r3d_program(render3d_st, "model.vert", "model.frag", "")
|
|
render3d_st.sc_prog_fol = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define ALPHA_TEST\n#define NEAR_FADE\n")
|
|
render3d_st.sc_prog_fol_depth = r3d_program(render3d_st, "model.vert", "depth.frag", "#define FOLIAGE\n#define WIND\n#define ALPHA_TEST\n#define NEAR_FADE\n")
|
|
render3d_st.sc_prog_fol_eq = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define EQ_PASS\n#define NEAR_FADE\n")
|
|
render3d_st.sc_prog_wind = r3d_program(render3d_st, "model.vert", "model.frag", "#define WIND\n")
|
|
render3d_st.sc_prog_blade = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define BLADE\n")
|
|
render3d_st.sc_prog_flower = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define FLOWER\n")
|
|
render3d_st.sc_prog_card = r3d_program(render3d_st, "model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define CARD\n")
|
|
render3d_st.sc_prog_card_shadow = r3d_program(render3d_st, "model.vert", "model.frag", "#define SHADOW_PASS\n#define WIND\n#define CARD\n")
|
|
render3d_st.sc_prog_card_cheap = r3d_program(render3d_st, "model.vert", "model.frag", "#define CARD\n#define CHEAP\n")
|
|
# a dry alpine meadow: brown-olive roots, straw with a little green at the tips
|
|
render3d_st.sc_blade_base = v3_new(0.045, 0.06, 0.025)
|
|
render3d_st.sc_blade_tip = v3_new(0.22, 0.27, 0.13)
|
|
render3d_st.sc_blade_tint = v3_new(1.0, 1.0, 1.0)
|
|
render3d_st.sc_prog_shadow = r3d_program(render3d_st, "model.vert", "shadow.frag", "#define SHADOW_PASS\n")
|
|
render3d_st.sc_prog_shadow_wind = r3d_program(render3d_st, "model.vert", "shadow.frag", "#define SHADOW_PASS\n#define WIND\n")
|
|
render3d_st.sc_prog_shadow_fol = r3d_program(render3d_st, "model.vert", "shadow.frag", "#define SHADOW_PASS\n#define WIND\n#define ALPHA_TEST\n")
|
|
render3d_st.sc_imp_prog = r3d_program(render3d_st, "impostor.vert", "impostor.frag", "")
|
|
render3d_st.sc_imp_prog_shadow = r3d_program(render3d_st, "impostor.vert", "impostor.frag", "#define SHADOW_PASS\n")
|
|
render3d_st.sc_bake_prog = r3d_program(render3d_st, "model.vert", "bake.frag", "")
|
|
render3d_st.sc_bake_flower_prog = r3d_program(render3d_st, "model.vert", "bake.frag", "#define FLOWER\n")
|
|
render3d_st.sc_bake_card_prog = r3d_program(render3d_st, "model.vert", "bake.frag", "#define CARD\n")
|
|
render3d_st.sc_card = mesh_card(render3d_st)
|
|
# a single identity instance, for baking
|
|
let one = gl_floats(INST_FLOATS)
|
|
for i in 0 .. INST_FLOATS { gl_put_bits(one, i, float_bits(0.0)) }
|
|
gl_put_bits(one, 3, float_bits(1.0)); gl_put_bits(one, 5, float_bits(1.0))
|
|
render3d_st.sc_ident_buf = gpu_buffer_new(render3d_st)
|
|
gpu_buffer_upload(render3d_st, render3d_st.sc_ident_buf, INST_FLOATS * 4, one, GPU_STATIC)
|
|
free(one)
|
|
render3d_st.sc_layers = new []Layer
|
|
}
|
|
|
|
# feed a mesh its instances from `buf`: attribute 3 = position + scale, 4 = sin, cos, seed, wind
|
|
function scatter_attach(render3d_st: Render3dState, m: Mesh, buf: int) -> void {
|
|
gpu_mesh_bind_instances(render3d_st, m, buf)
|
|
gpu_mesh_attr_inst(render3d_st, m, 3, 4, GPU_F32, INST_FLOATS * 4, 0)
|
|
gpu_mesh_attr_inst(render3d_st, m, 4, 4, GPU_F32, INST_FLOATS * 4, 16)
|
|
gpu_mesh_done(render3d_st, m)
|
|
}
|
|
|
|
# the GPU memory it makes is counted as VKM_SCATTER (R3D_VKMEM)
|
|
function layer_new(render3d_st: mut Render3dState, model: Model, cap: int, foliage: bool, wind: float, near: float, cull: float) -> Layer {
|
|
let was = render3d_st.gvk_tag
|
|
render3d_st.gvk_tag = VKM_SCATTER
|
|
let r = layer_new__t(render3d_st, model, cap, foliage, wind, near, cull)
|
|
render3d_st.gvk_tag = was
|
|
return r
|
|
}
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function layer_new__t(render3d_st: mut Render3dState, model: Model, cap: int, foliage: bool, wind: float, near: float, cull: float) -> Layer {
|
|
let l = new Layer
|
|
l.model = model
|
|
l.cap = cap
|
|
l.foliage = foliage
|
|
l.wind = wind
|
|
l.near = near
|
|
l.cull = cull
|
|
l.tint = v3_new(1.0, 1.0, 1.0)
|
|
# the whole capacity up front: a game writes l.inst directly (Maroon Lake's prints, trees and
|
|
# rocks do), so a layer cannot start smaller than it may be written to
|
|
l.have = cap
|
|
l.inst = floats(l.have * INST_FLOATS)
|
|
l.scratch = floats(l.have * INST_FLOATS)
|
|
l.last_cam = v3_new(100000.0, 0.0, 0.0)
|
|
l.buf = gpu_buffer_new(render3d_st)
|
|
l.imp_buf = gpu_buffer_new(render3d_st)
|
|
l.sh_buf = gpu_buffer_new(render3d_st)
|
|
l.fog_buf = gpu_buffer_new(render3d_st)
|
|
l.rough = 1.0
|
|
for i in 0 .. len(model.prims) { scatter_attach(render3d_st, model.prims[i].mesh, l.buf) }
|
|
push(render3d_st.sc_layers, l)
|
|
return l
|
|
}
|
|
|
|
# room for n instances (at most the layer's cap), doubling what there is so a fill costs a few copies;
|
|
# a caller writing l.inst itself reserves first
|
|
function layer_reserve(l: Layer, n: int) -> void { layer_room(l, n) }
|
|
@alloc_ok("a layer outgrowing its capacity, grown once to what it now holds (rare, and bounded by the scene)")
|
|
function layer_room(l: Layer, n: int) -> void {
|
|
if n <= l.have { return }
|
|
var want = max(l.have * 2, n)
|
|
if want > l.cap { want = l.cap }
|
|
let inst = floats(want * INST_FLOATS)
|
|
if l.count > 0 { mem_copy(data_of(inst), data_of(l.inst), l.count * INST_FLOATS * 4) }
|
|
free(l.inst)
|
|
free(l.scratch)
|
|
l.inst = inst
|
|
l.scratch = floats(want * INST_FLOATS)
|
|
l.have = want
|
|
}
|
|
|
|
function layer_add(l: Layer, x: float, y: float, z: float, scale: float, yaw: float, seed: float, wind: float) -> void {
|
|
if l.count >= l.cap { return }
|
|
layer_room(l, l.count + 1)
|
|
let o = l.count * INST_FLOATS
|
|
l.inst[o] = x; l.inst[o + 1] = y; l.inst[o + 2] = z; l.inst[o + 3] = scale
|
|
l.inst[o + 4] = Math.sin(yaw); l.inst[o + 5] = Math.cos(yaw); l.inst[o + 6] = seed; l.inst[o + 7] = wind
|
|
l.count += 1
|
|
}
|
|
|
|
# ---- impostors ---------------------------------------------------------------------
|
|
# the GPU memory it makes is counted as VKM_IMPOSTOR (R3D_VKMEM)
|
|
function impostor_bake(render3d_st: mut Render3dState, model: Model, tiles: int, tw: int, th: int) -> Impostor {
|
|
let was = render3d_st.gvk_tag
|
|
render3d_st.gvk_tag = VKM_IMPOSTOR
|
|
let r = impostor_bake__t(render3d_st, model, tiles, tw, th)
|
|
render3d_st.gvk_tag = was
|
|
return r
|
|
}
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function impostor_bake__t(render3d_st: mut Render3dState, model: Model, tiles: int, tw: int, th: int) -> Impostor {
|
|
let im = new Impostor
|
|
im.tiles = tiles
|
|
im.radius = model.radius * 1.02
|
|
im.height = model.height
|
|
im.model = model; im.tw = tw; im.th = th; im.flower = render3d_st.sc_bake_flower
|
|
impostor_paint(render3d_st, im)
|
|
return im
|
|
}
|
|
|
|
# the atlases drawn from the model: at the bake, and again when a fog that let them go opens
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function impostor_paint(render3d_st: mut Render3dState, im: Impostor) -> void {
|
|
let model = im.model
|
|
let tiles = im.tiles
|
|
let tw = im.tw
|
|
let th = im.th
|
|
let aw = tiles * tw
|
|
im.albedo = tex_target(render3d_st, aw, th, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
im.normal = tex_target(render3d_st, aw, th, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
let fbo = gpu_fb_new(render3d_st)
|
|
gpu_fb_bind(render3d_st, fbo)
|
|
gpu_fb_color(render3d_st, 0, im.albedo)
|
|
gpu_fb_color(render3d_st, 1, im.normal)
|
|
let rb = gpu_rb_new(render3d_st)
|
|
gpu_rb_storage(render3d_st, rb, GL_DEPTH_COMPONENT24, aw, th, 0)
|
|
gpu_fb_depth_rb(render3d_st, rb)
|
|
gpu_fb_draw_buffers(render3d_st, 2)
|
|
gpu_viewport(render3d_st, 0, 0, aw, th)
|
|
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 0.0)
|
|
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
|
gpu_depth_test(render3d_st, true)
|
|
gpu_depth_func(render3d_st, GL_LESS)
|
|
gpu_cull(render3d_st, false)
|
|
gpu_blend(render3d_st, false)
|
|
# the model's prims temporarily take the identity instance
|
|
for i in 0 .. len(model.prims) { scatter_attach(render3d_st, model.prims[i].mesh, render3d_st.sc_ident_buf) }
|
|
let view = m4_new(); let proj = m4_new()
|
|
let eye = floats(3); let at = floats(3); let up = v3_new(0.0, 1.0, 0.0)
|
|
let cy = model.ymin + model.height * 0.5
|
|
let r = im.radius
|
|
let hh = model.height * 0.5
|
|
var bake = render3d_st.sc_bake_prog
|
|
if im.flower { bake = render3d_st.sc_bake_flower_prog }
|
|
gpu_use_program(render3d_st, bake)
|
|
for t in 0 .. tiles {
|
|
let a = 2.0 * PI * (float(t) / float(tiles))
|
|
v3_set(at, 0.0, cy, 0.0)
|
|
# a touch of elevation (the viewer usually looks slightly down at a tree)
|
|
v3_set(eye, Math.sin(a) * (r * 4.0), cy + r * 0.5, -(Math.cos(a) * (r * 4.0)))
|
|
m4_look_at(view, eye, at, up)
|
|
m4_ortho(proj, -r, r, -hh, hh, 0.1, r * 9.0)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, bake, "u_view"), view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, bake, "u_proj"), proj)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_wind"), 0.0)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_flutter"), 0.0)
|
|
gpu_viewport(render3d_st, t * tw, 0, tw, th)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
r3d_bind_2d(render3d_st, bake, "u_diff", 0, pr.diff)
|
|
r3d_bind_2d(render3d_st, bake, "u_arm", 2, pr.arm)
|
|
prim_factors(render3d_st, bake, pr)
|
|
mesh_draw_instanced(render3d_st, pr.mesh, 1)
|
|
}
|
|
}
|
|
free(view); free(proj); free(eye); free(at); free(up)
|
|
gpu_fb_bind(render3d_st, 0)
|
|
gpu_fb_free(render3d_st, fbo)
|
|
gpu_rb_free(render3d_st, rb)
|
|
gpu_tex_bind(render3d_st, GPU_TEX2D, im.albedo)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
|
gpu_tex_bind(render3d_st, GPU_TEX2D, im.normal)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
|
gpu_check(render3d_st, "impostor bake")
|
|
# numbered, so every bake of a run survives to be compared (a card layer per species bakes one)
|
|
if render3d_st.sc_debug_dump {
|
|
render3d_st.sc_dump_n += 1
|
|
render3d_st.tex_dump_alpha = true; tex_dump(render3d_st, im.albedo, aw, th, `build/atlas_{render3d_st.sc_dump_n}_alpha.ppm`); render3d_st.tex_dump_alpha = false
|
|
tex_dump(render3d_st, im.albedo, aw, th, `build/atlas_{render3d_st.sc_dump_n}_color.ppm`)
|
|
}
|
|
}
|
|
|
|
# Give a layer a LOD chain. `dists` (float bits) are the outer distances of each level;
|
|
# the last one becomes the layer's `near` so the impostor (if any) starts there.
|
|
@alloc_ok("start-up: the device, its tables, the programs, the passes and the world's first textures are made once, before play")
|
|
function layer_set_lods(render3d_st: mut Render3dState, l: Layer, models: []Model, dists: floats) -> void {
|
|
l.lods = models
|
|
l.n_lods = len(models)
|
|
l.lod_dist = floats(l.n_lods); l.lod_card = words(l.n_lods); l.lod_buf = words(l.n_lods); l.n_lod = words(l.n_lods)
|
|
for k in 0 .. l.n_lods {
|
|
l.lod_dist[k] = dists[k]; l.lod_card[k] = 0; l.n_lod[k] = 0
|
|
l.lod_buf[k] = gpu_buffer_new(render3d_st)
|
|
let m = models[k]
|
|
for i in 0 .. len(m.prims) { scatter_attach(render3d_st, m.prims[i].mesh, l.lod_buf[k]) }
|
|
}
|
|
l.model = models[0]
|
|
l.near = dists[l.n_lods - 1]
|
|
if l.lvl == null { l.lvl = words(l.cap) }
|
|
}
|
|
# mark level k as the layer's crossed card (drawn with the card program and its atlas)
|
|
function layer_lod_card(l: Layer, k: int) -> void { l.lod_card[k] = 1 }
|
|
|
|
function layer_set_impostor(render3d_st: mut Render3dState, l: Layer, im: Impostor) -> void {
|
|
l.imp = im
|
|
scatter_attach(render3d_st, render3d_st.sc_card, l.imp_buf)
|
|
fog_impostor(render3d_st, l)
|
|
}
|
|
|
|
# ---- per frame -----------------------------------------------------------------------
|
|
# Partitions and gathers are redone only when the view changed enough to matter: the
|
|
# camera moved 1.5 m or turned about 2.5 degrees. Everything culled by the frustum keys
|
|
# off this one counter, so a turn re-gathers the streams and the grids together.
|
|
function scatter_begin_frame(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.sc_view_pos == null { render3d_st.sc_view_pos = v3_new(100000.0, 0.0, 0.0); render3d_st.sc_view_fwd = v3_new(0.0, 0.0, -1.0) }
|
|
if v3_dist(render3d_st.sc_view_pos, render3d_st.cam_pos) > 1.5 or v3_dot(render3d_st.sc_view_fwd, render3d_st.cam_fwd) < 0.999 {
|
|
render3d_st.sc_view_gen += 1
|
|
v3_copy(render3d_st.sc_view_pos, render3d_st.cam_pos)
|
|
v3_copy(render3d_st.sc_view_fwd, render3d_st.cam_fwd)
|
|
}
|
|
# GPU-culled layers dispatch before the first pass of the frame, so no pass is split for it
|
|
if render3d_st.sc_layers != null {
|
|
for i in 0 .. len(render3d_st.sc_layers) { if render3d_st.sc_layers[i].n_lods > 1 and render3d_st.sc_layers[i].imp != null { layer_update(render3d_st, render3d_st.sc_layers[i]) } }
|
|
}
|
|
}
|
|
|
|
# ---- the GPU-culled path ------------------------------------------------------------------
|
|
# On Vulkan (R3D_GPU_CULL=0 turns it off): a tree layer (a LOD chain of up to four levels sharing up
|
|
# to four materials, with an impostor, not streamed) is culled and split into its buckets by
|
|
# scatter_cull.comp, and its lit, prepass, impostor and shadow-LOD draws read the records that pass
|
|
# wrote - one draw per material covering every level. Nothing is partitioned or uploaded on the CPU
|
|
# when the view moves. PC camp benchmark: 2791 -> 2657 draws, 4.3 -> 4.1 s for 400 frames.
|
|
const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand
|
|
const SC_LOD_ROOM: int = 64 # a layer's LODs plus near and far, most (sc_lod_* are made this size)
|
|
const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp)
|
|
|
|
function layer_gpu_eligible(render3d_st: Render3dState, l: Layer) -> bool {
|
|
if l.n_lods < 2 or l.n_lods > 4 or l.imp == null or l.streamed or l.flower or l.blade or l.count == 0 { return false }
|
|
return layer_arena_ok(render3d_st, l)
|
|
}
|
|
|
|
# The merged meshes. Every level of a kit tree or rock carries the same materials in the same order
|
|
# (bark then needles; the rock's one), so each material becomes ONE mesh holding all its levels, and
|
|
# one indirect draw of several records draws every level of it: record (material, level) names that
|
|
# level's index and vertex range and its bucket's instances. A conifer's lit pass goes from eight
|
|
# draws to two. Anything that does not fit - a card level, a level with other materials, other
|
|
# attributes or 32-bit indices - keeps the CPU path.
|
|
function layer_arena_ok(render3d_st: Render3dState, l: Layer) -> bool {
|
|
let n_mat = len(l.lods[0].prims)
|
|
if n_mat == 0 or n_mat > 4 { return false }
|
|
for k in 0 .. l.n_lods {
|
|
if l.lod_card[k] == 1 { return false }
|
|
let m = l.lods[k]
|
|
if len(m.prims) != n_mat { return false }
|
|
for j in 0 .. n_mat {
|
|
let pm = m.prims[j].mesh
|
|
let p0 = l.lods[0].prims[j]
|
|
if m.prims[j].diff != p0.diff or m.prims[j].verts == 0 or pm.ebo == 0 or pm.itype != GL_UNSIGNED_SHORT { return false }
|
|
if gpu_buffer_map(render3d_st, pm.ebo) == null { return false }
|
|
for a in 0 .. 3 {
|
|
let o = a * GPU_ATTR_W
|
|
if p0.mesh.attrs[o + 1] == 0 or pm.attrs[o + 1] != p0.mesh.attrs[o + 1] or pm.attrs[o + 3] != pm.attrs[o + 1] * 4 { return false }
|
|
if gpu_buffer_map(render3d_st, pm.attrs[o]) == null { return false }
|
|
}
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function layer_arena_build(render3d_st: mut Render3dState, l: Layer) -> void {
|
|
let n_mat = len(l.lods[0].prims)
|
|
l.g_arena = new []Prim
|
|
l.g_first = words(16); l.g_base = words(16)
|
|
for i in 0 .. 16 { l.g_first[i] = 0; l.g_base[i] = 0 }
|
|
for j in 0 .. n_mat {
|
|
var nv = 0
|
|
var ni = 0
|
|
for k in 0 .. l.n_lods {
|
|
let pr = l.lods[k].prims[j]
|
|
l.g_first[j * 4 + k] = ni; l.g_base[j * 4 + k] = nv
|
|
nv += pr.verts; ni += pr.mesh.count
|
|
}
|
|
let p0 = l.lods[0].prims[j]
|
|
let m = gpu_mesh_new(render3d_st)
|
|
for a in 0 .. 3 {
|
|
let comps = p0.mesh.attrs[a * GPU_ATTR_W + 1]
|
|
let vb = bytes(nv * comps * 4 + 8)
|
|
for k in 0 .. l.n_lods {
|
|
let pr = l.lods[k].prims[j]
|
|
mem_copy(mem_off(vb, l.g_base[j * 4 + k] * comps * 4), gpu_buffer_map(render3d_st, pr.mesh.attrs[a * GPU_ATTR_W]), pr.verts * comps * 4)
|
|
}
|
|
gpu_mesh_vertices(render3d_st, m, vb, nv * comps * 4, GPU_STATIC)
|
|
gpu_mesh_attr(render3d_st, m, a, comps, GPU_F32, 0, 0, false)
|
|
free(vb)
|
|
}
|
|
let ib = bytes(ni * 2 + 8)
|
|
for k in 0 .. l.n_lods {
|
|
let pm = l.lods[k].prims[j].mesh
|
|
mem_copy(mem_off(ib, l.g_first[j * 4 + k] * 2), gpu_buffer_map(render3d_st, pm.ebo), pm.count * 2)
|
|
}
|
|
gpu_mesh_indices(render3d_st, m, ib, ni * 2, 2)
|
|
free(ib)
|
|
m.count = ni
|
|
gpu_mesh_done(render3d_st, m)
|
|
scatter_attach(render3d_st, m, l.g_dst)
|
|
let ap = new Prim
|
|
ap.mesh = m; ap.diff = p0.diff; ap.nrm = p0.nrm; ap.arm = p0.arm; ap.verts = nv; ap.name = p0.name
|
|
push(l.g_arena, ap)
|
|
}
|
|
l.g_model = new Model
|
|
l.g_model.prims = l.g_arena; l.g_model.height = l.lods[0].height; l.g_model.radius = l.lods[0].radius; l.g_model.ymin = l.lods[0].ymin
|
|
l.g_model_sh = new Model
|
|
var sh = l.n_lods - 1
|
|
if sh > 2 { sh = 2 }
|
|
l.g_model_sh.prims = l.g_arena; l.g_model_sh.height = l.lods[sh].height; l.g_model_sh.radius = l.lods[sh].radius; l.g_model_sh.ymin = l.lods[sh].ymin
|
|
}
|
|
|
|
function layer_gpu_prepare(render3d_st: mut Render3dState, l: Layer) -> bool {
|
|
if not render3d_st.sc_cull_tried {
|
|
render3d_st.sc_cull_tried = true
|
|
# Os.env is null when the variable is unset, and a compare reads through it: ask first
|
|
# on by default wherever there is compute; R3D_GPU_CULL=0 keeps the CPU partition (for comparing)
|
|
var off = false
|
|
if r3d_env_has(render3d_st, "R3D_GPU_CULL") { off = r3d_env(render3d_st, "R3D_GPU_CULL") == "0" }
|
|
if gpu_has_compute(render3d_st) and gpu_has_mdi(render3d_st) and not off {
|
|
render3d_st.sc_cull_prog = gpu_compute(render3d_st, "scatter_cull", 4)
|
|
if render3d_st.sc_cull_prog > 0 { print("r3d: scatter: tree layers are culled on the GPU") }
|
|
}
|
|
}
|
|
if render3d_st.sc_cull_prog == 0 or not layer_gpu_eligible(render3d_st, l) { return false }
|
|
if l.g_on and l.g_n == l.count { return true }
|
|
if l.g_src == 0 {
|
|
l.g_src = gpu_buffer_new(render3d_st); l.g_dst = gpu_buffer_new(render3d_st); l.g_cmds = gpu_buffer_new(render3d_st); l.g_counts = gpu_buffer_new(render3d_st)
|
|
gpu_buffer_gpu_owned(render3d_st, l.g_dst); gpu_buffer_gpu_owned(render3d_st, l.g_cmds); gpu_buffer_gpu_owned(render3d_st, l.g_counts)
|
|
}
|
|
let n = l.n_lods
|
|
let cap = l.count
|
|
gpu_buffer_upload(render3d_st, l.g_src, cap * INST_FLOATS * 4, data_of(l.inst), GPU_STATIC)
|
|
gpu_buffer_upload(render3d_st, l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC)
|
|
if l.g_arena == null { layer_arena_build(render3d_st, l) }
|
|
let n_mat = len(l.g_arena)
|
|
let rec = render3d_st.sc_gpu_rec
|
|
for i in 0 .. SC_RECS * 5 { rec[i] = 0 }
|
|
# material j, level k: that level's range of the merged mesh, its instances from bucket k
|
|
for j in 0 .. n_mat {
|
|
for k in 0 .. n {
|
|
let r = (j * 4 + k) * 5
|
|
rec[r] = l.lods[k].prims[j].mesh.count; rec[r + 2] = l.g_first[j * 4 + k]; rec[r + 3] = l.g_base[j * 4 + k]; rec[r + 4] = k * cap
|
|
}
|
|
}
|
|
rec[16 * 5] = render3d_st.sc_card.count; rec[16 * 5 + 4] = n * cap
|
|
# the shadow LOD: level 2's range, from buckets 0 .. 2
|
|
if n > 2 {
|
|
for j in 0 .. n_mat {
|
|
for b in 0 .. 3 {
|
|
let r = (17 + j * 3 + b) * 5
|
|
rec[r] = l.lods[2].prims[j].mesh.count; rec[r + 2] = l.g_first[j * 4 + 2]; rec[r + 3] = l.g_base[j * 4 + 2]; rec[r + 4] = b * cap
|
|
}
|
|
}
|
|
}
|
|
gpu_buffer_upload(render3d_st, l.g_cmds, SC_RECS * SC_REC_W, data_of(rec), GPU_DYNAMIC)
|
|
let zeros = render3d_st.sc_gpu_zeros
|
|
for i in 0 .. 5 { zeros[i] = 0 }
|
|
gpu_buffer_upload(render3d_st, l.g_counts, 20, data_of(zeros), GPU_DYNAMIC)
|
|
# the card casts every instance, as on the CPU path (layer_grid_build uploads this there)
|
|
l.n_sh = l.count
|
|
gpu_buffer_upload(render3d_st, l.sh_buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_STATIC)
|
|
l.g_n = l.count
|
|
l.g_on = true
|
|
return true
|
|
}
|
|
|
|
# the dispatch for the view as it stands: frustum, camera, distances, the layer's shape
|
|
function layer_gpu_cull(render3d_st: mut Render3dState, l: Layer) -> void {
|
|
let pr = render3d_st.sc_gpu_pr
|
|
for i in 0 .. 36 { pr[i] = 0 }
|
|
if render3d_st.cam_planes != null { for i in 0 .. 16 { pr[i] = float_bits(render3d_st.cam_planes[i]) } }
|
|
pr[16] = float_bits(render3d_st.cam_pos[0]); pr[17] = float_bits(render3d_st.cam_pos[1]); pr[18] = float_bits(render3d_st.cam_pos[2]); pr[19] = float_bits(r3d_reach(render3d_st, l.cull))
|
|
for k in 0 .. l.n_lods { pr[20 + k] = float_bits(l.lod_dist[k]); pr[24 + k] = len(l.lods[k].prims) }
|
|
pr[28] = l.count; pr[29] = l.count; pr[30] = l.n_lods; pr[31] = 1
|
|
# as layer_grid_gather pads a cell: the tallest instance, plus a margin
|
|
pr[32] = float_bits(l.lods[0].height * 2.0); pr[33] = float_bits(4.0)
|
|
let bufs = render3d_st.sc_gpu_bufs
|
|
bufs[0] = l.g_src; bufs[1] = l.g_dst; bufs[2] = l.g_cmds; bufs[3] = l.g_counts
|
|
gpu_dispatch(render3d_st, render3d_st.sc_cull_prog, data_of(pr), 144, bufs, 1)
|
|
}
|
|
|
|
# Sort a static layer's instances into square cells (call once, after placement; a
|
|
# large layer that was never gridded gets a 96 m grid on its first update). The
|
|
# shadow buffer is uploaded here once — casters are never culled by the view.
|
|
@alloc_ok("a world being set up (a stream, a layer, water, post, a bake): the map loading or swapping, not a frame")
|
|
function layer_grid_build(render3d_st: mut Render3dState, l: Layer, cell: float) -> void {
|
|
if l.count == 0 { return }
|
|
var minx = l.inst[0]; var maxx = minx; var minz = l.inst[2]; var maxz = minz
|
|
for i in 0 .. l.count {
|
|
let o = i * INST_FLOATS
|
|
minx = Math.min(minx, l.inst[o]); maxx = Math.max(maxx, l.inst[o])
|
|
minz = Math.min(minz, l.inst[o + 2]); maxz = Math.max(maxz, l.inst[o + 2])
|
|
}
|
|
l.gcell = cell; l.gx0 = minx; l.gz0 = minz
|
|
l.gnx = int((maxx - minx) / cell) + 1
|
|
l.gnz = int((maxz - minz) / cell) + 1
|
|
let ncell = l.gnx * l.gnz
|
|
l.gstart = words(ncell + 1)
|
|
l.gymin = floats(ncell); l.gymax = floats(ncell)
|
|
let cellof = words(l.count)
|
|
for c in 0 .. ncell + 1 { l.gstart[c] = 0 }
|
|
for i in 0 .. l.count {
|
|
let o = i * INST_FLOATS
|
|
let ix = int((l.inst[o] - minx) / cell)
|
|
let iz = int((l.inst[o + 2] - minz) / cell)
|
|
let c = iz * l.gnx + ix
|
|
cellof[i] = c
|
|
if l.gstart[c + 1] == 0 { l.gymin[c] = l.inst[o + 1]; l.gymax[c] = l.inst[o + 1] }
|
|
else { l.gymin[c] = Math.min(l.gymin[c], l.inst[o + 1]); l.gymax[c] = Math.max(l.gymax[c], l.inst[o + 1]) }
|
|
l.gstart[c + 1] += 1
|
|
}
|
|
for c in 0 .. ncell { l.gstart[c + 1] += l.gstart[c] }
|
|
let fill = words(ncell)
|
|
for c in 0 .. ncell { fill[c] = l.gstart[c] }
|
|
l.gsorted = floats(l.count * INST_FLOATS)
|
|
for i in 0 .. l.count {
|
|
let c = cellof[i]
|
|
let q = fill[c] * INST_FLOATS
|
|
fill[c] += 1
|
|
let o = i * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { l.gsorted[q + k] = l.inst[o + k] }
|
|
}
|
|
free(cellof); free(fill)
|
|
if l.vis == null { l.vis = floats(l.cap * INST_FLOATS) }
|
|
l.n_sh = l.count
|
|
gpu_buffer_upload(render3d_st, l.sh_buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_STATIC)
|
|
}
|
|
|
|
# gather the instances of the cells the camera can see (and that are within cull)
|
|
function layer_grid_gather(render3d_st: Render3dState, l: Layer) -> void {
|
|
let cell = l.gcell
|
|
let half = cell * 0.5
|
|
let reach = r3d_reach(render3d_st, l.cull) + cell * 0.71
|
|
var n = 0
|
|
for iz in 0 .. l.gnz {
|
|
let wz = l.gz0 + float(iz) * cell + half
|
|
for ix in 0 .. l.gnx {
|
|
let c = iz * l.gnx + ix
|
|
let cnt = l.gstart[c + 1] - l.gstart[c]
|
|
if cnt == 0 { continue }
|
|
let wx = l.gx0 + float(ix) * cell + half
|
|
if r3d_reach(render3d_st, l.cull) != 0.0 {
|
|
let dx = wx - render3d_st.cam_pos[0]; let dz = wz - render3d_st.cam_pos[2]
|
|
if Math.sqrt(dx * dx + dz * dz) > reach { continue }
|
|
}
|
|
let hy = (l.gymax[c] - l.gymin[c]) * 0.5
|
|
let cy = l.gymin[c] + hy
|
|
# pad by the tallest instance (scale 2 of the model's height) so crowns at the frame's edge stay
|
|
let r = Math.sqrt(half * half * 2.0 + hy * hy) + (l.model.height * 2.0 + 4.0)
|
|
if not cam_sphere_visible(render3d_st, wx, cy, wz, r) { continue }
|
|
mem_copy(mem_off(l.vis, n * INST_FLOATS * 4), mem_off(l.gsorted, l.gstart[c] * INST_FLOATS * 4), cnt * INST_FLOATS * 4)
|
|
n += cnt
|
|
}
|
|
}
|
|
l.n_vis = n
|
|
}
|
|
|
|
# Sort the gathered instances into their LOD levels (counting sort into the scratch),
|
|
# the impostor bucket last, and upload one buffer per level.
|
|
function layer_partition_lods(render3d_st: mut Render3dState, l: Layer, src: floats, total: int) -> void {
|
|
let n = l.n_lods
|
|
if n + 2 > SC_LOD_ROOM { return } # past the room made with the state
|
|
let counts = render3d_st.sc_lod_counts
|
|
for k in 0 .. n + 2 { counts[k] = 0 }
|
|
let lc = r3d_reach(render3d_st, l.cull)
|
|
let cull2 = lc * lc
|
|
let open = l.lod_dist[n - 1] == 0.0 # the last level runs out to the cull distance
|
|
for i in 0 .. total {
|
|
let o = i * INST_FLOATS
|
|
let dx = src[o] - render3d_st.cam_pos[0]; let dz = src[o + 2] - render3d_st.cam_pos[2]
|
|
let d2 = dx * dx + dz * dz
|
|
var lv = n + 1 # n + 1 = dropped
|
|
if lc == 0.0 or not (d2 > cull2) {
|
|
let d = Math.sqrt(d2)
|
|
lv = n # n = the impostor bucket
|
|
var k = 0
|
|
while k < n { if l.lod_dist[k] != 0.0 and d < l.lod_dist[k] { lv = k; k = n } else { k += 1 } }
|
|
if lv == n and open { lv = n - 1 }
|
|
if lv == n and l.imp == null { lv = n + 1 }
|
|
}
|
|
l.lvl[i] = lv
|
|
counts[lv] += 1
|
|
}
|
|
# prefix offsets (in instances) per bucket
|
|
let start = render3d_st.sc_lod_start
|
|
var acc = 0
|
|
for k in 0 .. n + 2 { start[k] = acc; acc += counts[k] }
|
|
let fill = render3d_st.sc_lod_fill
|
|
for k in 0 .. n + 2 { fill[k] = start[k] }
|
|
let tmp = l.scratch
|
|
for i in 0 .. total {
|
|
let lv = l.lvl[i]
|
|
if lv > n { continue }
|
|
let q = fill[lv] * INST_FLOATS
|
|
fill[lv] += 1
|
|
let o = i * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
}
|
|
for k in 0 .. n {
|
|
l.n_lod[k] = counts[k]
|
|
if counts[k] > 0 {
|
|
gpu_buffer_upload(render3d_st, l.lod_buf[k], counts[k] * INST_FLOATS * 4, mem_off(tmp, start[k] * INST_FLOATS * 4), GPU_DYNAMIC)
|
|
}
|
|
}
|
|
l.n_near = counts[0]
|
|
l.n_far = counts[n]
|
|
if render3d_st.sc_dbg_lod and total > 1000 { sc_debug_lods(render3d_st, l, counts, n, total, src) }
|
|
if l.n_far > 0 {
|
|
gpu_buffer_upload(render3d_st, l.imp_buf, l.n_far * INST_FLOATS * 4, mem_off(tmp, start[n] * INST_FLOATS * 4), GPU_DYNAMIC)
|
|
}
|
|
# casters: the whole (gathered) set from the shadow buffer, unless the impostor casts
|
|
if l.gcell == 0.0 {
|
|
l.n_sh = total
|
|
if total > 0 { gpu_buffer_upload(render3d_st, l.sh_buf, total * INST_FLOATS * 4, data_of(src), GPU_DYNAMIC) }
|
|
}
|
|
}
|
|
|
|
# split the instances by distance to the camera (only when the view changed)
|
|
function layer_update(render3d_st: mut Render3dState, l: Layer) -> void {
|
|
if render3d_st.sc_freeze { return }
|
|
if l.view_gen == render3d_st.sc_view_gen { return }
|
|
l.view_gen = render3d_st.sc_view_gen
|
|
if layer_gpu_prepare(render3d_st, l) { layer_gpu_cull(render3d_st, l); return }
|
|
let t_lu = gl_now_us()
|
|
let n_lu = l.count
|
|
# A streamed layer's instances were already gathered per visible chunk: no split, no
|
|
# per-instance loop — one upload, and the same buffer casts its shadows.
|
|
if l.streamed and l.imp == null and l.near == 0.0 and l.n_lods <= 1 {
|
|
l.n_near = l.count; l.n_far = 0; l.n_sh = l.count
|
|
gpu_buffer_upload(render3d_st, l.buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_DYNAMIC)
|
|
prof_layer_add(render3d_st, gl_now_us() - t_lu, l.count * INST_FLOATS * 4)
|
|
return
|
|
}
|
|
if l.gcell == 0.0 and not l.streamed and l.count > 2000 { layer_grid_build(render3d_st, l, 96.0) }
|
|
var src = l.inst
|
|
var total = l.count
|
|
if l.gcell != 0.0 { layer_grid_gather(render3d_st, l); src = l.vis; total = l.n_vis }
|
|
let near2 = l.near * l.near
|
|
let lc = r3d_reach(render3d_st, l.cull)
|
|
let cull2 = lc * lc
|
|
var nn = 0
|
|
var nf = 0
|
|
let far_off = l.cap * INST_FLOATS # far instances fill the scratch from its end backwards
|
|
let tmp = l.scratch
|
|
if l.n_lods > 1 {
|
|
layer_partition_lods(render3d_st, l, src, total)
|
|
prof_layer_add(render3d_st, gl_now_us() - t_lu, total * INST_FLOATS * 4)
|
|
return
|
|
}
|
|
var i = 0
|
|
while i < total {
|
|
let o = i * INST_FLOATS
|
|
let dx = src[o] - render3d_st.cam_pos[0]
|
|
let dz = src[o + 2] - render3d_st.cam_pos[2]
|
|
let d2 = dx * dx + dz * dz
|
|
if lc != 0.0 and d2 > cull2 { i += 1; continue }
|
|
if l.near == 0.0 or d2 < near2 {
|
|
let q = nn * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
nn += 1
|
|
} else if l.imp != null {
|
|
nf += 1
|
|
let q = far_off - nf * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
}
|
|
i += 1
|
|
}
|
|
l.n_near = nn
|
|
l.n_far = nf
|
|
if render3d_st.sc_debug_dump { sc_debug_layer(l, tmp, nn, nf) }
|
|
# Every instance, unculled and unsplit, for the shadow pass. What the camera draws is
|
|
# allowed to change with distance; what casts must not, or shadows blink in and out as
|
|
# you walk. This is the whole set, drawn one way, into every cascade.
|
|
if l.gcell == 0.0 {
|
|
l.n_sh = l.count
|
|
if l.count > 0 {
|
|
gpu_buffer_upload(render3d_st, l.sh_buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_DYNAMIC)
|
|
}
|
|
}
|
|
gpu_buffer_upload(render3d_st, l.buf, nn * INST_FLOATS * 4, data_of(tmp), GPU_DYNAMIC)
|
|
if nf > 0 {
|
|
gpu_buffer_upload(render3d_st, l.imp_buf, nf * INST_FLOATS * 4, mem_off(tmp, (far_off - nf * INST_FLOATS) * 4), GPU_DYNAMIC)
|
|
}
|
|
prof_layer_add(render3d_st, gl_now_us() - t_lu, (nn + nf + l.n_sh) * INST_FLOATS * 4)
|
|
}
|
|
|
|
function layer_program(render3d_st: Render3dState, l: Layer, shadow: bool, card: bool) -> int {
|
|
if card {
|
|
if shadow { return render3d_st.sc_prog_card_shadow }
|
|
if l.cheap { return render3d_st.sc_prog_card_cheap }
|
|
return render3d_st.sc_prog_card
|
|
}
|
|
if shadow {
|
|
if l.foliage and not l.blade and not l.flower { return render3d_st.sc_prog_shadow_fol }
|
|
if l.wind != 0.0 { return render3d_st.sc_prog_shadow_wind }
|
|
return render3d_st.sc_prog_shadow
|
|
}
|
|
if l.blade { return render3d_st.sc_prog_blade }
|
|
if l.flower { return render3d_st.sc_prog_flower }
|
|
if l.foliage {
|
|
if render3d_st.sc_prepass and render3d_st.sc_prog_fol_eq != 0 { return render3d_st.sc_prog_fol_eq }
|
|
return render3d_st.sc_prog_fol
|
|
}
|
|
if l.wind != 0.0 { return render3d_st.sc_prog_wind }
|
|
return render3d_st.sc_prog
|
|
}
|
|
|
|
# R3D_LODDBG=1 tints each LOD level (red, green, blue, yellow) and impostors magenta
|
|
# Can level k's casters (its instances lie between the previous level's distance and its own) put a
|
|
# shadow on anything the cascade being rendered covers? A receiver in that slice of view depth
|
|
# [near, far] stands between near - dy and far * K metres away on the ground: dy is the camera's height
|
|
# over the ground, K how far the frustum's corners reach past its depth. A prop's shadow falls at most
|
|
# about six times its height past it (the sun near ten degrees). A level outside that range cannot
|
|
# touch a pixel of the cascade, so leaving it out changes no shadow - unlike the old per-class skips
|
|
# at fixed distances, which dropped casters that did cast (see scatter_draw_casters). The flowers'
|
|
# mesh levels (6 - 30 m) stop being drawn into the three outer cascades. R3D_CAST_ALL=1 draws every
|
|
# level into every cascade, for comparing.
|
|
function layer_level_casts_here(render3d_st: mut Render3dState, l: Layer, k: int) -> bool {
|
|
if l.lod_dist == null { return true }
|
|
var dmin = 0.0
|
|
if k > 0 { dmin = l.lod_dist[k - 1] }
|
|
return cast_band_reaches(render3d_st, dmin, l.lod_dist[k], l.lods[k].height)
|
|
}
|
|
# Can something standing between dmin and dmax metres from the camera (dmax 0: no outer limit), this
|
|
# tall, put a shadow on anything the cascade being rendered covers? The flowers' levels ask it
|
|
# (layer_level_casts_here), and so does every actor (actor_draw_casters).
|
|
function cast_band_reaches(render3d_st: mut Render3dState, dmin: float, dmax: float, height: float) -> bool {
|
|
if render3d_st.sc_cast_all < 0 { render3d_st.sc_cast_all = 0; if r3d_env_has(render3d_st, "R3D_CAST_ALL") { render3d_st.sc_cast_all = 1 } }
|
|
if render3d_st.sc_cast_all == 1 or render3d_st.sh_split == null { return true }
|
|
if render3d_st.sc_cast_gen != render3d_st.sc_view_gen {
|
|
render3d_st.sc_cast_gen = render3d_st.sc_view_gen
|
|
render3d_st.sc_cast_dy = Math.abs(render3d_st.cam_pos[1] - terrain_height(render3d_st, render3d_st.cam_pos[0], render3d_st.cam_pos[2])) + 5.0
|
|
let half = render3d_st.cam_fov * 0.5
|
|
let t = Math.sin(half) / Math.cos(half)
|
|
let ta = t * render3d_st.cam_aspect
|
|
render3d_st.sc_cast_k = Math.sqrt(1.0 + (t * t + ta * ta)) * 1.1
|
|
}
|
|
let c = render3d_st.sh_cascade
|
|
var near = render3d_st.cam_near
|
|
# sunShadow cross-fades into this cascade from 0.85 of the previous one's split (lighting.glsl), so
|
|
# its receivers start there, not at the split: starting at the split changed 19 pixels in town
|
|
if c > 0 { near = render3d_st.sh_split[c - 1] * 0.85 }
|
|
let far = render3d_st.sh_split[c]
|
|
let reach = Math.max(Math.min(height * 6.0, 40.0), 8.0)
|
|
if dmax != 0.0 and dmax + reach + render3d_st.sc_cast_dy < near { return false }
|
|
if dmin - reach > far * render3d_st.sc_cast_k { return false }
|
|
return true
|
|
}
|
|
|
|
# `full` casts the layer's entire instance list out of sh_buf instead of the near
|
|
# partition out of l.buf. A layer with no impostor (the tree crowns' branch cards) has
|
|
# no cheap stand-in to cast from, so without this its shadow simply began at the near
|
|
# distance — which is the crown shadow that appeared as you walked up to a tree.
|
|
function layer_draw_near(render3d_st: mut Render3dState, l: Layer, shadow: bool, light_vp: floats, full: bool) -> void {
|
|
if l.n_lods > 1 and l.g_on and not shadow {
|
|
# one draw per material covering all of its levels (the merged meshes)
|
|
render3d_st.sc_ind_base = 0; render3d_st.sc_ind_n = l.n_lods
|
|
layer_draw_model(render3d_st, l, l.g_model, l.g_dst, 1, false, shadow, light_vp)
|
|
render3d_st.sc_ind_base = -1; render3d_st.sc_ind_n = 1
|
|
return
|
|
}
|
|
if l.n_lods > 1 {
|
|
# a LOD chain: every level from its own bucket (casters are what is drawn), and a caster level
|
|
# only into the cascades it can put a shadow in
|
|
for k in 0 .. l.n_lods {
|
|
if shadow and not layer_level_casts_here(render3d_st, l, k) { continue }
|
|
render3d_st.sc_dbg_level = k; layer_draw_model(render3d_st, l, l.lods[k], l.lod_buf[k], l.n_lod[k], l.lod_card[k] == 1, shadow, light_vp)
|
|
}
|
|
render3d_st.sc_dbg_level = -1
|
|
return
|
|
}
|
|
var vb = l.buf
|
|
var cnt = l.n_near
|
|
if full and not l.streamed { vb = l.sh_buf; cnt = l.n_sh }
|
|
layer_draw_model(render3d_st, l, l.model, vb, cnt, l.card, shadow, light_vp)
|
|
}
|
|
|
|
# draw `cnt` instances of `model` out of instance buffer `vb`, as a mesh or as the layer's card
|
|
function layer_draw_model(render3d_st: mut Render3dState, l: Layer, model: Model, vb: int, cnt: int, card: bool, shadow: bool, light_vp: floats) -> void {
|
|
if cnt == 0 { return }
|
|
let p = layer_program(render3d_st, l, shadow, card)
|
|
gpu_use_program(render3d_st, p)
|
|
# over the prepass: only the fragment the prepass kept, at exactly its depth (a texel
|
|
# it cut would otherwise pass LEQUAL over the terrain behind and draw the quad solid)
|
|
if p == render3d_st.sc_prog_fol_eq { gpu_depth_func(render3d_st, GL_EQUAL) }
|
|
var ground = 0.0
|
|
if l.grounded { ground = 1.0 }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_ground"), ground)
|
|
if l.grounded { terrain_bind_height(render3d_st, p) }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_time"), render3d_st.r3d_time)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_wind"), l.wind)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_flutter"), l.flutter)
|
|
var mh = 0.0
|
|
if not card and l.foliage { mh = model.height }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_model_h"), mh)
|
|
if not card and l.foliage and not shadow and render3d_st.sc_a2c { gpu_alpha_to_coverage(render3d_st, true) }
|
|
if card {
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_card_w"), l.atlas.radius * 2.0); u_f(render3d_st, gpu_uniform(render3d_st, p, "u_card_h"), l.atlas.height)
|
|
r3d_bind_2d(render3d_st, p, "u_diff", 0, l.atlas.albedo)
|
|
r3d_bind_2d(render3d_st, p, "u_nrm", 1, l.atlas.normal)
|
|
gpu_alpha_to_coverage(render3d_st, true)
|
|
}
|
|
if shadow { u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_light_vp"), light_vp) }
|
|
else {
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_view"), render3d_st.cam_view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_proj"), render3d_st.cam_proj)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), l.tint)
|
|
if render3d_st.sc_dbg_lod and render3d_st.sc_dbg_level >= 0 {
|
|
if render3d_st.sc_dbg_tint == null { render3d_st.sc_dbg_tint = v3_new(1.0, 1.0, 1.0) }
|
|
let k = render3d_st.sc_dbg_level
|
|
var r = 0.0; var g = 0.0; var b = 0.0
|
|
if k == 0 { r = 3.0 } else if k == 1 { g = 3.0 } else if k == 2 { b = 3.0 } else { r = 3.0; g = 3.0 }
|
|
v3_set(render3d_st.sc_dbg_tint, r, g, b)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), render3d_st.sc_dbg_tint)
|
|
}
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_rough_scale"), l.rough)
|
|
if l.blade { u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_blade_base"), render3d_st.sc_blade_base); u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_blade_tip"), render3d_st.sc_blade_tip) }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_cull"), r3d_reach(render3d_st, l.cull))
|
|
sky_bind_lighting(render3d_st, p)
|
|
shadow_bind(render3d_st, p)
|
|
fog_bind(render3d_st, p)
|
|
if l.foliage { u_f(render3d_st, gpu_uniform(render3d_st, p, "u_spec_scale"), 0.05) }
|
|
}
|
|
gpu_cull(render3d_st, false)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
scatter_attach(render3d_st, pr.mesh, vb)
|
|
if not card {
|
|
r3d_bind_2d(render3d_st, p, "u_diff", 0, pr.diff)
|
|
prim_factors(render3d_st, p, pr)
|
|
if not shadow { r3d_bind_2d(render3d_st, p, "u_nrm", 1, pr.nrm); r3d_bind_2d(render3d_st, p, "u_arm", 2, pr.arm) }
|
|
}
|
|
if render3d_st.sc_ind_base >= 0 { gpu_draw_mesh_indirect(render3d_st, pr.mesh, l.g_cmds, (render3d_st.sc_ind_base + i * render3d_st.sc_ind_stride) * SC_REC_W, render3d_st.sc_ind_n, 0, 0) }
|
|
else { mesh_draw_instanced(render3d_st, pr.mesh, cnt) }
|
|
}
|
|
gpu_alpha_to_coverage(render3d_st, false)
|
|
if p == render3d_st.sc_prog_fol_eq { gpu_depth_func(render3d_st, GL_LESS) }
|
|
}
|
|
|
|
function layer_draw_far(render3d_st: mut Render3dState, l: Layer, shadow: bool, light_vp: floats) -> void {
|
|
if l.imp == null or l.imp.albedo == 0 or (l.n_far == 0 and not l.g_on) { return }
|
|
var p = render3d_st.sc_imp_prog
|
|
if shadow { p = render3d_st.sc_imp_prog_shadow }
|
|
gpu_use_program(render3d_st, p)
|
|
let im = l.imp
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_radius"), im.radius)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_height"), im.height)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_tiles"), float(im.tiles))
|
|
r3d_bind_2d(render3d_st, p, "u_atlas_albedo", 0, im.albedo)
|
|
if shadow {
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_light_vp"), light_vp)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_face_dir"), render3d_st.sun_dir)
|
|
if render3d_st.r3d_debug_shadow and not render3d_st.sc_printed { render3d_st.sc_printed = true; print(`imp shadow prog {p} face_dir loc {gpu_uniform(render3d_st, p, "u_face_dir")} sun {fixed(render3d_st.sun_dir[0])} {fixed(render3d_st.sun_dir[1])} {fixed(render3d_st.sun_dir[2])} cam {fixed(render3d_st.cam_pos[0])} {fixed(render3d_st.cam_pos[1])} {fixed(render3d_st.cam_pos[2])} n_far {l.n_far}`) }
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_cam_pos"), render3d_st.cam_pos)
|
|
} else {
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_view"), render3d_st.cam_view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_proj"), render3d_st.cam_proj)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), l.tint)
|
|
if render3d_st.sc_dbg_lod { if render3d_st.sc_dbg_tint == null { render3d_st.sc_dbg_tint = v3_new(1.0, 1.0, 1.0) }; v3_set(render3d_st.sc_dbg_tint, 3.0, 0.0, 3.0); u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), render3d_st.sc_dbg_tint) }
|
|
r3d_bind_2d(render3d_st, p, "u_atlas_normal", 1, im.normal)
|
|
sky_bind_lighting(render3d_st, p)
|
|
shadow_bind(render3d_st, p)
|
|
fog_bind(render3d_st, p)
|
|
if l.foliage { u_f(render3d_st, gpu_uniform(render3d_st, p, "u_spec_scale"), 0.05) }
|
|
}
|
|
gpu_cull(render3d_st, false)
|
|
if not shadow and render3d_st.sc_a2c { gpu_alpha_to_coverage(render3d_st, true) }
|
|
if l.g_on {
|
|
scatter_attach(render3d_st, render3d_st.sc_card, l.g_dst)
|
|
gpu_draw_mesh_indirect(render3d_st, render3d_st.sc_card, l.g_cmds, 16 * SC_REC_W, 1, 0, 0)
|
|
} else {
|
|
scatter_attach(render3d_st, render3d_st.sc_card, l.imp_buf)
|
|
mesh_draw_instanced(render3d_st, render3d_st.sc_card, l.n_far)
|
|
}
|
|
gpu_alpha_to_coverage(render3d_st, false)
|
|
}
|
|
|
|
# Cast from the impostor card, always, for every instance in the layer. The lit pass may
|
|
# swap a scanned mesh in up close; the shadow must not, or a tree's shadow changes shape
|
|
# as you approach it. The card is also far the cheaper of the two, which is what pays for
|
|
# casting the whole set into all five cascades.
|
|
function layer_draw_shadow(render3d_st: mut Render3dState, l: Layer, light_vp: floats) -> void {
|
|
if l.n_sh == 0 or l.imp == null or l.imp.albedo == 0 { return }
|
|
let p = render3d_st.sc_imp_prog_shadow
|
|
gpu_use_program(render3d_st, p)
|
|
let im = l.imp
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_radius"), im.radius)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_height"), im.height)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_tiles"), float(im.tiles))
|
|
r3d_bind_2d(render3d_st, p, "u_atlas_albedo", 0, im.albedo)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_light_vp"), light_vp)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_face_dir"), render3d_st.sun_dir)
|
|
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_cam_pos"), render3d_st.cam_pos)
|
|
gpu_cull(render3d_st, false)
|
|
let buf = fog_casters(render3d_st, l)
|
|
let n = fog_casters_n(render3d_st, l)
|
|
if n == 0 { return }
|
|
scatter_attach(render3d_st, render3d_st.sc_card, buf)
|
|
mesh_draw_instanced(render3d_st, render3d_st.sc_card, n)
|
|
}
|
|
|
|
|
|
# ---- the foliage depth prepass -----------------------------------------------------------
|
|
# Exactly the instances and positions layer_draw_model will light (the same LOD buckets,
|
|
# the same vertex shader, the same ground and wind), into depth only.
|
|
function layer_draw_depth(render3d_st: mut Render3dState, l: Layer, model: Model, vb: int, cnt: int) -> void {
|
|
if cnt == 0 or model == null or vb == 0 { return }
|
|
let p = render3d_st.sc_prog_fol_depth
|
|
gpu_use_program(render3d_st, p)
|
|
var ground = 0.0
|
|
if l.grounded { ground = 1.0 }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_ground"), ground)
|
|
if l.grounded { terrain_bind_height(render3d_st, p) }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_time"), render3d_st.r3d_time)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_wind"), l.wind)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_flutter"), l.flutter)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_model_h"), model.height)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_view"), render3d_st.cam_view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_proj"), render3d_st.cam_proj)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_clip_y"), render3d_st.r3d_clip_y)
|
|
gpu_cull(render3d_st, false)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
scatter_attach(render3d_st, pr.mesh, vb)
|
|
r3d_bind_2d(render3d_st, p, "u_diff", 0, pr.diff)
|
|
prim_factors(render3d_st, p, pr)
|
|
if render3d_st.sc_ind_base >= 0 { gpu_draw_mesh_indirect(render3d_st, pr.mesh, l.g_cmds, (render3d_st.sc_ind_base + i * render3d_st.sc_ind_stride) * SC_REC_W, render3d_st.sc_ind_n, 0, 0) }
|
|
else { mesh_draw_instanced(render3d_st, pr.mesh, cnt) }
|
|
}
|
|
}
|
|
|
|
# every foliage mesh draw the lit pass will make with sc_prog_fol_eq: not blades, not
|
|
# flowers, not card levels (those keep their own alpha and draw as before)
|
|
function scatter_draw_depth(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.sc_prog_fol_depth == 0 { return }
|
|
for i in 0 .. len(render3d_st.sc_layers) {
|
|
let l = render3d_st.sc_layers[i]
|
|
if not l.foliage or l.blade or l.flower { continue }
|
|
if l.grass and sc_grass_replaced(render3d_st) { continue }
|
|
if render3d_st.r3d_no_trees and l.imp != null { continue }
|
|
layer_update(render3d_st, l)
|
|
if l.n_lods > 1 and l.g_on {
|
|
render3d_st.sc_ind_base = 0; render3d_st.sc_ind_n = l.n_lods
|
|
layer_draw_depth(render3d_st, l, l.g_model, l.g_dst, 1)
|
|
render3d_st.sc_ind_base = -1; render3d_st.sc_ind_n = 1
|
|
} else if l.n_lods > 1 {
|
|
# (a tree layer is flagged `card` for its distant level; its mesh levels still count)
|
|
for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { layer_draw_depth(render3d_st, l, l.lods[k], l.lod_buf[k], l.n_lod[k]) } }
|
|
} else if not l.card {
|
|
layer_draw_depth(render3d_st, l, l.model, l.buf, l.n_near)
|
|
}
|
|
}
|
|
gpu_cull(render3d_st, true)
|
|
}
|
|
# a layer the game flagged `grass` is ground cover the GPU blades stand in for: off while they draw
|
|
function sc_grass_replaced(render3d_st: Render3dState) -> bool { return render3d_st.sc_skip_grass or ((render3d_st.grass_on or render3d_st.grass_force) and not render3d_st.grass_env_off and render3d_st.grass_prog != 0) }
|
|
function scatter_draw(render3d_st: mut Render3dState) -> void {
|
|
for i in 0 .. len(render3d_st.sc_layers) {
|
|
let l = render3d_st.sc_layers[i]
|
|
if render3d_st.sc_skip_blade and l.blade { continue }
|
|
if render3d_st.sc_skip_flower and l.flower { continue }
|
|
if render3d_st.sc_skip_card and l.card { continue }
|
|
if l.grass and sc_grass_replaced(render3d_st) { continue }
|
|
if render3d_st.r3d_no_trees and l.imp != null and not l.card { continue }
|
|
layer_update(render3d_st, l)
|
|
layer_draw_near(render3d_st, l, false, null, false)
|
|
layer_draw_far(render3d_st, l, false, null)
|
|
}
|
|
gpu_cull(render3d_st, true)
|
|
}
|
|
# Nothing here is keyed off the cascade. Every skip that used to be — ground cover past
|
|
# the 250 m cascade, blades past the nearest, the scanned mesh past the second — made a
|
|
# whole class of caster vanish at a fixed distance, which is exactly the popping. A layer
|
|
# with an impostor now casts its entire instance list from the card in every cascade;
|
|
# only layers that have no impostor at all fall back to the mesh.
|
|
const SC_IMP_CAST_CASCADE: int = 2
|
|
function scatter_draw_casters(render3d_st: mut Render3dState, light_vp: floats) -> void {
|
|
for i in 0 .. len(render3d_st.sc_layers) {
|
|
let l = render3d_st.sc_layers[i]
|
|
if render3d_st.sc_skip_blade and l.blade { continue }
|
|
if render3d_st.sc_skip_card and l.card { continue }
|
|
if l.grass and sc_grass_replaced(render3d_st) { continue }
|
|
if render3d_st.r3d_no_trees and l.imp != null and not l.card { continue }
|
|
layer_update(render3d_st, l)
|
|
if l.imp != null {
|
|
# the impostor casts only into the far cascades: every tree on the map was drawn as a card
|
|
# into the near two as well, whose receivers (within 60 m) are shaded by the near trees'
|
|
# own LOD2 meshes, cast just below
|
|
if render3d_st.sh_cascade >= SC_IMP_CAST_CASCADE { layer_draw_shadow(render3d_st, l, light_vp) }
|
|
# Shadow LOD (the practice in every production engine: a caster uses a low mesh LOD,
|
|
# the billboard only far away). The card alone is a side-view silhouette and a
|
|
# crown of drooping needle cards is mostly slivers from the side, so the sun, which
|
|
# sees the crown from above, cast a trunk line with a few blobs. The near levels
|
|
# now also cast their LOD2 mesh, alpha-tested, on top of the card.
|
|
if l.n_lods > 2 and l.g_on {
|
|
render3d_st.sc_ind_base = 17; render3d_st.sc_ind_n = 3; render3d_st.sc_ind_stride = 3
|
|
layer_draw_model(render3d_st, l, l.g_model_sh, l.g_dst, 1, false, true, light_vp)
|
|
render3d_st.sc_ind_base = -1; render3d_st.sc_ind_n = 1; render3d_st.sc_ind_stride = 4
|
|
} else if l.n_lods > 2 {
|
|
for k in 0 .. 3 { layer_draw_model(render3d_st, l, l.lods[2], l.lod_buf[k], l.n_lod[k], false, true, light_vp) }
|
|
}
|
|
}
|
|
else { layer_draw_near(render3d_st, l, true, light_vp, true) }
|
|
}
|
|
prof_cpu_mark(render3d_st, "shadow scatter")
|
|
}
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# The distant-grass carpet: the clump cards rendered straight down into one tiling
|
|
# tile, so the ground beyond the blade rings carries the same clumps, colours and
|
|
# gaps as the near cover instead of a lawn scan (the far-field trick the big open
|
|
# worlds use: geometry up close, an authored ground texture that matches it beyond).
|
|
# Returns an RGBA8 texture (alpha = coverage), repeat-wrapped and mipmapped.
|
|
function cb_rnd(render3d_st: mut Render3dState) -> float {
|
|
render3d_st.cb_state = (render3d_st.cb_state * 1103515245 + 12345) & 0x7FFFFFFF
|
|
return float((render3d_st.cb_state >> 8) & 0xFFFF) / 65536.0
|
|
}
|
|
# the GPU memory it makes is counted as VKM_SCATTER (R3D_VKMEM)
|
|
function carpet_bake(render3d_st: mut Render3dState, layers: []Layer, count: int, tile: float, res: int) -> int {
|
|
let was = render3d_st.gvk_tag
|
|
render3d_st.gvk_tag = VKM_SCATTER
|
|
let r = carpet_bake__t(render3d_st, layers, count, tile, res)
|
|
render3d_st.gvk_tag = was
|
|
return r
|
|
}
|
|
function carpet_bake__t(render3d_st: mut Render3dState, layers: []Layer, count: int, tile: float, res: int) -> int {
|
|
let tex = tex_target(render3d_st, res, res, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
let fbo = gpu_fb_new(render3d_st)
|
|
gpu_fb_bind(render3d_st, fbo)
|
|
gpu_fb_color(render3d_st, 0, tex)
|
|
let rb = gpu_rb_new(render3d_st)
|
|
gpu_rb_storage(render3d_st, rb, GL_DEPTH_COMPONENT24, res, res, 0)
|
|
gpu_fb_depth_rb(render3d_st, rb)
|
|
gpu_fb_draw_buffers(render3d_st, 1)
|
|
gpu_viewport(render3d_st, 0, 0, res, res)
|
|
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 0.0)
|
|
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
|
gpu_depth_test(render3d_st, true)
|
|
gpu_depth_func(render3d_st, GL_LESS)
|
|
gpu_cull(render3d_st, false)
|
|
gpu_blend(render3d_st, false)
|
|
# the clumps, and eight wrapped copies so the tile's edges continue
|
|
let half = tile * 0.5
|
|
let n9 = count * 9
|
|
let inst = gl_floats(n9 * INST_FLOATS)
|
|
render3d_st.cb_state = 977
|
|
var k = 0
|
|
for i in 0 .. count {
|
|
let x = cb_rnd(render3d_st) * tile - half
|
|
let z = cb_rnd(render3d_st) * tile - half
|
|
let sc = 1.5 + cb_rnd(render3d_st)
|
|
let yaw = cb_rnd(render3d_st) * (2.0 * PI)
|
|
let sd = cb_rnd(render3d_st)
|
|
for oz in 0 .. 3 {
|
|
for ox in 0 .. 3 {
|
|
let px = x + float(ox - 1) * tile
|
|
let pz = z + float(oz - 1) * tile
|
|
gl_put_bits(inst, k, float_bits(px)); gl_put_bits(inst, k + 1, float_bits(0.0)); gl_put_bits(inst, k + 2, float_bits(pz)); gl_put_bits(inst, k + 3, float_bits(sc))
|
|
gl_put_bits(inst, k + 4, float_bits(Math.sin(yaw))); gl_put_bits(inst, k + 5, float_bits(Math.cos(yaw))); gl_put_bits(inst, k + 6, float_bits(sd)); gl_put_bits(inst, k + 7, float_bits(0.0))
|
|
k += INST_FLOATS
|
|
}
|
|
}
|
|
}
|
|
let buf = gpu_buffer_new(render3d_st)
|
|
gpu_buffer_upload(render3d_st, buf, gl_bytes_of(n9 * INST_FLOATS), inst, GPU_STATIC)
|
|
free(inst)
|
|
# straight down: the window is exactly one tile
|
|
let view = m4_new(); let proj = m4_new()
|
|
let eye = v3_new(0.0, 6.0, 0.0); let at = v3_new(0.0, 0.0, 0.0); let up = v3_new(0.0, 0.0, -1.0)
|
|
m4_look_at(view, eye, at, up)
|
|
m4_ortho(proj, -half, half, -half, half, 0.1, 12.0)
|
|
let bake = render3d_st.sc_bake_card_prog
|
|
gpu_use_program(render3d_st, bake)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, bake, "u_view"), view)
|
|
u_mat4(render3d_st, gpu_uniform(render3d_st, bake, "u_proj"), proj)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_wind"), 0.0)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_flutter"), 0.0)
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_time"), 0.0)
|
|
for li in 0 .. len(layers) {
|
|
let l = layers[li]
|
|
if l.atlas == null { continue }
|
|
u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_card_w"), l.atlas.radius * 2.0); u_f(render3d_st, gpu_uniform(render3d_st, bake, "u_card_h"), l.atlas.height)
|
|
r3d_bind_2d(render3d_st, bake, "u_diff", 0, l.atlas.albedo)
|
|
r3d_bind_2d(render3d_st, bake, "u_arm", 2, l.atlas.normal)
|
|
for i in 0 .. len(l.model.prims) {
|
|
let pr = l.model.prims[i]
|
|
scatter_attach(render3d_st, pr.mesh, buf)
|
|
mesh_draw_instanced(render3d_st, pr.mesh, n9)
|
|
}
|
|
}
|
|
free(view); free(proj); free(eye); free(at); free(up)
|
|
gpu_fb_bind(render3d_st, 0)
|
|
gpu_fb_free(render3d_st, fbo)
|
|
gpu_rb_free(render3d_st, rb)
|
|
gpu_buffer_free(render3d_st, buf)
|
|
carpet_finish(render3d_st, tex)
|
|
gpu_check(render3d_st, "carpet bake")
|
|
return tex
|
|
}
|
|
# the carpet tile made ready to wear: repeat-wrapped and mipmapped (a bake's too, bake_load.ludic)
|
|
function carpet_finish(render3d_st: mut Render3dState, tex: int) -> void {
|
|
gpu_tex_bind(render3d_st, GPU_TEX2D, tex)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_REPEAT)
|
|
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_paramf(render3d_st, GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, render3d_st.tex_anisotropy)
|
|
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
|
}
|
|
|
|
# ---- another map at run time ------------------------------------------------------------
|
|
# Every scattered layer released - its instance arrays and GL buffers, a card layer's own
|
|
# crossed-card mesh and baked atlas - and every stream feeding them. The models a layer drew
|
|
# belong to whoever loaded them and are kept, and so are the programs. A game rebuilding its
|
|
# world calls this, then places the new map's layers exactly as it did at boot.
|
|
function scatter_clear_all(render3d_st: mut Render3dState) -> void {
|
|
stream_clear_all(render3d_st)
|
|
if render3d_st.sc_layers == null { return }
|
|
for i in 0 .. len(render3d_st.sc_layers) {
|
|
let l = render3d_st.sc_layers[i]
|
|
if l.buf != 0 { gpu_buffer_free(render3d_st, l.buf) }
|
|
if l.imp_buf != 0 { gpu_buffer_free(render3d_st, l.imp_buf) }
|
|
if l.sh_buf != 0 { gpu_buffer_free(render3d_st, l.sh_buf) }
|
|
if l.fog_buf != 0 { gpu_buffer_free(render3d_st, l.fog_buf) }
|
|
if l.lod_buf != null { for k in 0 .. l.n_lods { gpu_buffer_free(render3d_st, l.lod_buf[k]) } }
|
|
if l.inst != null { free(l.inst) }
|
|
if l.scratch != null { free(l.scratch) }
|
|
if l.tint != null { free(l.tint) }
|
|
if l.last_cam != null { free(l.last_cam) }
|
|
if l.gstart != null { free(l.gstart) }
|
|
if l.gsorted != null { free(l.gsorted) }
|
|
if l.gymin != null { free(l.gymin) }
|
|
if l.gymax != null { free(l.gymax) }
|
|
if l.vis != null { free(l.vis) }
|
|
if l.lod_dist != null { free(l.lod_dist); free(l.lod_card); free(l.lod_buf); free(l.n_lod) }
|
|
if l.lvl != null { free(l.lvl) }
|
|
if l.g_arena != null { for k in 0 .. len(l.g_arena) { mesh_free(render3d_st, l.g_arena[k].mesh) } }
|
|
# layer_cards built this layer's crossed card and its atlas itself
|
|
if l.card and l.n_lods == 0 and l.model != null { for k in 0 .. len(l.model.prims) { mesh_free(render3d_st, l.model.prims[k].mesh) } }
|
|
if l.card and l.atlas != null {
|
|
gpu_tex_free(render3d_st, l.atlas.albedo)
|
|
gpu_tex_free(render3d_st, l.atlas.normal)
|
|
}
|
|
}
|
|
render3d_st.sc_layers = new []Layer
|
|
render3d_st.sc_view_gen += 1
|
|
}
|
|
|
|
# R3D_DUMP_ATLAS / the LOD debug switch: what a layer holds, printed (text made only then)
|
|
@alloc_ok("debug output, only under R3D_DUMP_ATLAS and the LOD debug switch")
|
|
function sc_debug_lods(render3d_st: Render3dState, l: Layer, counts: words, n: int, total: int, src: floats) -> void {
|
|
print(`lod partition: total {total} dropped {counts[n + 1]} far {counts[n]} l0 {counts[0]} l1 {counts[1]} l2 {counts[2]} l3 {counts[3]} dist0 {fixed(l.lod_dist[0])} dist3 {fixed(l.lod_dist[n - 1])} cull {fixed(l.cull)} cam {fixed(render3d_st.cam_pos[0])} {fixed(render3d_st.cam_pos[2])} first {fixed(src[0])} {fixed(src[2])}`)
|
|
}
|
|
@alloc_ok("debug output, only under R3D_DUMP_ATLAS and the LOD debug switch")
|
|
function sc_debug_layer(l: Layer, tmp: floats, nn: int, nf: int) -> void {
|
|
if l.imp != null and nn > 0 { print(`near full-mesh instances: {nn} (first at {fixed(tmp[0])} {fixed(tmp[1])} {fixed(tmp[2])})`) }
|
|
if l.imp != null {
|
|
print(`layer: near {nn} far {nf}`)
|
|
for k in 0 .. nn {
|
|
let q = k * INST_FLOATS
|
|
print(` near {fixed(tmp[q])} {fixed(tmp[q + 1])} {fixed(tmp[q + 2])} s {fixed(tmp[q + 3])}`)
|
|
}
|
|
}
|
|
}
|