Grass (Vulkan with multi-draw indirect): every visible tile is a record in one buffer, uploaded once a frame, and each band draws its records 256 at a time. A record's firstInstance is its place in the chunk times 65536; grass.vert's TILES variant reads that place's corner and indices per cell from u_tiles. R3D_GRASS_TILES=1 keeps a draw per tile. OpenGL is unchanged. Casters: a LOD level with no impostor is drawn into a shadow cascade only when its distance band, widened by six times its height, the camera's height over the ground and the frustum's corner reach, can touch that cascade's receivers. The flowers' mesh levels (6 - 30 m) leave the three outer cascades. R3D_CAST_ALL=1 draws every level everywhere. OpenGL frames byte-identical at all five viewpoints; alpha-tested shadow draws at a 460 -> 244. gpu_has_mdi() guards both this and the GPU-culled trees' multi-record draws. Camp: Mac Vulkan 2645 -> 2191 (grass) -> 2034 draws; PC 2657 -> 2046, self-tests 59/59, validation 0. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
1355 lines
58 KiB
Text
1355 lines
58 KiB
Text
# ============================================================================
|
|
# scatter.ludic — instanced vegetation and props. A Layer is one model placed
|
|
# many times (position, scale, yaw, seed, wind weight per instance). Each frame
|
|
# the instances are split by distance: the near ones draw as the full scanned
|
|
# mesh, the far ones as impostor cards baked from that mesh at load.
|
|
# ============================================================================
|
|
|
|
const INST_FLOATS: int = 8
|
|
|
|
property Impostor {
|
|
albedo: int = 0,
|
|
normal: int = 0,
|
|
tiles: int = 16,
|
|
radius: int = 0,
|
|
height: int = 0
|
|
}
|
|
property Layer {
|
|
model: Model,
|
|
imp: Impostor,
|
|
foliage: bool = false,
|
|
wind: int = 0, # float bits
|
|
tint: words,
|
|
inst: words, # INST_FLOATS per instance
|
|
count: int = 0,
|
|
cap: int = 0,
|
|
near: int = 0, # float bits; instances beyond it draw as impostors (or not at all)
|
|
cull: int = 0, # float bits; instances beyond it are skipped (0 = never)
|
|
buf: int = 0,
|
|
n_near: int = 0,
|
|
imp_buf: int = 0,
|
|
sh_buf: int = 0, # every instance, for shadow casting (no cull, no LOD split)
|
|
n_sh: int = 0,
|
|
n_far: int = 0,
|
|
scratch: words,
|
|
last_cam: words,
|
|
rough: int = 0,
|
|
blade: bool = false,
|
|
flower: bool = false,
|
|
card: bool = false,
|
|
cheap: bool = false, # distant cover: no shadows, no wind, flat lighting
|
|
atlas: Impostor,
|
|
streamed: bool = false, # fed by a Stream: already frustum-culled per chunk, no split needed
|
|
grounded: bool = false, # the vertex shader stands each instance on the drawn terrain
|
|
view_gen: int = -1, # sc_view_gen this layer's partition was built for
|
|
# static layers with many instances are sorted into a cell grid once, and only the
|
|
# cells inside the view frustum (and within cull) are partitioned each frame
|
|
gcell: int = 0, # cell size (float bits); 0 = no grid
|
|
gx0: int = 0,
|
|
gz0: int = 0,
|
|
gnx: int = 0,
|
|
gnz: int = 0,
|
|
gstart: words, # per cell: first index into gsorted (ncell + 1 entries)
|
|
gsorted: words, # the instances, grouped by cell
|
|
gymin: words, # per cell height range (float bits)
|
|
gymax: words,
|
|
vis: words, # the instances gathered from visible cells this frame
|
|
n_vis: int = 0,
|
|
# A LOD chain: lods[k] is drawn for instances within lod_dist[k] (and beyond lod_dist[k-1]);
|
|
# past the last level the impostor takes over (or, if the last distance is 0, the last
|
|
# level runs out to the cull distance). lod_card[k] = 1 marks a level that is the layer's
|
|
# crossed card carrying its atlas (cover keeps its baked card as the far level).
|
|
lods: []Model,
|
|
n_lods: int = 0,
|
|
# The GPU-culled path (Vulkan, phase 38): every instance in g_src, a compute pass packs the
|
|
# visible ones per bucket into g_dst and writes the instance counts of the draw records in
|
|
# g_cmds (see scatter_cull.comp for the record layout). g_on once it is set up for g_n instances.
|
|
g_on: bool = false,
|
|
g_n: int = 0,
|
|
g_src: int = 0,
|
|
g_dst: int = 0,
|
|
g_cmds: int = 0,
|
|
g_counts: int = 0,
|
|
g_arena: []Prim, # one merged mesh per material, holding every level's copy (layer_arena_build)
|
|
g_first: words, # material * 4 + level: that level's first index in the merged mesh
|
|
g_base: words, # ... and its first vertex
|
|
g_model: Model, # the merged meshes as a model the draws take (level 0's height)
|
|
g_model_sh: Model, # the same with level 2's height, for the shadow LOD
|
|
lod_dist: words,
|
|
lod_card: words,
|
|
lod_buf: words,
|
|
n_lod: words,
|
|
lvl: words # scratch: the level chosen per gathered instance
|
|
}
|
|
|
|
# the procedural models' one layout: position, normal, uv, interleaved at 32 bytes
|
|
function sc_model_layout(m: Mesh) -> void {
|
|
gpu_mesh_attr(m, 0, 3, GPU_F32, 32, 0, false)
|
|
gpu_mesh_attr(m, 1, 3, GPU_F32, 32, 12, false)
|
|
gpu_mesh_attr(m, 2, 2, GPU_F32, 32, 24, false)
|
|
}
|
|
|
|
# Two crossed unit quads (x in [-0.5, 0.5], y in [0, 1]), attribute 0 = pos,
|
|
# 1 = the quad's facing normal, 2 = uv. Scaled per layer to the atlas card size.
|
|
function model_cross_card() -> Model {
|
|
let model = new Model
|
|
model.prims = new []Prim
|
|
let pr = new Prim
|
|
let m = gpu_mesh_new()
|
|
let v = gl_floats(8 * 8)
|
|
var k = 0
|
|
for q in 0 .. 2 {
|
|
for c in 0 .. 4 {
|
|
var sx = f_neg(F_HALF); var sy = F_ZERO; var u = F_ZERO; var vv = F_ZERO
|
|
if c == 1 or c == 2 { sx = F_HALF; u = F_ONE }
|
|
if c == 2 or c == 3 { sy = F_ONE; vv = F_ONE }
|
|
if q == 0 { gl_put_bits(v, k, sx); gl_put_bits(v, k + 1, sy); gl_put_bits(v, k + 2, F_ZERO); gl_put_bits(v, k + 3, F_ZERO); gl_put_bits(v, k + 4, F_ZERO); gl_put_bits(v, k + 5, f_neg1()) }
|
|
else { gl_put_bits(v, k, F_ZERO); gl_put_bits(v, k + 1, sy); gl_put_bits(v, k + 2, sx); gl_put_bits(v, k + 3, f_neg1()); gl_put_bits(v, k + 4, F_ZERO); gl_put_bits(v, k + 5, F_ZERO) }
|
|
gl_put_bits(v, k + 6, u); gl_put_bits(v, k + 7, vv)
|
|
k += 8
|
|
}
|
|
}
|
|
gpu_mesh_vertices(m, v, gl_bytes_of(64), GPU_STATIC)
|
|
sc_model_layout(m)
|
|
free(v)
|
|
let idx = words(12)
|
|
idx[0] = 0; idx[1] = 1; idx[2] = 2; idx[3] = 0; idx[4] = 2; idx[5] = 3
|
|
idx[6] = 4; idx[7] = 5; idx[8] = 6; idx[9] = 4; idx[10] = 6; idx[11] = 7
|
|
gpu_mesh_indices(m, idx, 48, 4)
|
|
free(idx)
|
|
m.count = 12
|
|
gpu_mesh_done(m)
|
|
pr.mesh = m
|
|
if gltf_white == 0 { gltf_white = tex_solid(200, 200, 200, 255); gltf_flat = tex_solid(128, 128, 255, 255) }
|
|
pr.diff = gltf_white; pr.nrm = gltf_flat; pr.arm = gltf_white
|
|
push(model.prims, pr)
|
|
model.radius = F_HALF; model.height = F_ONE; model.tris = 4
|
|
return model
|
|
}
|
|
|
|
# A card layer: crossed cards carrying a single-tile atlas baked from `scan`.
|
|
function layer_cards(scan: Model, cap: int, wind: int, cull: int) -> Layer {
|
|
let l = layer_new(model_cross_card(), cap, true, wind, F_ZERO, cull)
|
|
l.card = true
|
|
l.atlas = impostor_bake(scan, 1, 512, 512)
|
|
return l
|
|
}
|
|
|
|
# A lupine spike (1 m tall): a stem of two crossed quads (uv.x in [0,1]) and
|
|
# seven tiers of crossed floret quads (uv.x in [1,2]), coloured in the shader.
|
|
function model_lupine() -> Model {
|
|
let model = new Model
|
|
model.prims = new []Prim
|
|
let pr = new Prim
|
|
let m = gpu_mesh_new()
|
|
# quads: stem x2 + tiers 12 x 2 + 3 leaves = 29 quads
|
|
let nq = 29
|
|
let v = gl_floats(nq * 4 * 8)
|
|
let idx = words(nq * 6)
|
|
var k = 0
|
|
var qi = 0
|
|
for q in 0 .. nq {
|
|
var w = fl(0.012); var y0 = F_ZERO; var y1 = fl(0.62); var ukind = F_ZERO
|
|
var ang = F_ZERO
|
|
if q >= 2 and q < 26 {
|
|
let tier = (q - 2) / 2
|
|
let t = fr(tier, 12)
|
|
w = f_mul(fl(0.05), f_sub(fl(1.1), t))
|
|
y0 = f_add(fl(0.27), f_mul(t, fl(0.36)))
|
|
y1 = f_add(y0, fl(0.045))
|
|
ukind = F_ONE
|
|
ang = f_mul(fr(tier, 12), fl(2.1))
|
|
if (q & 1) == 1 { ang = f_add(ang, f_mul(F_PI, F_HALF)) }
|
|
} else if q >= 26 {
|
|
# a rosette of three leaves near the ground
|
|
w = fl(0.09); y0 = fl(0.02); y1 = fl(0.2); ukind = F_TWO
|
|
ang = f_mul(fr(q - 26, 3), f_mul(F_TWO, F_PI))
|
|
} else {
|
|
if (q & 1) == 1 { ang = f_add(ang, f_mul(F_PI, F_HALF)) }
|
|
}
|
|
let cx = f_mul(f_cos(ang), w); let cz = f_mul(f_sin(ang), w)
|
|
for c in 0 .. 4 {
|
|
var sx = f_neg1(); var sy = y0; var u = F_ZERO
|
|
if c == 1 or c == 2 { sx = F_ONE; u = F_ONE }
|
|
if c == 2 or c == 3 { sy = y1 }
|
|
gl_put_bits(v, k, f_mul(cx, sx)); gl_put_bits(v, k + 1, sy); gl_put_bits(v, k + 2, f_mul(cz, sx))
|
|
gl_put_bits(v, k + 3, f_neg(cz)); gl_put_bits(v, k + 4, fl(0.2)); gl_put_bits(v, k + 5, cx)
|
|
gl_put_bits(v, k + 6, f_add(ukind, u))
|
|
var vv = sy
|
|
if ukind == F_ONE { vv = f_div(f_sub(sy, fl(0.27)), fl(0.4)) }
|
|
if ukind == F_TWO { vv = f_div(f_sub(sy, fl(0.02)), fl(0.18)) }
|
|
gl_put_bits(v, k + 7, vv)
|
|
k += 8
|
|
}
|
|
let b = q * 4
|
|
idx[qi] = b; idx[qi + 1] = b + 1; idx[qi + 2] = b + 2; idx[qi + 3] = b; idx[qi + 4] = b + 2; idx[qi + 5] = b + 3
|
|
qi += 6
|
|
}
|
|
gpu_mesh_vertices(m, v, gl_bytes_of(nq * 4 * 8), GPU_STATIC)
|
|
sc_model_layout(m)
|
|
free(v)
|
|
gpu_mesh_indices(m, idx, nq * 6 * 4, 4)
|
|
free(idx)
|
|
m.count = nq * 6
|
|
gpu_mesh_done(m)
|
|
pr.mesh = m
|
|
if gltf_white == 0 { gltf_white = tex_solid(200, 200, 200, 255); gltf_flat = tex_solid(128, 128, 255, 255) }
|
|
pr.diff = gltf_white; pr.nrm = gltf_flat; pr.arm = gltf_white
|
|
push(model.prims, pr)
|
|
model.radius = fl(0.08); model.height = fl(0.65); model.tris = nq * 2
|
|
return model
|
|
}
|
|
|
|
# A dense lupine for baking into a card: a stem, ~220 small floret quads in a
|
|
# tapering spiral (uv.x in [1,2]) and five leaves (uv.x in [2,3]).
|
|
function model_lupine_dense() -> Model {
|
|
let model = new Model
|
|
model.prims = new []Prim
|
|
let pr = new Prim
|
|
let m = gpu_mesh_new()
|
|
let nfl = 220
|
|
let nq = 2 + nfl + 5
|
|
let v = gl_floats(nq * 4 * 8)
|
|
let idx = words(nq * 6)
|
|
var k = 0
|
|
var qi = 0
|
|
seed(5)
|
|
for q in 0 .. nq {
|
|
var w = fl(0.008); var y0 = F_ZERO; var y1 = fl(0.66); var ukind = F_ZERO
|
|
var ang = F_ZERO; var ox = F_ZERO; var oz = F_ZERO; var tilt = F_ZERO
|
|
if q >= 2 and q < 2 + nfl {
|
|
let t = fr(q - 2, nfl)
|
|
let yy = f_add(fl(0.28), f_mul(t, fl(0.4)))
|
|
ang = f_mul(fi(q), fl(2.39996)) # golden angle spiral
|
|
let rad = f_mul(fl(0.055), f_sub(fl(1.05), t))
|
|
ox = f_mul(f_cos(ang), rad); oz = f_mul(f_sin(ang), rad)
|
|
w = f_mul(fl(0.028), f_sub(fl(1.1), f_mul(t, fl(0.5))))
|
|
y0 = f_sub(yy, fl(0.016)); y1 = f_add(yy, fl(0.016))
|
|
ukind = F_ONE
|
|
tilt = fl(0.6)
|
|
} else if q >= 2 + nfl {
|
|
w = fl(0.05); y0 = fl(0.03); y1 = fl(0.16); ukind = F_TWO
|
|
ang = f_mul(fr(q - 2 - nfl, 5), f_mul(F_TWO, F_PI))
|
|
ox = f_mul(f_cos(ang), fl(0.05)); oz = f_mul(f_sin(ang), fl(0.05))
|
|
} else {
|
|
if (q & 1) == 1 { ang = f_mul(F_PI, F_HALF) }
|
|
}
|
|
# the quad faces outward (its normal along the spiral radius), leaning out by `tilt`
|
|
let nx = f_cos(ang); let nz = f_sin(ang)
|
|
let tx = f_neg(nz); let tz = nx # tangent (quad width direction)
|
|
for c in 0 .. 4 {
|
|
var sx = f_neg1(); var sy = y0; var u = F_ZERO
|
|
if c == 1 or c == 2 { sx = F_ONE; u = F_ONE }
|
|
if c == 2 or c == 3 { sy = y1 }
|
|
var lean = F_ZERO
|
|
if c == 2 or c == 3 { lean = f_mul(tilt, w) }
|
|
gl_put_bits(v, k, f_add(f_add(ox, f_mul(tx, f_mul(sx, w))), f_mul(nx, lean)))
|
|
gl_put_bits(v, k + 1, sy)
|
|
gl_put_bits(v, k + 2, f_add(f_add(oz, f_mul(tz, f_mul(sx, w))), f_mul(nz, lean)))
|
|
gl_put_bits(v, k + 3, nx); gl_put_bits(v, k + 4, fl(0.35)); gl_put_bits(v, k + 5, nz)
|
|
gl_put_bits(v, k + 6, f_add(ukind, u))
|
|
var vv = sy
|
|
if ukind == F_ONE { vv = f_div(f_sub(sy, fl(0.27)), fl(0.42)) }
|
|
if ukind == F_TWO { vv = f_div(f_sub(sy, fl(0.03)), fl(0.19)) }
|
|
gl_put_bits(v, k + 7, vv)
|
|
k += 8
|
|
}
|
|
let b = q * 4
|
|
idx[qi] = b; idx[qi + 1] = b + 1; idx[qi + 2] = b + 2; idx[qi + 3] = b; idx[qi + 4] = b + 2; idx[qi + 5] = b + 3
|
|
qi += 6
|
|
}
|
|
gpu_mesh_vertices(m, v, gl_bytes_of(nq * 4 * 8), GPU_STATIC)
|
|
sc_model_layout(m)
|
|
free(v)
|
|
gpu_mesh_indices(m, idx, nq * 6 * 4, 4)
|
|
free(idx)
|
|
m.count = nq * 6
|
|
gpu_mesh_done(m)
|
|
pr.mesh = m
|
|
if gltf_white == 0 { gltf_white = tex_solid(200, 200, 200, 255); gltf_flat = tex_solid(128, 128, 255, 255) }
|
|
pr.diff = gltf_white; pr.nrm = gltf_flat; pr.arm = gltf_white
|
|
push(model.prims, pr)
|
|
model.radius = fl(0.11); model.height = fl(0.68); model.tris = nq * 2
|
|
return model
|
|
}
|
|
|
|
# A procedural grass blade (1 m tall, 5 cm wide, curved): 5 rows of 2 vertices.
|
|
function model_blade() -> Model {
|
|
let model = new Model
|
|
model.prims = new []Prim
|
|
let pr = new Prim
|
|
let m = gpu_mesh_new()
|
|
let rows = 5
|
|
let v = gl_floats(rows * 2 * 8)
|
|
var k = 0
|
|
for r in 0 .. rows {
|
|
let t = fr(r, rows - 1)
|
|
# never a zero-width tip: a sliver triangle extrapolates its attributes wildly
|
|
let taper = f_max(f_sub(F_ONE, f_mul(t, f_mul(t, f_sqrt(t)))), fl(0.12))
|
|
let hw = f_mul(fl(0.05), taper)
|
|
let bend = f_mul(f_mul(t, t), fl(0.28))
|
|
for sd in 0 .. 2 {
|
|
var x = f_neg(hw)
|
|
if sd == 1 { x = hw }
|
|
gl_put_bits(v, k, x); gl_put_bits(v, k + 1, t); gl_put_bits(v, k + 2, bend)
|
|
gl_put_bits(v, k + 3, F_ZERO); gl_put_bits(v, k + 4, fl(0.3)); gl_put_bits(v, k + 5, F_ONE)
|
|
gl_put_bits(v, k + 6, fi(sd)); gl_put_bits(v, k + 7, t)
|
|
k += 8
|
|
}
|
|
}
|
|
gpu_mesh_vertices(m, v, gl_bytes_of(rows * 2 * 8), GPU_STATIC)
|
|
sc_model_layout(m)
|
|
free(v)
|
|
let ni = (rows - 1) * 6
|
|
let idx = words(ni)
|
|
k = 0
|
|
for r in 0 .. rows - 1 {
|
|
let a = r * 2
|
|
idx[k] = a; idx[k + 1] = a + 1; idx[k + 2] = a + 2
|
|
idx[k + 3] = a + 1; idx[k + 4] = a + 3; idx[k + 5] = a + 2
|
|
k += 6
|
|
}
|
|
gpu_mesh_indices(m, idx, ni * 4, 4)
|
|
free(idx)
|
|
m.count = ni
|
|
gpu_mesh_done(m)
|
|
pr.mesh = m
|
|
if gltf_white == 0 { gltf_white = tex_solid(200, 200, 200, 255); gltf_flat = tex_solid(128, 128, 255, 255) }
|
|
pr.diff = gltf_white; pr.nrm = gltf_flat; pr.arm = gltf_white
|
|
push(model.prims, pr)
|
|
model.radius = fl(0.05); model.height = F_ONE; model.tris = ni / 3
|
|
return model
|
|
}
|
|
|
|
var sc_prog: int = 0
|
|
var sc_prog_fol: int = 0
|
|
var sc_prog_wind: int = 0
|
|
var sc_prog_blade: int = 0
|
|
var sc_prog_flower: int = 0
|
|
var sc_prog_card: int = 0
|
|
var sc_prog_card_shadow: int = 0
|
|
var sc_prog_card_cheap: int = 0
|
|
var sc_blade_base: words = null
|
|
var sc_blade_tip: words = null
|
|
var sc_blade_tint: words = null
|
|
var sc_prog_shadow: int = 0
|
|
var sc_prog_shadow_wind: int = 0
|
|
var sc_prog_shadow_fol: int = 0 # foliage meshes: alpha-tested casters
|
|
# The foliage depth prepass (render.ludic): the near tree LODs write depth first with a
|
|
# shader that only runs the alpha test, then the lit pass shades them with no discard and
|
|
# an equal depth test, so a pixel of needles is lit once rather than once for every card
|
|
# stacked behind it. In a dense stand at 4K that overdraw was the largest pass in the frame.
|
|
var sc_prog_fol_depth: int = 0
|
|
var sc_prog_fol_eq: int = 0
|
|
var sc_prepass: bool = false # the main pass is drawing over what the prepass laid down
|
|
var sc_imp_prog: int = 0
|
|
var sc_imp_prog_shadow: int = 0
|
|
var sc_bake_prog: int = 0
|
|
var sc_bake_card_prog: int = 0
|
|
var sc_bake_flower_prog: int = 0
|
|
var sc_card: Mesh = null
|
|
var sc_ident_buf: int = 0
|
|
var sc_layers: []Layer = null
|
|
var sc_debug_dump: bool = false
|
|
var sc_printed: bool = false
|
|
var sc_a2c: bool = true
|
|
|
|
function scatter_init() -> void {
|
|
# R3D_DUMP_ATLAS: every impostor and card atlas the run bakes, to build/atlas_<n>_{color,alpha}.ppm
|
|
sc_debug_dump = Os.has_env("R3D_DUMP_ATLAS")
|
|
sc_prog = r3d_program("model.vert", "model.frag", "")
|
|
sc_prog_fol = r3d_program("model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define ALPHA_TEST\n")
|
|
sc_prog_fol_depth = r3d_program("model.vert", "depth.frag", "#define FOLIAGE\n#define WIND\n#define ALPHA_TEST\n")
|
|
sc_prog_fol_eq = r3d_program("model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define EQ_PASS\n")
|
|
sc_prog_wind = r3d_program("model.vert", "model.frag", "#define WIND\n")
|
|
sc_prog_blade = r3d_program("model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define BLADE\n")
|
|
sc_prog_flower = r3d_program("model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define FLOWER\n")
|
|
sc_prog_card = r3d_program("model.vert", "model.frag", "#define FOLIAGE\n#define WIND\n#define CARD\n")
|
|
sc_prog_card_shadow = r3d_program("model.vert", "model.frag", "#define SHADOW_PASS\n#define WIND\n#define CARD\n")
|
|
sc_prog_card_cheap = r3d_program("model.vert", "model.frag", "#define CARD\n#define CHEAP\n")
|
|
# a dry alpine meadow: brown-olive roots, straw with a little green at the tips
|
|
sc_blade_base = v3_new(fl(0.045), fl(0.06), fl(0.025))
|
|
sc_blade_tip = v3_new(fl(0.22), fl(0.27), fl(0.13))
|
|
sc_blade_tint = v3_new(F_ONE, F_ONE, F_ONE)
|
|
sc_prog_shadow = r3d_program("model.vert", "shadow.frag", "#define SHADOW_PASS\n")
|
|
sc_prog_shadow_wind = r3d_program("model.vert", "shadow.frag", "#define SHADOW_PASS\n#define WIND\n")
|
|
sc_prog_shadow_fol = r3d_program("model.vert", "shadow.frag", "#define SHADOW_PASS\n#define WIND\n#define ALPHA_TEST\n")
|
|
sc_imp_prog = r3d_program("impostor.vert", "impostor.frag", "")
|
|
sc_imp_prog_shadow = r3d_program("impostor.vert", "impostor.frag", "#define SHADOW_PASS\n")
|
|
sc_bake_prog = r3d_program("model.vert", "bake.frag", "")
|
|
sc_bake_flower_prog = r3d_program("model.vert", "bake.frag", "#define FLOWER\n")
|
|
sc_bake_card_prog = r3d_program("model.vert", "bake.frag", "#define CARD\n")
|
|
sc_card = mesh_card()
|
|
# a single identity instance, for baking
|
|
let one = gl_floats(INST_FLOATS)
|
|
for i in 0 .. INST_FLOATS { gl_put_bits(one, i, F_ZERO) }
|
|
gl_put_bits(one, 3, F_ONE); gl_put_bits(one, 5, F_ONE)
|
|
sc_ident_buf = gpu_buffer_new()
|
|
gpu_buffer_upload(sc_ident_buf, INST_FLOATS * 4, one, GPU_STATIC)
|
|
free(one)
|
|
sc_layers = new []Layer
|
|
}
|
|
|
|
# feed a mesh its instances from `buf`: attribute 3 = position + scale, 4 = sin, cos, seed, wind
|
|
function scatter_attach(m: Mesh, buf: int) -> void {
|
|
gpu_mesh_bind_instances(m, buf)
|
|
gpu_mesh_attr_inst(m, 3, 4, GPU_F32, INST_FLOATS * 4, 0)
|
|
gpu_mesh_attr_inst(m, 4, 4, GPU_F32, INST_FLOATS * 4, 16)
|
|
gpu_mesh_done(m)
|
|
}
|
|
|
|
function layer_new(model: Model, cap: int, foliage: bool, wind: int, near: int, cull: int) -> Layer {
|
|
let l = new Layer
|
|
l.model = model
|
|
l.cap = cap
|
|
l.foliage = foliage
|
|
l.wind = wind
|
|
l.near = near
|
|
l.cull = cull
|
|
l.tint = v3_new(F_ONE, F_ONE, F_ONE)
|
|
l.inst = words(cap * INST_FLOATS)
|
|
l.scratch = words(cap * INST_FLOATS)
|
|
l.last_cam = v3_new(fi(100000), F_ZERO, F_ZERO)
|
|
l.buf = gpu_buffer_new()
|
|
l.imp_buf = gpu_buffer_new()
|
|
l.sh_buf = gpu_buffer_new()
|
|
l.rough = F_ONE
|
|
for i in 0 .. len(model.prims) { scatter_attach(model.prims[i].mesh, l.buf) }
|
|
push(sc_layers, l)
|
|
return l
|
|
}
|
|
|
|
function layer_add(l: Layer, x: int, y: int, z: int, scale: int, yaw: int, seed: int, wind: int) -> void {
|
|
if l.count >= l.cap { return }
|
|
let o = l.count * INST_FLOATS
|
|
l.inst[o] = x; l.inst[o + 1] = y; l.inst[o + 2] = z; l.inst[o + 3] = scale
|
|
l.inst[o + 4] = f_sin(yaw); l.inst[o + 5] = f_cos(yaw); l.inst[o + 6] = seed; l.inst[o + 7] = wind
|
|
l.count += 1
|
|
}
|
|
|
|
# ---- impostors ---------------------------------------------------------------------
|
|
var sc_bake_flower: bool = false
|
|
var sc_dump_n: int = 0
|
|
function impostor_bake(model: Model, tiles: int, tw: int, th: int) -> Impostor {
|
|
let im = new Impostor
|
|
im.tiles = tiles
|
|
im.radius = f_mul(model.radius, fl(1.02))
|
|
im.height = model.height
|
|
let aw = tiles * tw
|
|
im.albedo = tex_target(aw, th, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
im.normal = tex_target(aw, th, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
let fbo = gpu_fb_new()
|
|
gpu_fb_bind(fbo)
|
|
gpu_fb_color(0, im.albedo)
|
|
gpu_fb_color(1, im.normal)
|
|
let rb = gpu_rb_new()
|
|
gpu_rb_storage(rb, GL_DEPTH_COMPONENT24, aw, th, 0)
|
|
gpu_fb_depth_rb(rb)
|
|
gpu_fb_draw_buffers(2)
|
|
gpu_viewport(0, 0, aw, th)
|
|
gpu_clear_color(0.0, 0.0, 0.0, 0.0)
|
|
gpu_clear(GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
|
gpu_depth_test(true)
|
|
gpu_depth_func(GL_LESS)
|
|
gpu_cull(false)
|
|
gpu_blend(false)
|
|
# the model's prims temporarily take the identity instance
|
|
for i in 0 .. len(model.prims) { scatter_attach(model.prims[i].mesh, sc_ident_buf) }
|
|
let view = m4_new(); let proj = m4_new()
|
|
let eye = words(3); let at = words(3); let up = v3_new(F_ZERO, F_ONE, F_ZERO)
|
|
let cy = f_add(model.ymin, f_mul(model.height, F_HALF))
|
|
let r = im.radius
|
|
let hh = f_mul(model.height, F_HALF)
|
|
var bake = sc_bake_prog
|
|
if sc_bake_flower { bake = sc_bake_flower_prog }
|
|
gpu_use_program(bake)
|
|
for t in 0 .. tiles {
|
|
let a = f_mul(f_mul(F_TWO, F_PI), fr(t, tiles))
|
|
v3_set(at, F_ZERO, cy, F_ZERO)
|
|
# a touch of elevation (the viewer usually looks slightly down at a tree)
|
|
v3_set(eye, f_mul(f_sin(a), f_mul(r, fi(4))), f_add(cy, f_mul(r, fl(0.5))), f_neg(f_mul(f_cos(a), f_mul(r, fi(4)))))
|
|
m4_look_at(view, eye, at, up)
|
|
m4_ortho(proj, f_neg(r), r, f_neg(hh), hh, fl(0.1), f_mul(r, fi(9)))
|
|
u_mat4(gpu_uniform(bake, "u_view"), view)
|
|
u_mat4(gpu_uniform(bake, "u_proj"), proj)
|
|
u_f(gpu_uniform(bake, "u_wind"), F_ZERO)
|
|
gpu_viewport(t * tw, 0, tw, th)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
r3d_bind_2d(bake, "u_diff", 0, pr.diff)
|
|
r3d_bind_2d(bake, "u_arm", 2, pr.arm)
|
|
mesh_draw_instanced(pr.mesh, 1)
|
|
}
|
|
}
|
|
free(view); free(proj); free(eye); free(at); free(up)
|
|
gpu_fb_bind(0)
|
|
gpu_fb_free(fbo)
|
|
gpu_rb_free(rb)
|
|
gpu_tex_bind(GPU_TEX2D, im.albedo)
|
|
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_mips(GPU_TEX2D)
|
|
gpu_tex_bind(GPU_TEX2D, im.normal)
|
|
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_mips(GPU_TEX2D)
|
|
gpu_check("impostor bake")
|
|
# numbered, so every bake of a run survives to be compared (a card layer per species bakes one)
|
|
if sc_debug_dump {
|
|
sc_dump_n += 1
|
|
tex_dump_alpha = true; tex_dump(im.albedo, aw, th, `build/atlas_{sc_dump_n}_alpha.ppm`); tex_dump_alpha = false
|
|
tex_dump(im.albedo, aw, th, `build/atlas_{sc_dump_n}_color.ppm`)
|
|
}
|
|
return im
|
|
}
|
|
|
|
# Give a layer a LOD chain. `dists` (float bits) are the outer distances of each level;
|
|
# the last one becomes the layer's `near` so the impostor (if any) starts there.
|
|
function layer_set_lods(l: Layer, models: []Model, dists: words) -> void {
|
|
l.lods = models
|
|
l.n_lods = len(models)
|
|
l.lod_dist = words(l.n_lods); l.lod_card = words(l.n_lods); l.lod_buf = words(l.n_lods); l.n_lod = words(l.n_lods)
|
|
for k in 0 .. l.n_lods {
|
|
l.lod_dist[k] = dists[k]; l.lod_card[k] = 0; l.n_lod[k] = 0
|
|
l.lod_buf[k] = gpu_buffer_new()
|
|
let m = models[k]
|
|
for i in 0 .. len(m.prims) { scatter_attach(m.prims[i].mesh, l.lod_buf[k]) }
|
|
}
|
|
l.model = models[0]
|
|
l.near = dists[l.n_lods - 1]
|
|
if l.lvl == null { l.lvl = words(l.cap) }
|
|
}
|
|
# mark level k as the layer's crossed card (drawn with the card program and its atlas)
|
|
function layer_lod_card(l: Layer, k: int) -> void { l.lod_card[k] = 1 }
|
|
|
|
function layer_set_impostor(l: Layer, im: Impostor) -> void {
|
|
l.imp = im
|
|
scatter_attach(sc_card, l.imp_buf)
|
|
}
|
|
|
|
# ---- per frame -----------------------------------------------------------------------
|
|
# Partitions and gathers are redone only when the view changed enough to matter: the
|
|
# camera moved 1.5 m or turned about 2.5 degrees. Everything culled by the frustum keys
|
|
# off this one counter, so a turn re-gathers the streams and the grids together.
|
|
var sc_view_gen: int = 1
|
|
var sc_view_pos: words = null
|
|
var sc_view_fwd: words = null
|
|
function scatter_begin_frame() -> void {
|
|
if sc_view_pos == null { sc_view_pos = v3_new(fi(100000), F_ZERO, F_ZERO); sc_view_fwd = v3_new(F_ZERO, F_ZERO, f_neg(F_ONE)) }
|
|
if f_gt(v3_dist(sc_view_pos, cam_pos), fl(1.5)) or f_ls(v3_dot(sc_view_fwd, cam_fwd), fl(0.999)) {
|
|
sc_view_gen += 1
|
|
v3_copy(sc_view_pos, cam_pos)
|
|
v3_copy(sc_view_fwd, cam_fwd)
|
|
}
|
|
# GPU-culled layers dispatch before the first pass of the frame, so no pass is split for it
|
|
if sc_layers != null {
|
|
for i in 0 .. len(sc_layers) { if sc_layers[i].n_lods > 1 and sc_layers[i].imp != null { layer_update(sc_layers[i]) } }
|
|
}
|
|
}
|
|
|
|
# ---- the GPU-culled path ------------------------------------------------------------------
|
|
# On Vulkan (R3D_GPU_CULL=0 turns it off): a tree layer (a LOD chain of up to four levels sharing up
|
|
# to four materials, with an impostor, not streamed) is culled and split into its buckets by
|
|
# scatter_cull.comp, and its lit, prepass, impostor and shadow-LOD draws read the records that pass
|
|
# wrote - one draw per material covering every level. Nothing is partitioned or uploaded on the CPU
|
|
# when the view moves. PC camp benchmark: 2791 -> 2657 draws, 4.3 -> 4.1 s for 400 frames.
|
|
const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand
|
|
const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp)
|
|
var sc_cull_prog: int = 0
|
|
var sc_cull_tried: bool = false
|
|
var sc_ind_base: int = -1 # >= 0: layer_draw_model / layer_draw_depth draw records from here
|
|
var sc_ind_n: int = 1 # records each of those draws covers: one per level of a material
|
|
var sc_ind_stride: int = 4 # records between one material's first and the next's (3 for the shadow LOD)
|
|
|
|
function layer_gpu_eligible(l: Layer) -> bool {
|
|
if l.n_lods < 2 or l.n_lods > 4 or l.imp == null or l.streamed or l.flower or l.blade or l.count == 0 { return false }
|
|
return layer_arena_ok(l)
|
|
}
|
|
|
|
# The merged meshes. Every level of a kit tree or rock carries the same materials in the same order
|
|
# (bark then needles; the rock's one), so each material becomes ONE mesh holding all its levels, and
|
|
# one indirect draw of several records draws every level of it: record (material, level) names that
|
|
# level's index and vertex range and its bucket's instances. A conifer's lit pass goes from eight
|
|
# draws to two. Anything that does not fit - a card level, a level with other materials, other
|
|
# attributes or 32-bit indices - keeps the CPU path.
|
|
function layer_arena_ok(l: Layer) -> bool {
|
|
let n_mat = len(l.lods[0].prims)
|
|
if n_mat == 0 or n_mat > 4 { return false }
|
|
for k in 0 .. l.n_lods {
|
|
if l.lod_card[k] == 1 { return false }
|
|
let m = l.lods[k]
|
|
if len(m.prims) != n_mat { return false }
|
|
for j in 0 .. n_mat {
|
|
let pm = m.prims[j].mesh
|
|
let p0 = l.lods[0].prims[j]
|
|
if m.prims[j].diff != p0.diff or m.prims[j].verts == 0 or pm.ebo == 0 or pm.itype != GL_UNSIGNED_SHORT { return false }
|
|
if gpu_buffer_map(pm.ebo) == null { return false }
|
|
for a in 0 .. 3 {
|
|
let o = a * GPU_ATTR_W
|
|
if p0.mesh.attrs[o + 1] == 0 or pm.attrs[o + 1] != p0.mesh.attrs[o + 1] or pm.attrs[o + 3] != pm.attrs[o + 1] * 4 { return false }
|
|
if gpu_buffer_map(pm.attrs[o]) == null { return false }
|
|
}
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
function layer_arena_build(l: Layer) -> void {
|
|
let n_mat = len(l.lods[0].prims)
|
|
l.g_arena = new []Prim
|
|
l.g_first = words(16); l.g_base = words(16)
|
|
for i in 0 .. 16 { l.g_first[i] = 0; l.g_base[i] = 0 }
|
|
for j in 0 .. n_mat {
|
|
var nv = 0
|
|
var ni = 0
|
|
for k in 0 .. l.n_lods {
|
|
let pr = l.lods[k].prims[j]
|
|
l.g_first[j * 4 + k] = ni; l.g_base[j * 4 + k] = nv
|
|
nv += pr.verts; ni += pr.mesh.count
|
|
}
|
|
let p0 = l.lods[0].prims[j]
|
|
let m = gpu_mesh_new()
|
|
for a in 0 .. 3 {
|
|
let comps = p0.mesh.attrs[a * GPU_ATTR_W + 1]
|
|
let vb = bytes(nv * comps * 4 + 8)
|
|
for k in 0 .. l.n_lods {
|
|
let pr = l.lods[k].prims[j]
|
|
mem_copy(mem_off(vb, l.g_base[j * 4 + k] * comps * 4), gpu_buffer_map(pr.mesh.attrs[a * GPU_ATTR_W]), pr.verts * comps * 4)
|
|
}
|
|
gpu_mesh_vertices(m, vb, nv * comps * 4, GPU_STATIC)
|
|
gpu_mesh_attr(m, a, comps, GPU_F32, 0, 0, false)
|
|
free(vb)
|
|
}
|
|
let ib = bytes(ni * 2 + 8)
|
|
for k in 0 .. l.n_lods {
|
|
let pm = l.lods[k].prims[j].mesh
|
|
mem_copy(mem_off(ib, l.g_first[j * 4 + k] * 2), gpu_buffer_map(pm.ebo), pm.count * 2)
|
|
}
|
|
gpu_mesh_indices(m, ib, ni * 2, 2)
|
|
free(ib)
|
|
m.count = ni
|
|
gpu_mesh_done(m)
|
|
scatter_attach(m, l.g_dst)
|
|
let ap = new Prim
|
|
ap.mesh = m; ap.diff = p0.diff; ap.nrm = p0.nrm; ap.arm = p0.arm; ap.verts = nv; ap.name = p0.name
|
|
push(l.g_arena, ap)
|
|
}
|
|
l.g_model = new Model
|
|
l.g_model.prims = l.g_arena; l.g_model.height = l.lods[0].height; l.g_model.radius = l.lods[0].radius; l.g_model.ymin = l.lods[0].ymin
|
|
l.g_model_sh = new Model
|
|
var sh = l.n_lods - 1
|
|
if sh > 2 { sh = 2 }
|
|
l.g_model_sh.prims = l.g_arena; l.g_model_sh.height = l.lods[sh].height; l.g_model_sh.radius = l.lods[sh].radius; l.g_model_sh.ymin = l.lods[sh].ymin
|
|
}
|
|
|
|
function layer_gpu_prepare(l: Layer) -> bool {
|
|
if not sc_cull_tried {
|
|
sc_cull_tried = true
|
|
# Os.env is null when the variable is unset, and a compare reads through it: ask first
|
|
# on by default wherever there is compute; R3D_GPU_CULL=0 keeps the CPU partition (for comparing)
|
|
var off = false
|
|
if Os.has_env("R3D_GPU_CULL") { off = Os.env("R3D_GPU_CULL") == "0" }
|
|
if gpu_has_compute() and gpu_has_mdi() and not off {
|
|
sc_cull_prog = gpu_compute("scatter_cull", 4)
|
|
if sc_cull_prog > 0 { print("r3d: scatter: tree layers are culled on the GPU") }
|
|
}
|
|
}
|
|
if sc_cull_prog == 0 or not layer_gpu_eligible(l) { return false }
|
|
if l.g_on and l.g_n == l.count { return true }
|
|
if l.g_src == 0 {
|
|
l.g_src = gpu_buffer_new(); l.g_dst = gpu_buffer_new(); l.g_cmds = gpu_buffer_new(); l.g_counts = gpu_buffer_new()
|
|
gpu_buffer_gpu_owned(l.g_dst); gpu_buffer_gpu_owned(l.g_cmds); gpu_buffer_gpu_owned(l.g_counts)
|
|
}
|
|
let n = l.n_lods
|
|
let cap = l.count
|
|
gpu_buffer_upload(l.g_src, cap * INST_FLOATS * 4, l.inst, GPU_STATIC)
|
|
gpu_buffer_upload(l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC)
|
|
if l.g_arena == null { layer_arena_build(l) }
|
|
let n_mat = len(l.g_arena)
|
|
let rec = words(SC_RECS * 5)
|
|
for i in 0 .. SC_RECS * 5 { rec[i] = 0 }
|
|
# material j, level k: that level's range of the merged mesh, its instances from bucket k
|
|
for j in 0 .. n_mat {
|
|
for k in 0 .. n {
|
|
let r = (j * 4 + k) * 5
|
|
rec[r] = l.lods[k].prims[j].mesh.count; rec[r + 2] = l.g_first[j * 4 + k]; rec[r + 3] = l.g_base[j * 4 + k]; rec[r + 4] = k * cap
|
|
}
|
|
}
|
|
rec[16 * 5] = sc_card.count; rec[16 * 5 + 4] = n * cap
|
|
# the shadow LOD: level 2's range, from buckets 0 .. 2
|
|
if n > 2 {
|
|
for j in 0 .. n_mat {
|
|
for b in 0 .. 3 {
|
|
let r = (17 + j * 3 + b) * 5
|
|
rec[r] = l.lods[2].prims[j].mesh.count; rec[r + 2] = l.g_first[j * 4 + 2]; rec[r + 3] = l.g_base[j * 4 + 2]; rec[r + 4] = b * cap
|
|
}
|
|
}
|
|
}
|
|
gpu_buffer_upload(l.g_cmds, SC_RECS * SC_REC_W, rec, GPU_DYNAMIC)
|
|
let zeros = words(5)
|
|
for i in 0 .. 5 { zeros[i] = 0 }
|
|
gpu_buffer_upload(l.g_counts, 20, zeros, GPU_DYNAMIC)
|
|
free(rec); free(zeros)
|
|
# the card casts every instance, as on the CPU path (layer_grid_build uploads this there)
|
|
l.n_sh = l.count
|
|
gpu_buffer_upload(l.sh_buf, l.count * INST_FLOATS * 4, l.inst, GPU_STATIC)
|
|
l.g_n = l.count
|
|
l.g_on = true
|
|
return true
|
|
}
|
|
|
|
# the dispatch for the view as it stands: frustum, camera, distances, the layer's shape
|
|
function layer_gpu_cull(l: Layer) -> void {
|
|
let pr = words(36)
|
|
for i in 0 .. 36 { pr[i] = 0 }
|
|
if cam_planes != null { for i in 0 .. 16 { pr[i] = cam_planes[i] } }
|
|
pr[16] = cam_pos[0]; pr[17] = cam_pos[1]; pr[18] = cam_pos[2]; pr[19] = l.cull
|
|
for k in 0 .. l.n_lods { pr[20 + k] = l.lod_dist[k]; pr[24 + k] = len(l.lods[k].prims) }
|
|
pr[28] = l.count; pr[29] = l.count; pr[30] = l.n_lods; pr[31] = 1
|
|
# as layer_grid_gather pads a cell: the tallest instance, plus a margin
|
|
pr[32] = f_mul(l.lods[0].height, F_TWO); pr[33] = fi(4)
|
|
let bufs = words(4)
|
|
bufs[0] = l.g_src; bufs[1] = l.g_dst; bufs[2] = l.g_cmds; bufs[3] = l.g_counts
|
|
gpu_dispatch(sc_cull_prog, pr, 144, bufs, 1)
|
|
free(pr); free(bufs)
|
|
}
|
|
|
|
# Sort a static layer's instances into square cells (call once, after placement; a
|
|
# large layer that was never gridded gets a 96 m grid on its first update). The
|
|
# shadow buffer is uploaded here once — casters are never culled by the view.
|
|
function layer_grid_build(l: Layer, cell: int) -> void {
|
|
if l.count == 0 { return }
|
|
var minx = l.inst[0]; var maxx = minx; var minz = l.inst[2]; var maxz = minz
|
|
for i in 0 .. l.count {
|
|
let o = i * INST_FLOATS
|
|
minx = f_min(minx, l.inst[o]); maxx = f_max(maxx, l.inst[o])
|
|
minz = f_min(minz, l.inst[o + 2]); maxz = f_max(maxz, l.inst[o + 2])
|
|
}
|
|
l.gcell = cell; l.gx0 = minx; l.gz0 = minz
|
|
l.gnx = f_to_int(f_div(f_sub(maxx, minx), cell)) + 1
|
|
l.gnz = f_to_int(f_div(f_sub(maxz, minz), cell)) + 1
|
|
let ncell = l.gnx * l.gnz
|
|
l.gstart = words(ncell + 1)
|
|
l.gymin = words(ncell); l.gymax = words(ncell)
|
|
let cellof = words(l.count)
|
|
for c in 0 .. ncell + 1 { l.gstart[c] = 0 }
|
|
for i in 0 .. l.count {
|
|
let o = i * INST_FLOATS
|
|
let ix = f_to_int(f_div(f_sub(l.inst[o], minx), cell))
|
|
let iz = f_to_int(f_div(f_sub(l.inst[o + 2], minz), cell))
|
|
let c = iz * l.gnx + ix
|
|
cellof[i] = c
|
|
if l.gstart[c + 1] == 0 { l.gymin[c] = l.inst[o + 1]; l.gymax[c] = l.inst[o + 1] }
|
|
else { l.gymin[c] = f_min(l.gymin[c], l.inst[o + 1]); l.gymax[c] = f_max(l.gymax[c], l.inst[o + 1]) }
|
|
l.gstart[c + 1] += 1
|
|
}
|
|
for c in 0 .. ncell { l.gstart[c + 1] += l.gstart[c] }
|
|
let fill = words(ncell)
|
|
for c in 0 .. ncell { fill[c] = l.gstart[c] }
|
|
l.gsorted = words(l.count * INST_FLOATS)
|
|
for i in 0 .. l.count {
|
|
let c = cellof[i]
|
|
let q = fill[c] * INST_FLOATS
|
|
fill[c] += 1
|
|
let o = i * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { l.gsorted[q + k] = l.inst[o + k] }
|
|
}
|
|
free(cellof); free(fill)
|
|
if l.vis == null { l.vis = words(l.cap * INST_FLOATS) }
|
|
l.n_sh = l.count
|
|
gpu_buffer_upload(l.sh_buf, l.count * INST_FLOATS * 4, l.inst, GPU_STATIC)
|
|
}
|
|
|
|
# gather the instances of the cells the camera can see (and that are within cull)
|
|
function layer_grid_gather(l: Layer) -> void {
|
|
let cell = l.gcell
|
|
let half = f_mul(cell, F_HALF)
|
|
let reach = f_add(l.cull, f_mul(cell, fl(0.71)))
|
|
var n = 0
|
|
for iz in 0 .. l.gnz {
|
|
let wz = f_add(f_add(l.gz0, f_mul(fi(iz), cell)), half)
|
|
for ix in 0 .. l.gnx {
|
|
let c = iz * l.gnx + ix
|
|
let cnt = l.gstart[c + 1] - l.gstart[c]
|
|
if cnt == 0 { continue }
|
|
let wx = f_add(f_add(l.gx0, f_mul(fi(ix), cell)), half)
|
|
if l.cull != 0 {
|
|
let dx = f_sub(wx, cam_pos[0]); let dz = f_sub(wz, cam_pos[2])
|
|
if f_gt(f_sqrt(f_add(f_mul(dx, dx), f_mul(dz, dz))), reach) { continue }
|
|
}
|
|
let hy = f_mul(f_sub(l.gymax[c], l.gymin[c]), F_HALF)
|
|
let cy = f_add(l.gymin[c], hy)
|
|
# pad by the tallest instance (scale 2 of the model's height) so crowns at the frame's edge stay
|
|
let r = f_add(f_sqrt(f_add(f_mul(f_mul(half, half), F_TWO), f_mul(hy, hy))), f_add(f_mul(l.model.height, F_TWO), fi(4)))
|
|
if not cam_sphere_visible(wx, cy, wz, r) { continue }
|
|
mem_copy(mem_off(l.vis, n * INST_FLOATS * 4), mem_off(l.gsorted, l.gstart[c] * INST_FLOATS * 4), cnt * INST_FLOATS * 4)
|
|
n += cnt
|
|
}
|
|
}
|
|
l.n_vis = n
|
|
}
|
|
|
|
# Sort the gathered instances into their LOD levels (counting sort into the scratch),
|
|
# the impostor bucket last, and upload one buffer per level.
|
|
function layer_partition_lods(l: Layer, src: words, total: int) -> void {
|
|
let n = l.n_lods
|
|
let counts = words(n + 2)
|
|
for k in 0 .. n + 2 { counts[k] = 0 }
|
|
let cull2 = f_mul(l.cull, l.cull)
|
|
let open = l.lod_dist[n - 1] == 0 # the last level runs out to the cull distance
|
|
for i in 0 .. total {
|
|
let o = i * INST_FLOATS
|
|
let dx = f_sub(src[o], cam_pos[0]); let dz = f_sub(src[o + 2], cam_pos[2])
|
|
let d2 = f_add(f_mul(dx, dx), f_mul(dz, dz))
|
|
var lv = n + 1 # n + 1 = dropped
|
|
if l.cull == 0 or not f_gt(d2, cull2) {
|
|
let d = f_sqrt(d2)
|
|
lv = n # n = the impostor bucket
|
|
var k = 0
|
|
while k < n { if l.lod_dist[k] != 0 and f_ls(d, l.lod_dist[k]) { lv = k; k = n } else { k += 1 } }
|
|
if lv == n and open { lv = n - 1 }
|
|
if lv == n and l.imp == null { lv = n + 1 }
|
|
}
|
|
l.lvl[i] = lv
|
|
counts[lv] += 1
|
|
}
|
|
# prefix offsets (in instances) per bucket
|
|
let start = words(n + 2)
|
|
var acc = 0
|
|
for k in 0 .. n + 2 { start[k] = acc; acc += counts[k] }
|
|
let fill = words(n + 2)
|
|
for k in 0 .. n + 2 { fill[k] = start[k] }
|
|
let tmp = l.scratch
|
|
for i in 0 .. total {
|
|
let lv = l.lvl[i]
|
|
if lv > n { continue }
|
|
let q = fill[lv] * INST_FLOATS
|
|
fill[lv] += 1
|
|
let o = i * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
}
|
|
for k in 0 .. n {
|
|
l.n_lod[k] = counts[k]
|
|
if counts[k] > 0 {
|
|
gpu_buffer_upload(l.lod_buf[k], counts[k] * INST_FLOATS * 4, mem_off(tmp, start[k] * INST_FLOATS * 4), GPU_DYNAMIC)
|
|
}
|
|
}
|
|
l.n_near = counts[0]
|
|
l.n_far = counts[n]
|
|
if sc_dbg_lod and total > 1000 { print(`lod partition: total {total} dropped {counts[n + 1]} far {counts[n]} l0 {counts[0]} l1 {counts[1]} l2 {counts[2]} l3 {counts[3]} dist0 {f_fx(l.lod_dist[0])} dist3 {f_fx(l.lod_dist[n - 1])} cull {f_fx(l.cull)} cam {f_fx(cam_pos[0])} {f_fx(cam_pos[2])} first {f_fx(src[0])} {f_fx(src[2])}`) }
|
|
if l.n_far > 0 {
|
|
gpu_buffer_upload(l.imp_buf, l.n_far * INST_FLOATS * 4, mem_off(tmp, start[n] * INST_FLOATS * 4), GPU_DYNAMIC)
|
|
}
|
|
# casters: the whole (gathered) set from the shadow buffer, unless the impostor casts
|
|
if l.gcell == 0 {
|
|
l.n_sh = total
|
|
if total > 0 { gpu_buffer_upload(l.sh_buf, total * INST_FLOATS * 4, src, GPU_DYNAMIC) }
|
|
}
|
|
free(counts); free(start); free(fill)
|
|
}
|
|
|
|
# split the instances by distance to the camera (only when the view changed)
|
|
var sc_freeze: bool = false # a secondary pass (reflection) reuses the partition
|
|
function layer_update(l: Layer) -> void {
|
|
if sc_freeze { return }
|
|
if l.view_gen == sc_view_gen { return }
|
|
l.view_gen = sc_view_gen
|
|
if layer_gpu_prepare(l) { layer_gpu_cull(l); return }
|
|
let t_lu = gl_now_us()
|
|
let n_lu = l.count
|
|
# A streamed layer's instances were already gathered per visible chunk: no split, no
|
|
# per-instance loop — one upload, and the same buffer casts its shadows.
|
|
if l.streamed and l.imp == null and l.near == 0 and l.n_lods <= 1 {
|
|
l.n_near = l.count; l.n_far = 0; l.n_sh = l.count
|
|
gpu_buffer_upload(l.buf, l.count * INST_FLOATS * 4, l.inst, GPU_DYNAMIC)
|
|
prof_layer_add(gl_now_us() - t_lu, l.count * INST_FLOATS * 4)
|
|
return
|
|
}
|
|
if l.gcell == 0 and not l.streamed and l.count > 2000 { layer_grid_build(l, fi(96)) }
|
|
var src = l.inst
|
|
var total = l.count
|
|
if l.gcell != 0 { layer_grid_gather(l); src = l.vis; total = l.n_vis }
|
|
let near2 = f_mul(l.near, l.near)
|
|
let cull2 = f_mul(l.cull, l.cull)
|
|
var nn = 0
|
|
var nf = 0
|
|
let far_off = l.cap * INST_FLOATS # far instances fill the scratch from its end backwards
|
|
let tmp = l.scratch
|
|
if l.n_lods > 1 {
|
|
layer_partition_lods(l, src, total)
|
|
prof_layer_add(gl_now_us() - t_lu, total * INST_FLOATS * 4)
|
|
return
|
|
}
|
|
var i = 0
|
|
while i < total {
|
|
let o = i * INST_FLOATS
|
|
let dx = f_sub(src[o], cam_pos[0])
|
|
let dz = f_sub(src[o + 2], cam_pos[2])
|
|
let d2 = f_add(f_mul(dx, dx), f_mul(dz, dz))
|
|
if l.cull != 0 and f_gt(d2, cull2) { i += 1; continue }
|
|
if l.near == 0 or f_ls(d2, near2) {
|
|
let q = nn * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
nn += 1
|
|
} else if l.imp != null {
|
|
nf += 1
|
|
let q = far_off - nf * INST_FLOATS
|
|
for k in 0 .. INST_FLOATS { tmp[q + k] = src[o + k] }
|
|
}
|
|
i += 1
|
|
}
|
|
l.n_near = nn
|
|
l.n_far = nf
|
|
if l.imp != null and nn > 0 and sc_debug_dump { print(`near full-mesh instances: {nn} (first at {f_fx(tmp[0])} {f_fx(tmp[1])} {f_fx(tmp[2])})`) }
|
|
if sc_debug_dump and l.imp != null {
|
|
print(`layer: near {nn} far {nf}`)
|
|
for k in 0 .. nn { let q = k * INST_FLOATS; print(` near {f_fx(tmp[q])} {f_fx(tmp[q + 1])} {f_fx(tmp[q + 2])} s {f_fx(tmp[q + 3])}`) }
|
|
}
|
|
# Every instance, unculled and unsplit, for the shadow pass. What the camera draws is
|
|
# allowed to change with distance; what casts must not, or shadows blink in and out as
|
|
# you walk. This is the whole set, drawn one way, into every cascade.
|
|
if l.gcell == 0 {
|
|
l.n_sh = l.count
|
|
if l.count > 0 {
|
|
gpu_buffer_upload(l.sh_buf, l.count * INST_FLOATS * 4, l.inst, GPU_DYNAMIC)
|
|
}
|
|
}
|
|
gpu_buffer_upload(l.buf, nn * INST_FLOATS * 4, tmp, GPU_DYNAMIC)
|
|
if nf > 0 {
|
|
gpu_buffer_upload(l.imp_buf, nf * INST_FLOATS * 4, mem_off(tmp, (far_off - nf * INST_FLOATS) * 4), GPU_DYNAMIC)
|
|
}
|
|
prof_layer_add(gl_now_us() - t_lu, (nn + nf + l.n_sh) * INST_FLOATS * 4)
|
|
}
|
|
|
|
function layer_program(l: Layer, shadow: bool, card: bool) -> int {
|
|
if card {
|
|
if shadow { return sc_prog_card_shadow }
|
|
if l.cheap { return sc_prog_card_cheap }
|
|
return sc_prog_card
|
|
}
|
|
if shadow {
|
|
if l.foliage and not l.blade and not l.flower { return sc_prog_shadow_fol }
|
|
if l.wind != 0 { return sc_prog_shadow_wind }
|
|
return sc_prog_shadow
|
|
}
|
|
if l.blade { return sc_prog_blade }
|
|
if l.flower { return sc_prog_flower }
|
|
if l.foliage {
|
|
if sc_prepass and sc_prog_fol_eq != 0 { return sc_prog_fol_eq }
|
|
return sc_prog_fol
|
|
}
|
|
if l.wind != 0 { return sc_prog_wind }
|
|
return sc_prog
|
|
}
|
|
|
|
var sc_dbg_blade: int = 0
|
|
# R3D_LODDBG=1 tints each LOD level (red, green, blue, yellow) and impostors magenta
|
|
var sc_dbg_level: int = -1
|
|
var sc_dbg_lod: bool = false
|
|
var sc_dbg_tint: words = null
|
|
# Can level k's casters (its instances lie between the previous level's distance and its own) put a
|
|
# shadow on anything the cascade being rendered covers? A receiver in that slice of view depth
|
|
# [near, far] stands between near - dy and far * K metres away on the ground: dy is the camera's height
|
|
# over the ground, K how far the frustum's corners reach past its depth. A prop's shadow falls at most
|
|
# about six times its height past it (the sun near ten degrees). A level outside that range cannot
|
|
# touch a pixel of the cascade, so leaving it out changes no shadow - unlike the old per-class skips
|
|
# at fixed distances, which dropped casters that did cast (see scatter_draw_casters). The flowers'
|
|
# mesh levels (6 - 30 m) stop being drawn into the three outer cascades. R3D_CAST_ALL=1 draws every
|
|
# level into every cascade, for comparing.
|
|
var sc_cast_all: int = -1
|
|
var sc_cast_gen: int = -1
|
|
var sc_cast_dy: int = 0
|
|
var sc_cast_k: int = 0
|
|
function layer_level_casts_here(l: Layer, k: int) -> bool {
|
|
if sc_cast_all < 0 { sc_cast_all = 0; if Os.has_env("R3D_CAST_ALL") { sc_cast_all = 1 } }
|
|
if sc_cast_all == 1 or sh_split == null or l.lod_dist == null { return true }
|
|
if sc_cast_gen != sc_view_gen {
|
|
sc_cast_gen = sc_view_gen
|
|
sc_cast_dy = f_add(f_abs(f_sub(cam_pos[1], terrain_height(cam_pos[0], cam_pos[2]))), fi(5))
|
|
let half = f_mul(cam_fov, F_HALF)
|
|
let t = f_div(f_sin(half), f_cos(half))
|
|
let ta = f_mul(t, cam_aspect)
|
|
sc_cast_k = f_mul(f_sqrt(f_add(F_ONE, f_add(f_mul(t, t), f_mul(ta, ta)))), fl(1.1))
|
|
}
|
|
let c = sh_cascade
|
|
var near = cam_near
|
|
if c > 0 { near = sh_split[c - 1] }
|
|
let far = sh_split[c]
|
|
var dmin = F_ZERO
|
|
if k > 0 { dmin = l.lod_dist[k - 1] }
|
|
let dmax = l.lod_dist[k] # 0: open, out to the layer's cull distance
|
|
let reach = f_max(f_min(f_mul(l.lods[k].height, fi(6)), fi(40)), fi(8))
|
|
if dmax != 0 and f_ls(f_add(f_add(dmax, reach), sc_cast_dy), near) { return false }
|
|
if f_gt(f_sub(dmin, reach), f_mul(far, sc_cast_k)) { return false }
|
|
return true
|
|
}
|
|
|
|
# `full` casts the layer's entire instance list out of sh_buf instead of the near
|
|
# partition out of l.buf. A layer with no impostor (the tree crowns' branch cards) has
|
|
# no cheap stand-in to cast from, so without this its shadow simply began at the near
|
|
# distance — which is the crown shadow that appeared as you walked up to a tree.
|
|
function layer_draw_near(l: Layer, shadow: bool, light_vp: words, full: bool) -> void {
|
|
if l.n_lods > 1 and l.g_on and not shadow {
|
|
# one draw per material covering all of its levels (the merged meshes)
|
|
sc_ind_base = 0; sc_ind_n = l.n_lods
|
|
layer_draw_model(l, l.g_model, l.g_dst, 1, false, shadow, light_vp)
|
|
sc_ind_base = -1; sc_ind_n = 1
|
|
return
|
|
}
|
|
if l.n_lods > 1 {
|
|
# a LOD chain: every level from its own bucket (casters are what is drawn), and a caster level
|
|
# only into the cascades it can put a shadow in
|
|
for k in 0 .. l.n_lods {
|
|
if shadow and not layer_level_casts_here(l, k) { continue }
|
|
sc_dbg_level = k; layer_draw_model(l, l.lods[k], l.lod_buf[k], l.n_lod[k], l.lod_card[k] == 1, shadow, light_vp)
|
|
}
|
|
sc_dbg_level = -1
|
|
return
|
|
}
|
|
var vb = l.buf
|
|
var cnt = l.n_near
|
|
if full and not l.streamed { vb = l.sh_buf; cnt = l.n_sh }
|
|
layer_draw_model(l, l.model, vb, cnt, l.card, shadow, light_vp)
|
|
}
|
|
|
|
# draw `cnt` instances of `model` out of instance buffer `vb`, as a mesh or as the layer's card
|
|
function layer_draw_model(l: Layer, model: Model, vb: int, cnt: int, card: bool, shadow: bool, light_vp: words) -> void {
|
|
if cnt == 0 { return }
|
|
let p = layer_program(l, shadow, card)
|
|
gpu_use_program(p)
|
|
# over the prepass: only the fragment the prepass kept, at exactly its depth (a texel
|
|
# it cut would otherwise pass LEQUAL over the terrain behind and draw the quad solid)
|
|
if p == sc_prog_fol_eq { gpu_depth_func(GL_EQUAL) }
|
|
var ground = F_ZERO
|
|
if l.grounded { ground = F_ONE }
|
|
u_f(gpu_uniform(p, "u_ground"), ground)
|
|
if l.grounded { terrain_bind_height(p) }
|
|
u_f(gpu_uniform(p, "u_time"), r3d_time)
|
|
u_f(gpu_uniform(p, "u_wind"), l.wind)
|
|
var mh = F_ZERO
|
|
if not card and l.foliage { mh = model.height }
|
|
u_f(gpu_uniform(p, "u_model_h"), mh)
|
|
if not card and l.foliage and not shadow and sc_a2c { gpu_alpha_to_coverage(true) }
|
|
if card {
|
|
u_f(gpu_uniform(p, "u_card_w"), f_mul(l.atlas.radius, F_TWO)); u_f(gpu_uniform(p, "u_card_h"), l.atlas.height)
|
|
r3d_bind_2d(p, "u_diff", 0, l.atlas.albedo)
|
|
r3d_bind_2d(p, "u_nrm", 1, l.atlas.normal)
|
|
gpu_alpha_to_coverage(true)
|
|
}
|
|
if shadow { u_mat4(gpu_uniform(p, "u_light_vp"), light_vp) }
|
|
else {
|
|
u_mat4(gpu_uniform(p, "u_view"), cam_view)
|
|
u_mat4(gpu_uniform(p, "u_proj"), cam_proj)
|
|
u_v3(gpu_uniform(p, "u_tint"), l.tint)
|
|
if sc_dbg_lod and sc_dbg_level >= 0 {
|
|
if sc_dbg_tint == null { sc_dbg_tint = v3_new(F_ONE, F_ONE, F_ONE) }
|
|
let k = sc_dbg_level
|
|
var r = F_ZERO; var g = F_ZERO; var b = F_ZERO
|
|
if k == 0 { r = fi(3) } else if k == 1 { g = fi(3) } else if k == 2 { b = fi(3) } else { r = fi(3); g = fi(3) }
|
|
v3_set(sc_dbg_tint, r, g, b)
|
|
u_v3(gpu_uniform(p, "u_tint"), sc_dbg_tint)
|
|
}
|
|
u_f(gpu_uniform(p, "u_rough_scale"), l.rough)
|
|
if l.blade { u_v3(gpu_uniform(p, "u_blade_base"), sc_blade_base); u_v3(gpu_uniform(p, "u_blade_tip"), sc_blade_tip) }
|
|
u_f(gpu_uniform(p, "u_cull"), l.cull)
|
|
sky_bind_lighting(p)
|
|
shadow_bind(p)
|
|
fog_bind(p)
|
|
if l.foliage { u_f(gpu_uniform(p, "u_spec_scale"), fl(0.05)) }
|
|
}
|
|
gpu_cull(false)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
scatter_attach(pr.mesh, vb)
|
|
if not card {
|
|
r3d_bind_2d(p, "u_diff", 0, pr.diff)
|
|
if not shadow { r3d_bind_2d(p, "u_nrm", 1, pr.nrm); r3d_bind_2d(p, "u_arm", 2, pr.arm) }
|
|
}
|
|
if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i * sc_ind_stride) * SC_REC_W, sc_ind_n, 0, 0) }
|
|
else { mesh_draw_instanced(pr.mesh, cnt) }
|
|
}
|
|
gpu_alpha_to_coverage(false)
|
|
if p == sc_prog_fol_eq { gpu_depth_func(GL_LESS) }
|
|
}
|
|
|
|
function layer_draw_far(l: Layer, shadow: bool, light_vp: words) -> void {
|
|
if l.imp == null or (l.n_far == 0 and not l.g_on) { return }
|
|
var p = sc_imp_prog
|
|
if shadow { p = sc_imp_prog_shadow }
|
|
gpu_use_program(p)
|
|
let im = l.imp
|
|
u_f(gpu_uniform(p, "u_radius"), im.radius)
|
|
u_f(gpu_uniform(p, "u_height"), im.height)
|
|
u_f(gpu_uniform(p, "u_tiles"), fi(im.tiles))
|
|
r3d_bind_2d(p, "u_atlas_albedo", 0, im.albedo)
|
|
if shadow {
|
|
u_mat4(gpu_uniform(p, "u_light_vp"), light_vp)
|
|
u_v3(gpu_uniform(p, "u_face_dir"), sun_dir)
|
|
if r3d_debug_shadow and not sc_printed { sc_printed = true; print(`imp shadow prog {p} face_dir loc {gpu_uniform(p, "u_face_dir")} sun {f_fx(sun_dir[0])} {f_fx(sun_dir[1])} {f_fx(sun_dir[2])} cam {f_fx(cam_pos[0])} {f_fx(cam_pos[1])} {f_fx(cam_pos[2])} n_far {l.n_far}`) }
|
|
u_v3(gpu_uniform(p, "u_cam_pos"), cam_pos)
|
|
} else {
|
|
u_mat4(gpu_uniform(p, "u_view"), cam_view)
|
|
u_mat4(gpu_uniform(p, "u_proj"), cam_proj)
|
|
u_v3(gpu_uniform(p, "u_tint"), l.tint)
|
|
if sc_dbg_lod { if sc_dbg_tint == null { sc_dbg_tint = v3_new(F_ONE, F_ONE, F_ONE) }; v3_set(sc_dbg_tint, fi(3), F_ZERO, fi(3)); u_v3(gpu_uniform(p, "u_tint"), sc_dbg_tint) }
|
|
r3d_bind_2d(p, "u_atlas_normal", 1, im.normal)
|
|
sky_bind_lighting(p)
|
|
shadow_bind(p)
|
|
fog_bind(p)
|
|
if l.foliage { u_f(gpu_uniform(p, "u_spec_scale"), fl(0.05)) }
|
|
}
|
|
gpu_cull(false)
|
|
if not shadow and sc_a2c { gpu_alpha_to_coverage(true) }
|
|
if l.g_on {
|
|
scatter_attach(sc_card, l.g_dst)
|
|
gpu_draw_mesh_indirect(sc_card, l.g_cmds, 16 * SC_REC_W, 1, 0, 0)
|
|
} else {
|
|
scatter_attach(sc_card, l.imp_buf)
|
|
mesh_draw_instanced(sc_card, l.n_far)
|
|
}
|
|
gpu_alpha_to_coverage(false)
|
|
}
|
|
|
|
# Cast from the impostor card, always, for every instance in the layer. The lit pass may
|
|
# swap a scanned mesh in up close; the shadow must not, or a tree's shadow changes shape
|
|
# as you approach it. The card is also far the cheaper of the two, which is what pays for
|
|
# casting the whole set into all five cascades.
|
|
function layer_draw_shadow(l: Layer, light_vp: words) -> void {
|
|
if l.n_sh == 0 or l.imp == null { return }
|
|
let p = sc_imp_prog_shadow
|
|
gpu_use_program(p)
|
|
let im = l.imp
|
|
u_f(gpu_uniform(p, "u_radius"), im.radius)
|
|
u_f(gpu_uniform(p, "u_height"), im.height)
|
|
u_f(gpu_uniform(p, "u_tiles"), fi(im.tiles))
|
|
r3d_bind_2d(p, "u_atlas_albedo", 0, im.albedo)
|
|
u_mat4(gpu_uniform(p, "u_light_vp"), light_vp)
|
|
u_v3(gpu_uniform(p, "u_face_dir"), sun_dir)
|
|
u_v3(gpu_uniform(p, "u_cam_pos"), cam_pos)
|
|
gpu_cull(false)
|
|
scatter_attach(sc_card, l.sh_buf)
|
|
mesh_draw_instanced(sc_card, l.n_sh)
|
|
}
|
|
|
|
var sc_skip_blade: bool = false
|
|
var sc_skip_flower: bool = false
|
|
var sc_skip_card: bool = false
|
|
|
|
# ---- the foliage depth prepass -----------------------------------------------------------
|
|
# Exactly the instances and positions layer_draw_model will light (the same LOD buckets,
|
|
# the same vertex shader, the same ground and wind), into depth only.
|
|
function layer_draw_depth(l: Layer, model: Model, vb: int, cnt: int) -> void {
|
|
if cnt == 0 or model == null or vb == 0 { return }
|
|
let p = sc_prog_fol_depth
|
|
gpu_use_program(p)
|
|
var ground = F_ZERO
|
|
if l.grounded { ground = F_ONE }
|
|
u_f(gpu_uniform(p, "u_ground"), ground)
|
|
if l.grounded { terrain_bind_height(p) }
|
|
u_f(gpu_uniform(p, "u_time"), r3d_time)
|
|
u_f(gpu_uniform(p, "u_wind"), l.wind)
|
|
u_f(gpu_uniform(p, "u_model_h"), model.height)
|
|
u_mat4(gpu_uniform(p, "u_view"), cam_view)
|
|
u_mat4(gpu_uniform(p, "u_proj"), cam_proj)
|
|
u_f(gpu_uniform(p, "u_clip_y"), r3d_clip_y)
|
|
gpu_cull(false)
|
|
for i in 0 .. len(model.prims) {
|
|
let pr = model.prims[i]
|
|
scatter_attach(pr.mesh, vb)
|
|
r3d_bind_2d(p, "u_diff", 0, pr.diff)
|
|
if sc_ind_base >= 0 { gpu_draw_mesh_indirect(pr.mesh, l.g_cmds, (sc_ind_base + i * sc_ind_stride) * SC_REC_W, sc_ind_n, 0, 0) }
|
|
else { mesh_draw_instanced(pr.mesh, cnt) }
|
|
}
|
|
}
|
|
|
|
# every foliage mesh draw the lit pass will make with sc_prog_fol_eq: not blades, not
|
|
# flowers, not card levels (those keep their own alpha and draw as before)
|
|
function scatter_draw_depth() -> void {
|
|
if sc_prog_fol_depth == 0 { return }
|
|
for i in 0 .. len(sc_layers) {
|
|
let l = sc_layers[i]
|
|
if not l.foliage or l.blade or l.flower { continue }
|
|
if r3d_no_trees and l.imp != null { continue }
|
|
layer_update(l)
|
|
if l.n_lods > 1 and l.g_on {
|
|
sc_ind_base = 0; sc_ind_n = l.n_lods
|
|
layer_draw_depth(l, l.g_model, l.g_dst, 1)
|
|
sc_ind_base = -1; sc_ind_n = 1
|
|
} else if l.n_lods > 1 {
|
|
# (a tree layer is flagged `card` for its distant level; its mesh levels still count)
|
|
for k in 0 .. l.n_lods { if l.lod_card[k] != 1 { layer_draw_depth(l, l.lods[k], l.lod_buf[k], l.n_lod[k]) } }
|
|
} else if not l.card {
|
|
layer_draw_depth(l, l.model, l.buf, l.n_near)
|
|
}
|
|
}
|
|
gpu_cull(true)
|
|
}
|
|
function scatter_draw() -> void {
|
|
for i in 0 .. len(sc_layers) {
|
|
let l = sc_layers[i]
|
|
if sc_skip_blade and l.blade { continue }
|
|
if sc_skip_flower and l.flower { continue }
|
|
if sc_skip_card and l.card { continue }
|
|
if r3d_no_trees and l.imp != null and not l.card { continue }
|
|
layer_update(l)
|
|
layer_draw_near(l, false, null, false)
|
|
layer_draw_far(l, false, null)
|
|
}
|
|
gpu_cull(true)
|
|
}
|
|
# Nothing here is keyed off the cascade. Every skip that used to be — ground cover past
|
|
# the 250 m cascade, blades past the nearest, the scanned mesh past the second — made a
|
|
# whole class of caster vanish at a fixed distance, which is exactly the popping. A layer
|
|
# with an impostor now casts its entire instance list from the card in every cascade;
|
|
# only layers that have no impostor at all fall back to the mesh.
|
|
function scatter_draw_casters(light_vp: words) -> void {
|
|
for i in 0 .. len(sc_layers) {
|
|
let l = sc_layers[i]
|
|
if sc_skip_blade and l.blade { continue }
|
|
if sc_skip_card and l.card { continue }
|
|
if r3d_no_trees and l.imp != null and not l.card { continue }
|
|
layer_update(l)
|
|
if l.imp != null {
|
|
layer_draw_shadow(l, light_vp)
|
|
# Shadow LOD (the practice in every production engine: a caster uses a low mesh LOD,
|
|
# the billboard only far away). The card alone is a side-view silhouette and a
|
|
# crown of drooping needle cards is mostly slivers from the side, so the sun, which
|
|
# sees the crown from above, cast a trunk line with a few blobs. The near levels
|
|
# now also cast their LOD2 mesh, alpha-tested, on top of the card.
|
|
if l.n_lods > 2 and l.g_on {
|
|
sc_ind_base = 17; sc_ind_n = 3; sc_ind_stride = 3
|
|
layer_draw_model(l, l.g_model_sh, l.g_dst, 1, false, true, light_vp)
|
|
sc_ind_base = -1; sc_ind_n = 1; sc_ind_stride = 4
|
|
} else if l.n_lods > 2 {
|
|
for k in 0 .. 3 { layer_draw_model(l, l.lods[2], l.lod_buf[k], l.n_lod[k], false, true, light_vp) }
|
|
}
|
|
}
|
|
else { layer_draw_near(l, true, light_vp, true) }
|
|
}
|
|
}
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# The distant-grass carpet: the clump cards rendered straight down into one tiling
|
|
# tile, so the ground beyond the blade rings carries the same clumps, colours and
|
|
# gaps as the near cover instead of a lawn scan (the far-field trick the big open
|
|
# worlds use: geometry up close, an authored ground texture that matches it beyond).
|
|
# Returns an RGBA8 texture (alpha = coverage), repeat-wrapped and mipmapped.
|
|
var cb_state: int = 12345
|
|
function cb_rnd() -> int {
|
|
cb_state = (cb_state * 1103515245 + 12345) & 0x7FFFFFFF
|
|
return fr((cb_state >> 8) & 0xFFFF, 65536)
|
|
}
|
|
function carpet_bake(layers: []Layer, count: int, tile: int, res: int) -> int {
|
|
let tex = tex_target(res, res, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
|
|
let fbo = gpu_fb_new()
|
|
gpu_fb_bind(fbo)
|
|
gpu_fb_color(0, tex)
|
|
let rb = gpu_rb_new()
|
|
gpu_rb_storage(rb, GL_DEPTH_COMPONENT24, res, res, 0)
|
|
gpu_fb_depth_rb(rb)
|
|
gpu_fb_draw_buffers(1)
|
|
gpu_viewport(0, 0, res, res)
|
|
gpu_clear_color(0.0, 0.0, 0.0, 0.0)
|
|
gpu_clear(GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
|
gpu_depth_test(true)
|
|
gpu_depth_func(GL_LESS)
|
|
gpu_cull(false)
|
|
gpu_blend(false)
|
|
# the clumps, and eight wrapped copies so the tile's edges continue
|
|
let half = f_mul(tile, F_HALF)
|
|
let n9 = count * 9
|
|
let inst = gl_floats(n9 * INST_FLOATS)
|
|
cb_state = 977
|
|
var k = 0
|
|
for i in 0 .. count {
|
|
let x = f_sub(f_mul(cb_rnd(), tile), half)
|
|
let z = f_sub(f_mul(cb_rnd(), tile), half)
|
|
let sc = f_add(fl(1.5), cb_rnd())
|
|
let yaw = f_mul(cb_rnd(), f_mul(F_TWO, F_PI))
|
|
let sd = cb_rnd()
|
|
for oz in 0 .. 3 {
|
|
for ox in 0 .. 3 {
|
|
let px = f_add(x, f_mul(fi(ox - 1), tile))
|
|
let pz = f_add(z, f_mul(fi(oz - 1), tile))
|
|
gl_put_bits(inst, k, px); gl_put_bits(inst, k + 1, F_ZERO); gl_put_bits(inst, k + 2, pz); gl_put_bits(inst, k + 3, sc)
|
|
gl_put_bits(inst, k + 4, f_sin(yaw)); gl_put_bits(inst, k + 5, f_cos(yaw)); gl_put_bits(inst, k + 6, sd); gl_put_bits(inst, k + 7, F_ZERO)
|
|
k += INST_FLOATS
|
|
}
|
|
}
|
|
}
|
|
let buf = gpu_buffer_new()
|
|
gpu_buffer_upload(buf, gl_bytes_of(n9 * INST_FLOATS), inst, GPU_STATIC)
|
|
free(inst)
|
|
# straight down: the window is exactly one tile
|
|
let view = m4_new(); let proj = m4_new()
|
|
let eye = v3_new(F_ZERO, fi(6), F_ZERO); let at = v3_new(F_ZERO, F_ZERO, F_ZERO); let up = v3_new(F_ZERO, F_ZERO, f_neg1())
|
|
m4_look_at(view, eye, at, up)
|
|
m4_ortho(proj, f_neg(half), half, f_neg(half), half, fl(0.1), fi(12))
|
|
let bake = sc_bake_card_prog
|
|
gpu_use_program(bake)
|
|
u_mat4(gpu_uniform(bake, "u_view"), view)
|
|
u_mat4(gpu_uniform(bake, "u_proj"), proj)
|
|
u_f(gpu_uniform(bake, "u_wind"), F_ZERO)
|
|
u_f(gpu_uniform(bake, "u_time"), F_ZERO)
|
|
for li in 0 .. len(layers) {
|
|
let l = layers[li]
|
|
if l.atlas == null { continue }
|
|
u_f(gpu_uniform(bake, "u_card_w"), f_mul(l.atlas.radius, F_TWO)); u_f(gpu_uniform(bake, "u_card_h"), l.atlas.height)
|
|
r3d_bind_2d(bake, "u_diff", 0, l.atlas.albedo)
|
|
r3d_bind_2d(bake, "u_arm", 2, l.atlas.normal)
|
|
for i in 0 .. len(l.model.prims) {
|
|
let pr = l.model.prims[i]
|
|
scatter_attach(pr.mesh, buf)
|
|
mesh_draw_instanced(pr.mesh, n9)
|
|
}
|
|
}
|
|
free(view); free(proj); free(eye); free(at); free(up)
|
|
gpu_fb_bind(0)
|
|
gpu_fb_free(fbo)
|
|
gpu_rb_free(rb)
|
|
gpu_buffer_free(buf)
|
|
gpu_tex_bind(GPU_TEX2D, tex)
|
|
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
|
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_REPEAT)
|
|
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
|
gpu_tex_paramf(GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, tex_anisotropy)
|
|
gpu_tex_mips(GPU_TEX2D)
|
|
gpu_check("carpet bake")
|
|
return tex
|
|
}
|
|
|
|
# ---- another map at run time ------------------------------------------------------------
|
|
# Every scattered layer released - its instance arrays and GL buffers, a card layer's own
|
|
# crossed-card mesh and baked atlas - and every stream feeding them. The models a layer drew
|
|
# belong to whoever loaded them and are kept, and so are the programs. A game rebuilding its
|
|
# world calls this, then places the new map's layers exactly as it did at boot.
|
|
function scatter_clear_all() -> void {
|
|
stream_clear_all()
|
|
if sc_layers == null { return }
|
|
for i in 0 .. len(sc_layers) {
|
|
let l = sc_layers[i]
|
|
if l.buf != 0 { gpu_buffer_free(l.buf) }
|
|
if l.imp_buf != 0 { gpu_buffer_free(l.imp_buf) }
|
|
if l.sh_buf != 0 { gpu_buffer_free(l.sh_buf) }
|
|
if l.lod_buf != null { for k in 0 .. l.n_lods { gpu_buffer_free(l.lod_buf[k]) } }
|
|
if l.inst != null { free(l.inst) }
|
|
if l.scratch != null { free(l.scratch) }
|
|
if l.tint != null { free(l.tint) }
|
|
if l.last_cam != null { free(l.last_cam) }
|
|
if l.gstart != null { free(l.gstart) }
|
|
if l.gsorted != null { free(l.gsorted) }
|
|
if l.gymin != null { free(l.gymin) }
|
|
if l.gymax != null { free(l.gymax) }
|
|
if l.vis != null { free(l.vis) }
|
|
if l.lod_dist != null { free(l.lod_dist); free(l.lod_card); free(l.lod_buf); free(l.n_lod) }
|
|
if l.lvl != null { free(l.lvl) }
|
|
if l.g_arena != null { for k in 0 .. len(l.g_arena) { mesh_free(l.g_arena[k].mesh) } }
|
|
# layer_cards built this layer's crossed card and its atlas itself
|
|
if l.card and l.n_lods == 0 and l.model != null { for k in 0 .. len(l.model.prims) { mesh_free(l.model.prims[k].mesh) } }
|
|
if l.card and l.atlas != null {
|
|
gpu_tex_free(l.atlas.albedo)
|
|
gpu_tex_free(l.atlas.normal)
|
|
}
|
|
}
|
|
sc_layers = new []Layer
|
|
sc_view_gen += 1
|
|
}
|