r3d/runtime: allocs, keeps and births at 0 on this side

render3d: shadow_fit, water_reflection_pass, layer_partition_lods and the
GPU cull's scratch are made with the state; v3_dist is scalar; the pushes
into lists sized at start-up, the caps probe, the table growth, the loads
and the constructors declared with their bounds (one statement a line);
the renderer's name made once with the device; the two error messages
given back; the dead lupine models removed.

runtime: a component's text is held interned in its value cell (one copy
per distinct text), so the getter's own text goes with its frame instead
of being kept by ludic.ui's model - 80 of the 83 keeps.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-28 20:10:59 +03:00
parent 5be2422c84
commit f993c4a36a
18 changed files with 105 additions and 177 deletions

View file

@ -54,7 +54,7 @@ Geometry, terrain and world
29. GPU instancing with divisor attributes and per-instance transforms — `scatter.ludic`, `shaders/model.vert`
30. Baked impostors (16-view atlas with coverage and normals, hull-normal blending, LOD switch) — `scatter.ludic` `impostor_bake`, `shaders/impostor.*`
31. Baked branch-card trees: wedge slices of a scanned crown (sectors × height bands) on radial cards over the real trunk mesh — `scatter.ludic` `impostor_bake_wedges`, `model_branch_cards`
32. Card atlases baked from scanned grass clumps and a dense procedural lupine, instanced as crossed cards — `scatter.ludic` `layer_cards`, `model_lupine_dense`
32. Card atlases baked from scanned grass clumps and a dense procedural lupine, instanced as crossed cards — `scatter.ludic` `layer_cards`
33. Procedural grass blades with wind, dryness and base occlusion; distant blades widen, lie flat to the ground normal and sink into the carpet at the ring's edge — `scatter.ludic` `model_blade`, `shaders/model.vert`, `shaders/model.frag` `BLADE`
34. Wind animation (height-weighted sway, per-instance phase) — `shaders/model.vert` `WIND`
35. Chunk-streamed ground cover with distance bands (blades in four rings, densest underfoot), deterministic per-chunk generation, nearest-first amortised over frames with an instance budget — `stream.ludic`

View file

@ -119,6 +119,7 @@ function actor_remove(render3d_st: mut Render3dState, a: Actor) -> void {
function actor_keep(render3d_st: mut Render3dState, a: Actor) -> void {
if a == null { return }
a.row = len(render3d_st.ac_actors)
@alloc_ok("within the stage's room made at start-up (@max 2048)")
push(render3d_st.ac_actors, a)
}
# colour one named part of the model (a material name from the file)
@ -470,6 +471,7 @@ function actor_release(render3d_st: mut Render3dState, a: Actor) -> void {
a.model = null; a.skin = null; a.ocol = null
a.id = 0
if render3d_st.ac_spare == null { return } # made in actor_init, with its room
@alloc_ok("within the spares' room made at start-up (@max 2048)")
push(render3d_st.ac_spare, a)
}

View file

@ -119,6 +119,7 @@ function ds_kind_name(k: int) -> pointer {
return "indirect"
}
# a per-frame mean with one decimal
@alloc_ok("the draw-count report: printed once, at the end of a run that asked for it")
function ds_per(render3d_st: Render3dState, v: long) -> string {
let x = (v * 10) / render3d_st.ds_frames
return `{string(x / 10)}.{string(x % 10)}`

View file

@ -166,6 +166,7 @@ export state Render3dState {
gvk_family: int = -1
gvk_mp: bytes = null # VkPhysicalDeviceMemoryProperties
gvk_device_name: string = ""
gvk_renderer_name: string = "" # "<device> (Vulkan)", made once with the device
gvk_why: string = "" # why the device did not come up, for the fallback notice
gvk_max_aniso: int = 0x3F800000 # the device's anisotropy limit as float bits; 1.0 when it has none
gvk_want_surface: bool = false # set before gvk_init when the renderer will present to a window
@ -692,7 +693,23 @@ export state Render3dState {
stream_debug_n: int = 0
stream_scratch: floats = null
gvk_pipe_bufs: words = null # a pipeline's vertex buffers, while it is made
gsl_up: floats = floats(3) # the camera's up for DLSS's constants
gsl_up: floats = floats(3)
sh_fit_cview: floats = floats(3) # shadow_fit's scratch
sh_fit_corners: floats = floats(24)
sh_fit_center: floats = floats(3)
sh_fit_eye: floats = floats(3)
sh_fit_up: floats = floats(3)
sh_fit_out: floats = floats(16)
water_eye: floats = floats(3) # water_reflection_pass's scratch
water_mirror: floats = floats(16)
water_mv: floats = floats(16)
sc_lod_counts: words = words(64) # layer_partition_lods' scratch (SC_LOD_ROOM)
sc_lod_start: words = words(64)
sc_lod_fill: words = words(64)
sc_gpu_rec: words = words(29 * 5) # layer_gpu_prepare's (SC_RECS * 5) and layer_gpu_cull's
sc_gpu_zeros: words = words(5)
sc_gpu_pr: words = words(36)
sc_gpu_bufs: words = words(4) # the camera's up for DLSS's constants
ov_nine_buf: floats = floats(16) # ov_nine's corners and uvs
gvk_ac: pointer = null # the VkAllocationCallbacks every create and destroy is given (R3D_ALLOC_VK=1)
gvk_prime_b: words = null # buffers made since the frame began, read once so MoltenVK makes their

View file

@ -21,6 +21,7 @@ const F_PI: int = 0x40490FDB
# the n 4x4 matrices laid end to end in `m`, a view of each. Views are made once, where their
# buffer is: a view is a small allocation, and one per matrix per frame is one nobody frees.
@alloc_ok("views over a skeleton's matrices: made once per skeleton, with it")
function m4_views(m: floats, n: int) -> [][]float {
let out = new [][]float
var i = 0
@ -40,6 +41,7 @@ function f_rad(deg: float) -> float { return deg * (PI / 180.0) }
function f_fx(x: float) -> fixed { return fixed(x) }
# ---- vectors ----------------------------------------------------------------
@alloc_ok("a constructor: what it makes is its caller's (records made at start-up or once per figure)")
function v3_new(x: float, y: float, z: float) -> floats {
let v = floats(3)
v[0] = x
@ -78,14 +80,15 @@ function v3_normalize(o: floats, a: floats) -> void {
v3_scale(o, a, inv)
}
function v3_dist(a: floats, b: floats) -> float {
let t = floats(3)
v3_sub(t, a, b)
let d = v3_len(t)
free(t)
let dx = a[0] - b[0]
let dy = a[1] - b[1]
let dz = a[2] - b[2]
let d = Math.sqrt(dx * dx + dy * dy + dz * dz)
return d
}
# ---- matrices (column-major, m[col*4 + row]) ------------------------------------
@alloc_ok("a constructor: what it makes is its caller's (records made at start-up or once per figure)")
function m4_new() -> floats { let m = floats(16); m4_identity(m); return m }
function m4_identity(m: floats) -> void {
for i in 0 .. 16 { m[i] = 0.0 }

View file

@ -163,7 +163,7 @@ function gpu_query_result(render3d_st: mut Render3dState, id: int, out: words) -
# ---- the context --------------------------------------------------------------------
function gpu_open(render3d_st: mut Render3dState, w: int, h: int, title: string) -> bool { return gvk_open(render3d_st, w, h, title) }
function gpu_vsync(render3d_st: mut Render3dState, on: int) -> void { render3d_st.gvk_vsync = on != 0; if render3d_st.gvk_swap != 0 { render3d_st.gvk_swap_stale = true } }
function gpu_renderer_name(render3d_st: Render3dState) -> string { return `{render3d_st.gvk_device_name} (Vulkan)` }
function gpu_renderer_name(render3d_st: Render3dState) -> string { return render3d_st.gvk_renderer_name }
function gpu_resize_check(render3d_st: mut Render3dState) -> bool { return gvk_resize_check(render3d_st) }
# the finished frame: presented to the window, or (headless) the GPU's work finished
function gpu_present(render3d_st: mut Render3dState) -> void { gvk_present(render3d_st) }
@ -225,6 +225,7 @@ function gpu_ext_in(props: bytes, n: int, want: string) -> bool {
# R3D_CAPS=rtx50|rtx40|rtx30|amd|intel|none pretends to be a Windows machine with that GPU,
# so the settings screen can be shot and tested anywhere
@alloc_ok("a developer switch (R3D_CAPS): the caps faked once at start-up")
function gpu_caps_fake(render3d_st: mut Render3dState, kind: string) -> void {
render3d_st.gpu_cap_windows = true
if kind == "none" { return }
@ -237,6 +238,7 @@ function gpu_caps_fake(render3d_st: mut Render3dState, kind: string) -> void {
render3d_st.gpu_cap_device = `test NVIDIA GeForce RTX {kind[3 .. 5]}`
}
@alloc_ok("start-up: the machine's capabilities, probed once (or on a settings page asking)")
function gpu_caps_probe(render3d_st: mut Render3dState) -> void {
if render3d_st.gpu_cap_probed { return }
render3d_st.gpu_cap_probed = true
@ -518,6 +520,7 @@ const GPU_TX_W: int = 12 # per handle: kind, w, h, layers, ifmt, min, m
function gpu_gl_target(kind: int) -> int { if kind == GPU_TEX2D_ARRAY { return GL_TEXTURE_2D_ARRAY }; return GL_TEXTURE_2D }
# the record for a handle, growing the table when a new name is larger than it
@alloc_ok("a texture handle past the 4096 made at start-up: the table doubles to hold it (ids are reused, so it stops)")
function gpu_tx_at(render3d_st: mut Render3dState, tex: int) -> int {
if tex <= 0 { return -1 }
if tex >= render3d_st.gpu_tx_cap {
@ -642,6 +645,7 @@ function gpu_bind_sampler(render3d_st: mut Render3dState, prog: int, name: strin
# a pass drawing into an incomplete target says which one; unset, it calls nothing.
const GPU_FB_W: int = 8 # per handle: colour 0, colour 1, depth texture, depth layer + 1, colour rb, depth rb, samples, colour layer + 1
@alloc_ok("a framebuffer handle past the 1024 made at start-up: the table doubles to hold it (bounded by targets)")
function gpu_fb_at(render3d_st: mut Render3dState, fb: int) -> int {
if fb <= 0 { return -1 }
if fb >= render3d_st.gpu_fb_cap {

View file

@ -131,6 +131,7 @@ function gvk_init(render3d_st: mut Render3dState) -> bool {
render3d_st.gvk_pd = Vk.get_ptr(devs, pick * 8)
Vk.get_physical_device_properties(render3d_st.gvk_pd, props)
render3d_st.gvk_device_name = string(Vk.at(props, VkPhysicalDeviceProperties_deviceName))
render3d_st.gvk_renderer_name = `{render3d_st.gvk_device_name} (Vulkan)`
Vk.put_i32(cnt, 0, 0)
Vk.get_physical_device_queue_family_properties(render3d_st.gvk_pd, cnt, null)
@ -363,7 +364,12 @@ function gvk_mem_raw(render3d_st: mut Render3dState, t: int, size: int, host: bo
let out = gvk_tmp(render3d_st, 8)
render3d_st.gvk_mk_mem += 1
let r = Vk.allocate_memory(render3d_st.gvk_dev, mai, render3d_st.gvk_ac, out)
if r != VK_SUCCESS { gvk_note(render3d_st, `r3d: vulkan: vkAllocateMemory of {size} bytes failed (VkResult {r}, {render3d_st.gvk_n_allocs} allocations live)`); return zero }
if r != VK_SUCCESS {
let msg = `r3d: vulkan: vkAllocateMemory of {size} bytes failed (VkResult {r}, {render3d_st.gvk_n_allocs} allocations live)`
gvk_note(render3d_st, msg)
free(msg)
return zero
}
render3d_st.gvk_n_allocs += 1
let mem = gvk_handle(out)
out_map[0] = null
@ -378,7 +384,12 @@ function gvk_mem_raw(render3d_st: mut Render3dState, t: int, size: int, host: bo
function gvk_fr_put(render3d_st: mut Render3dState, b: int, off: int, len_: int) -> void {
if len_ <= 0 { return }
for i in 0 .. len(render3d_st.gvk_fr_blk) { if render3d_st.gvk_fr_len[i] == 0 { render3d_st.gvk_fr_blk[i] = b; render3d_st.gvk_fr_off[i] = off; render3d_st.gvk_fr_len[i] = len_; return } }
push(render3d_st.gvk_fr_blk, b); push(render3d_st.gvk_fr_off, off); push(render3d_st.gvk_fr_len, len_)
@alloc_ok("within the free ranges' room made at start-up (@max 4096)")
push(render3d_st.gvk_fr_blk, b)
@alloc_ok("within the free ranges' room made at start-up (@max 4096)")
push(render3d_st.gvk_fr_off, off)
@alloc_ok("within the free ranges' room made at start-up (@max 4096)")
push(render3d_st.gvk_fr_len, len_)
}
# Memory for a resource from its requirements (a VkMemoryRequirements): an allocation id, 0 if none.
@ -510,6 +521,7 @@ function gvk_mem_free(render3d_st: mut Render3dState, a: int) -> void {
}
}
render3d_st.gvk_al_len[a] = 0; render3d_st.gvk_al_mem[a] = zero; render3d_st.gvk_al_map[a] = null; render3d_st.gvk_al_blk[a] = -1
@alloc_ok("within the spare ids' room made at start-up (@max 4096)")
push(render3d_st.gvk_al_spare, a)
}
function gvk_mem_id(x: long) -> int { return int(x) }

View file

@ -341,6 +341,7 @@ function gvk_input_format(t: string) -> int {
return VK_FORMAT_R32G32B32A32_SFLOAT
}
# 64 KB of zeros, the buffer every unfed input reads (one vec4 per instance, up to 4096)
@alloc_ok("the one buffer of zeros every unfed input reads: made once")
function gvk_zero_vbuf_get(render3d_st: mut Render3dState) -> int {
if render3d_st.gvk_zero_vbuf == 0 {
render3d_st.gvk_zero_vbuf = gvk_buf_new(render3d_st)
@ -604,7 +605,12 @@ function gvk_pipeline(render3d_st: mut Render3dState, p: int, m: Mesh, st: GvkSt
# the create infos are read by the create and go with it: a pipeline first met in play kept all of them
free(stages); free(attrs); free(bnds); free(vin); free(ias); free(vps); free(rs); free(ms); free(ds)
free(cba); free(cbs); free(dyn_states); free(dys); free(formats); free(prci); free(gpci); free(out)
if r != VK_SUCCESS { gvk_fail(render3d_st, `vkCreateGraphicsPipelines for {v.vs} + {v.fs}`, r); return zero }
if r != VK_SUCCESS {
let what = `vkCreateGraphicsPipelines for {v.vs} + {v.fs}`
gvk_fail(render3d_st, what, r)
free(what)
return zero
}
# R3D_VK_PROF names each pipeline as it is made: one made during play is a stall a warm-up missed
if gvk_prof(render3d_st) { render3d_st.gvk_n_pipe_new += 1; print(`r3d: vulkan pipeline {len(render3d_st.gvk_pipe) + 1}: {v.vs} + {v.fs}, {n_color} colour format {color_fmt}, depth {depth_fmt}, {samples}x`) }
push(render3d_st.gvk_pipe_keys, key)
@ -832,9 +838,11 @@ function gvk_draw_set(render3d_st: mut Render3dState, p: int, tx: words, tx_w: i
let zero: long = 0
let v = render3d_st.gvk_prog_var[p]
gvk_uniform_blocks(render3d_st, p)
@alloc_ok("filled to 4096 programs at start-up (@max 4096)")
while len(render3d_st.gvk_ub_frame) <= p { push(render3d_st.gvk_ub_frame, 0); push(render3d_st.gvk_ub_offv, 0); push(render3d_st.gvk_ub_offf, 0) }
let nt = len(v.t_name)
# the key: every texture as this draw resolves it, and its sampler
@alloc_ok("filled to 130 at start-up (@max 130: 64 textures a draw)")
while len(render3d_st.gvk_sc_tmp) < nt * 2 + 2 { push(render3d_st.gvk_sc_tmp, zero) }
for t in 0 .. nt {
var tex = render3d_st.gvk_prog_tex[p][t]

View file

@ -62,6 +62,7 @@ function gvk_channel_bytes(ifmt: int) -> int {
function gvk_tex_give_back(render3d_st: mut Render3dState, tex: int) -> void {
if tex <= 0 or tex >= len(render3d_st.gvk_tex_image) or render3d_st.gvk_tex_image[tex] == 0 { return }
gvk_tex_release(render3d_st, tex)
@alloc_ok("within the spare ids' room made at start-up (@max 4096)")
push(render3d_st.gvk_tex_spare, tex)
}
@alloc_ok("made once per resource and kept for its life (a texture, program, sampler, view, layout or memory block is created when first asked for)")
@ -672,8 +673,13 @@ function gvk_buf_release(render3d_st: mut Render3dState, b: int) -> void {
let zero: long = 0
if gvk_buf_busy(render3d_st, b) {
# a draw recorded this frame (or the frame in flight) still reads it: destroy it once the frame has been submitted
push(render3d_st.gvk_retired_buf, render3d_st.gvk_buf[b]); push(render3d_st.gvk_retired_mem, render3d_st.gvk_buf_mem[b])
@alloc_ok("within the retire lists' room made at start-up (@max 4096)")
push(render3d_st.gvk_retired_buf, render3d_st.gvk_buf[b])
@alloc_ok("within the retire lists' room made at start-up (@max 4096)")
push(render3d_st.gvk_retired_mem, render3d_st.gvk_buf_mem[b])
@alloc_ok("within the retire lists' room made at start-up (@max 4096)")
while len(render3d_st.gvk_retired_frame) < len(render3d_st.gvk_retired_buf) - 1 { push(render3d_st.gvk_retired_frame, 0) }
@alloc_ok("within the retire lists' room made at start-up (@max 4096)")
push(render3d_st.gvk_retired_frame, render3d_st.gvk_buf_used[b])
} else {
render3d_st.gvk_mk_x_buf += 1
@ -695,6 +701,7 @@ function gvk_retire_flush(render3d_st: mut Render3dState) -> void { gvk_retire_u
# the frame being recorded read stays
function gvk_retire_upto(render3d_st: mut Render3dState, done: int) -> void {
if render3d_st.gvk_retired_buf == null { return }
@alloc_ok("within the retire lists' room made at start-up (@max 4096)")
while len(render3d_st.gvk_retired_frame) < len(render3d_st.gvk_retired_buf) { push(render3d_st.gvk_retired_frame, 0) }
# compacted in place: three fresh lists a frame were never given back
var w = 0
@ -720,6 +727,7 @@ function gvk_retire_upto(render3d_st: mut Render3dState, done: int) -> void {
# A buffer a compute pass writes: a draw later in the same frame reads what the GPU put there,
# so it is never swapped for fresh storage because it was used this frame (gvk_buf_reserve).
function gvk_buf_gpu_owned(render3d_st: mut Render3dState, b: int) -> void {
@alloc_ok("within the room made at start-up (@max 16384 buffers)")
while len(render3d_st.gvk_buf_gpu) <= b { push(render3d_st.gvk_buf_gpu, 0) }
render3d_st.gvk_buf_gpu[b] = 1
}

View file

@ -16,6 +16,7 @@ function r3d_set_paths(render3d_st: mut Render3dState, root: string, assets: str
# The install root, as the compiler computes it: $LUDIC_HOME, else the directory of the
# `ludic` on PATH. A game running from its own tree finds this package there.
@alloc_ok("once a run: the install root, asked only when the shaders are not beside the program")
function r3d_home() -> string {
let env = Os.env("LUDIC_HOME")
if env != null and env != "" { return env }

View file

@ -4,6 +4,7 @@
# major matrices, o may alias its inputs unless stated.
# ============================================================================
@alloc_ok("a constructor: what it makes is its caller's (records made at start-up or once per figure)")
function q_new() -> floats { let q = floats(4); q_identity(q); return q }
function q_identity(q: floats) -> void { q[0] = 0.0; q[1] = 0.0; q[2] = 0.0; q[3] = 1.0 }
function q_set(q: words, x: int, y: int, z: int, w: int) -> void { q[0] = x; q[1] = y; q[2] = z; q[3] = w }

View file

@ -144,145 +144,8 @@ function layer_cards(render3d_st: mut Render3dState, scan: Model, cap: int, wind
return l
}
# A lupine spike (1 m tall): a stem of two crossed quads (uv.x in [0,1]) and
# seven tiers of crossed floret quads (uv.x in [1,2]), coloured in the shader.
function model_lupine(render3d_st: mut Render3dState) -> Model {
let model = new Model
model.prims = new []Prim
let pr = new Prim
let m = gpu_mesh_new(render3d_st)
# quads: stem x2 + tiers 12 x 2 + 3 leaves = 29 quads
let nq = 29
let v = gl_floats(nq * 4 * 8)
let idx = words(nq * 6)
var k = 0
var qi = 0
for q in 0 .. nq {
var w = 0.012; var y0 = 0.0; var y1 = 0.62; var ukind = 0.0
var ang = 0.0
if q >= 2 and q < 26 {
let tier = (q - 2) / 2
let t = float(tier) / 12.0
w = 0.05 * (1.1 - t)
y0 = 0.27 + t * 0.36
y1 = y0 + 0.045
ukind = 1.0
ang = float(tier) / 12.0 * 2.1
if (q & 1) == 1 { ang = ang + PI * 0.5 }
} else if q >= 26 {
# a rosette of three leaves near the ground
w = 0.09; y0 = 0.02; y1 = 0.2; ukind = 2.0
ang = float(q - 26) / 3.0 * (2.0 * PI)
} else {
if (q & 1) == 1 { ang = ang + PI * 0.5 }
}
let cx = Math.cos(ang) * w; let cz = Math.sin(ang) * w
for c in 0 .. 4 {
var sx = -1.0; var sy = y0; var u = 0.0
if c == 1 or c == 2 { sx = 1.0; u = 1.0 }
if c == 2 or c == 3 { sy = y1 }
gl_put_bits(v, k, float_bits(cx * sx)); gl_put_bits(v, k + 1, float_bits(sy)); gl_put_bits(v, k + 2, float_bits(cz * sx))
gl_put_bits(v, k + 3, float_bits(-cz)); gl_put_bits(v, k + 4, float_bits(0.2)); gl_put_bits(v, k + 5, float_bits(cx))
gl_put_bits(v, k + 6, float_bits(ukind + u))
var vv = sy
if ukind == 1.0 { vv = (sy - 0.27) / 0.4 }
if ukind == 2.0 { vv = (sy - 0.02) / 0.18 }
gl_put_bits(v, k + 7, float_bits(vv))
k += 8
}
let b = q * 4
idx[qi] = b; idx[qi + 1] = b + 1; idx[qi + 2] = b + 2; idx[qi + 3] = b; idx[qi + 4] = b + 2; idx[qi + 5] = b + 3
qi += 6
}
gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(nq * 4 * 8), GPU_STATIC)
sc_model_layout(render3d_st, m)
free(v)
gpu_mesh_indices(render3d_st, m, data_of(idx), nq * 6 * 4, 4)
free(idx)
m.count = nq * 6
gpu_mesh_done(render3d_st, m)
pr.mesh = m
if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) }
pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white
push(model.prims, pr)
model.radius = 0.08; model.height = 0.65; model.tris = nq * 2
return model
}
# A dense lupine for baking into a card: a stem, ~220 small floret quads in a
# tapering spiral (uv.x in [1,2]) and five leaves (uv.x in [2,3]).
function model_lupine_dense(render3d_st: mut Render3dState) -> Model {
let model = new Model
model.prims = new []Prim
let pr = new Prim
let m = gpu_mesh_new(render3d_st)
let nfl = 220
let nq = 2 + nfl + 5
let v = gl_floats(nq * 4 * 8)
let idx = words(nq * 6)
var k = 0
var qi = 0
seed(5)
for q in 0 .. nq {
var w = 0.008; var y0 = 0.0; var y1 = 0.66; var ukind = 0.0
var ang = 0.0; var ox = 0.0; var oz = 0.0; var tilt = 0.0
if q >= 2 and q < 2 + nfl {
let t = float(q - 2) / float(nfl)
let yy = 0.28 + t * 0.4
ang = float(q) * 2.39996 # golden angle spiral
let rad = 0.055 * (1.05 - t)
ox = Math.cos(ang) * rad; oz = Math.sin(ang) * rad
w = 0.028 * (1.1 - t * 0.5)
y0 = yy - 0.016; y1 = yy + 0.016
ukind = 1.0
tilt = 0.6
} else if q >= 2 + nfl {
w = 0.05; y0 = 0.03; y1 = 0.16; ukind = 2.0
ang = float(q - 2 - nfl) / 5.0 * (2.0 * PI)
ox = Math.cos(ang) * 0.05; oz = Math.sin(ang) * 0.05
} else {
if (q & 1) == 1 { ang = PI * 0.5 }
}
# the quad faces outward (its normal along the spiral radius), leaning out by `tilt`
let nx = Math.cos(ang); let nz = Math.sin(ang)
let tx = -nz; let tz = nx # tangent (quad width direction)
for c in 0 .. 4 {
var sx = -1.0; var sy = y0; var u = 0.0
if c == 1 or c == 2 { sx = 1.0; u = 1.0 }
if c == 2 or c == 3 { sy = y1 }
var lean = 0.0
if c == 2 or c == 3 { lean = tilt * w }
gl_put_bits(v, k, float_bits(ox + tx * (sx * w) + nx * lean))
gl_put_bits(v, k + 1, float_bits(sy))
gl_put_bits(v, k + 2, float_bits(oz + tz * (sx * w) + nz * lean))
gl_put_bits(v, k + 3, float_bits(nx)); gl_put_bits(v, k + 4, float_bits(0.35)); gl_put_bits(v, k + 5, float_bits(nz))
gl_put_bits(v, k + 6, float_bits(ukind + u))
var vv = sy
if ukind == 1.0 { vv = (sy - 0.27) / 0.42 }
if ukind == 2.0 { vv = (sy - 0.03) / 0.19 }
gl_put_bits(v, k + 7, float_bits(vv))
k += 8
}
let b = q * 4
idx[qi] = b; idx[qi + 1] = b + 1; idx[qi + 2] = b + 2; idx[qi + 3] = b; idx[qi + 4] = b + 2; idx[qi + 5] = b + 3
qi += 6
}
gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(nq * 4 * 8), GPU_STATIC)
sc_model_layout(render3d_st, m)
free(v)
gpu_mesh_indices(render3d_st, m, data_of(idx), nq * 6 * 4, 4)
free(idx)
m.count = nq * 6
gpu_mesh_done(render3d_st, m)
pr.mesh = m
if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) }
pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white
push(model.prims, pr)
model.radius = 0.11; model.height = 0.68; model.tris = nq * 2
return model
}
# A procedural grass blade (1 m tall, 5 cm wide, curved): 5 rows of 2 vertices.
@alloc_ok("a model: made once by the program that asks for it, and kept with its layer")
function model_blade(render3d_st: mut Render3dState) -> Model {
let model = new Model
model.prims = new []Prim
@ -553,6 +416,7 @@ function scatter_begin_frame(render3d_st: mut Render3dState) -> void {
# wrote - one draw per material covering every level. Nothing is partitioned or uploaded on the CPU
# when the view moves. PC camp benchmark: 2791 -> 2657 draws, 4.3 -> 4.1 s for 400 frames.
const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand
const SC_LOD_ROOM: int = 64 # a layer's LODs plus near and far, most (sc_lod_* are made this size)
const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp)
function layer_gpu_eligible(render3d_st: Render3dState, l: Layer) -> bool {
@ -661,7 +525,7 @@ function layer_gpu_prepare(render3d_st: mut Render3dState, l: Layer) -> bool {
gpu_buffer_upload(render3d_st, l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC)
if l.g_arena == null { layer_arena_build(render3d_st, l) }
let n_mat = len(l.g_arena)
let rec = words(SC_RECS * 5)
let rec = render3d_st.sc_gpu_rec
for i in 0 .. SC_RECS * 5 { rec[i] = 0 }
# material j, level k: that level's range of the merged mesh, its instances from bucket k
for j in 0 .. n_mat {
@ -681,10 +545,9 @@ function layer_gpu_prepare(render3d_st: mut Render3dState, l: Layer) -> bool {
}
}
gpu_buffer_upload(render3d_st, l.g_cmds, SC_RECS * SC_REC_W, data_of(rec), GPU_DYNAMIC)
let zeros = words(5)
let zeros = render3d_st.sc_gpu_zeros
for i in 0 .. 5 { zeros[i] = 0 }
gpu_buffer_upload(render3d_st, l.g_counts, 20, data_of(zeros), GPU_DYNAMIC)
free(rec); free(zeros)
# the card casts every instance, as on the CPU path (layer_grid_build uploads this there)
l.n_sh = l.count
gpu_buffer_upload(render3d_st, l.sh_buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_STATIC)
@ -695,7 +558,7 @@ function layer_gpu_prepare(render3d_st: mut Render3dState, l: Layer) -> bool {
# the dispatch for the view as it stands: frustum, camera, distances, the layer's shape
function layer_gpu_cull(render3d_st: mut Render3dState, l: Layer) -> void {
let pr = words(36)
let pr = render3d_st.sc_gpu_pr
for i in 0 .. 36 { pr[i] = 0 }
if render3d_st.cam_planes != null { for i in 0 .. 16 { pr[i] = float_bits(render3d_st.cam_planes[i]) } }
pr[16] = float_bits(render3d_st.cam_pos[0]); pr[17] = float_bits(render3d_st.cam_pos[1]); pr[18] = float_bits(render3d_st.cam_pos[2]); pr[19] = float_bits(l.cull)
@ -703,10 +566,9 @@ function layer_gpu_cull(render3d_st: mut Render3dState, l: Layer) -> void {
pr[28] = l.count; pr[29] = l.count; pr[30] = l.n_lods; pr[31] = 1
# as layer_grid_gather pads a cell: the tallest instance, plus a margin
pr[32] = float_bits(l.lods[0].height * 2.0); pr[33] = float_bits(4.0)
let bufs = words(4)
let bufs = render3d_st.sc_gpu_bufs
bufs[0] = l.g_src; bufs[1] = l.g_dst; bufs[2] = l.g_cmds; bufs[3] = l.g_counts
gpu_dispatch(render3d_st, render3d_st.sc_cull_prog, data_of(pr), 144, bufs, 1)
free(pr); free(bufs)
}
# Sort a static layer's instances into square cells (call once, after placement; a
@ -789,7 +651,8 @@ function layer_grid_gather(render3d_st: Render3dState, l: Layer) -> void {
# the impostor bucket last, and upload one buffer per level.
function layer_partition_lods(render3d_st: mut Render3dState, l: Layer, src: floats, total: int) -> void {
let n = l.n_lods
let counts = words(n + 2)
if n + 2 > SC_LOD_ROOM { return } # past the room made with the state
let counts = render3d_st.sc_lod_counts
for k in 0 .. n + 2 { counts[k] = 0 }
let cull2 = l.cull * l.cull
let open = l.lod_dist[n - 1] == 0.0 # the last level runs out to the cull distance
@ -810,10 +673,10 @@ function layer_partition_lods(render3d_st: mut Render3dState, l: Layer, src: flo
counts[lv] += 1
}
# prefix offsets (in instances) per bucket
let start = words(n + 2)
let start = render3d_st.sc_lod_start
var acc = 0
for k in 0 .. n + 2 { start[k] = acc; acc += counts[k] }
let fill = words(n + 2)
let fill = render3d_st.sc_lod_fill
for k in 0 .. n + 2 { fill[k] = start[k] }
let tmp = l.scratch
for i in 0 .. total {
@ -841,7 +704,6 @@ function layer_partition_lods(render3d_st: mut Render3dState, l: Layer, src: flo
l.n_sh = total
if total > 0 { gpu_buffer_upload(render3d_st, l.sh_buf, total * INST_FLOATS * 4, data_of(src), GPU_DYNAMIC) }
}
free(counts); free(start); free(fill)
}
# split the instances by distance to the camera (only when the view changed)

View file

@ -80,8 +80,10 @@ function shadow_fit(render3d_st: mut Render3dState, c: int, near: float, far: fl
# map resampled every frame: that is the crawl and flicker seen while moving.
m4_perspective(render3d_st.sh_tmp_proj, render3d_st.cam_fov, render3d_st.cam_aspect, near, far)
m4_inverse(render3d_st.sh_tmp_inv, render3d_st.sh_tmp_proj) # NDC -> view space
let cview = v3_new(0.0, 0.0, 0.0)
let corners = floats(24)
# the fit's scratch is the state's, made once (a cascade fitted every frame made six of them)
let cview = render3d_st.sh_fit_cview
v3_set(cview, 0.0, 0.0, 0.0)
let corners = render3d_st.sh_fit_corners
for i in 0 .. 8 {
var x = -1.0; var y = -1.0; var z = -1.0
if (i & 1) != 0 { x = 1.0 }
@ -102,16 +104,16 @@ function shadow_fit(render3d_st: mut Render3dState, c: int, near: float, far: fl
radius = radius * 1.05
# the slice centre back into world space
m4_inverse(render3d_st.sh_tmp_vp, render3d_st.cam_view)
let center = floats(3)
let center = render3d_st.sh_fit_center
m4_xform_point(center, render3d_st.sh_tmp_vp, cview[0], cview[1], cview[2])
free(cview)
# light view: from far along the sun direction, looking at the centre
let eye = floats(3)
let eye = render3d_st.sh_fit_eye
# casters up to ~900 m toward the sun (a mountain across the valley), and the
# slice itself behind the centre: a tight depth range keeps the bias small
let back = radius + 900.0
v3_madd(eye, center, render3d_st.sun_dir, back)
let up = v3_new(0.0, 1.0, 0.0)
let up = render3d_st.sh_fit_up
v3_set(up, 0.0, 1.0, 0.0)
m4_look_at(render3d_st.sh_tmp_view, eye, center, up)
# snap the ortho window to the shadow texel grid
let texel = radius * 2.0 / float(render3d_st.shadow_res)
@ -123,10 +125,9 @@ function shadow_fit(render3d_st: mut Render3dState, c: int, near: float, far: fl
m4_ortho(render3d_st.sh_tmp_proj, nr + ox, radius + ox, nr + oy, radius + oy, 1.0, zfar)
render3d_st.sh_range[c] = zfar - 1.0
render3d_st.sh_texel[c] = texel
let out = floats(16)
let out = render3d_st.sh_fit_out
m4_mul(out, render3d_st.sh_tmp_proj, render3d_st.sh_tmp_view)
for i in 0 .. 16 { render3d_st.sh_vp[c * 16 + i] = out[i] }
free(out); free(eye); free(up); free(center); free(corners)
}
function shadow_cascade_vp(render3d_st: Render3dState, c: int) -> floats { return render3d_st.sh_vp_v[c] }

View file

@ -214,6 +214,7 @@ function skin_bind(render3d_st: mut Render3dState, sk: Skin, prog: int) -> void
# the same skeleton posed on its own: shares the rest data, owns the pose and the matrices
@creates(SkinClone)
@alloc_ok("one per figure placed (a person, an animal): actor_release frees it with the actor")
function skin_clone(src: Skin) -> Skin {
let sk = new Skin
sk.n_nodes = src.n_nodes; sk.par = src.par; sk.walk = src.walk

View file

@ -193,6 +193,7 @@ function stream_evict(render3d_st: mut Render3dState, s: Stream) -> void {
w += 1
} else {
c.off = -1
@alloc_ok("within the stream's record pool, all made with the stream (STREAM_MAX_CHUNKS)")
push(s.spare, c)
}
i += 1
@ -263,6 +264,7 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float,
if need > 0 { mem_copy(mem_off(data_of(s.arena), s.top * 4), data_of(render3d_st.stream_scratch), need * 4) }
c.off = s.top
s.top += need
@alloc_ok("within the stream's record pool, all made with the stream (STREAM_MAX_CHUNKS)")
push(s.chunks, c); s.keys[s.n] = key; stream_remember(s, key, s.n); s.n += 1
c.used = render3d_st.stream_walk_no
} else { loose = true }
@ -287,6 +289,7 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float,
l.count += c.count
render3d_st.stream_us_gather = render3d_st.stream_us_gather + (gl_now_us() - tg)
}
@alloc_ok("within the stream's record pool, all made with the stream (STREAM_MAX_CHUNKS)")
if loose and c != render3d_st.stream_loose { push(s.spare, c) }
}
}

View file

@ -36,6 +36,7 @@ function r3d_set_anisotropy(render3d_st: mut Render3dState, level: int) -> void
if keep > 0 { gpu_tex_bind(render3d_st, GPU_TEX2D, keep) }
}
@alloc_ok("a load: a file read whole, kept or freed by its caller")
function r3d_read_file(render3d_st: mut Render3dState, path: pointer) -> pointer {
let f = file_open(path, "rb")
if f == null { return null }
@ -483,6 +484,7 @@ function tex_dump(render3d_st: mut Render3dState, tex: int, w: int, h: int, path
# A binary PPM (P6, what Gl.screenshot writes) as an RGB8 texture, box-filtered down by
# `shrink` (a photo thumbnail); 0 when the file is missing.
@alloc_ok("a load: an image read from disk")
function tex_load_ppm(render3d_st: mut Render3dState, path: pointer, shrink: int) -> int {
let d = r3d_read_file(render3d_st, path)
if d == null { return 0 }

View file

@ -36,15 +36,16 @@ function water_reflection_pass(render3d_st: mut Render3dState) -> void {
# the mirrored camera: view' = view * R, R reflecting y about the surface (y' = 2L - y).
# R has determinant -1, so the winding flips (front faces culled below) and the image
# lands exactly where the main camera's pixels expect the reflection.
let eye = v3_new(sx, 2.0 * render3d_st.water_level - sy, sz)
let refl = m4_new()
# the pass's scratch is the state's, made once
let eye = render3d_st.water_eye
v3_set(eye, sx, 2.0 * render3d_st.water_level - sy, sz)
let refl = render3d_st.water_mirror
m4_identity(refl)
refl[5] = -1.0
refl[13] = 2.0 * render3d_st.water_level
let mv = floats(16)
let mv = render3d_st.water_mv
m4_mul(mv, render3d_st.water_saved, refl)
m4_copy(render3d_st.cam_view, mv)
free(mv); free(refl)
let fwd = words(3); let up = words(3); let at = words(3)
m4_mul(render3d_st.cam_vp, render3d_st.cam_proj, render3d_st.cam_view)
m4_inverse(render3d_st.cam_inv_vp, render3d_st.cam_vp)
v3_copy(render3d_st.cam_pos, eye)
@ -79,7 +80,6 @@ function water_reflection_pass(render3d_st: mut Render3dState) -> void {
m4_copy(render3d_st.cam_vp, render3d_st.water_saved_vp)
m4_copy(render3d_st.cam_inv_vp, render3d_st.water_saved_ivp)
v3_set(render3d_st.cam_pos, sx, sy, sz)
free(eye); free(fwd); free(up); free(at)
gpu_fb_bind(render3d_st, 0)
}

View file

@ -96,9 +96,11 @@ function value_set_float(o: Val, key: pointer, x: float) -> void {
let v = value_slot(o, key, 7)
value_num_set(v, float_bits(x))
}
# a component's text held interned: one copy per distinct text, so the caller's own (a template built
# every frame) is given back with its frame rather than kept by the model
function value_set_str(o: Val, key: pointer, s: pointer) -> void {
let v = value_slot(o, key, 4)
v.txt = s
v.txt = intern(s)
}
function value_set_bool(o: Val, key: pointer, b: bool) -> void {
let v = value_slot(o, key, 3)
@ -142,7 +144,7 @@ function value_set_strs(o: Val, key: pointer, xs: []string) -> void {
let l = value_list_fit(o, key, len(xs))
for i in 0 .. len(xs) {
let v = value_item(l, i, 4)
v.txt = xs[i]
v.txt = intern(xs[i])
}
}
function value_set_bools(o: Val, key: pointer, xs: []bool) -> void {
@ -197,7 +199,7 @@ function value_into_float(into: Val, x: float) -> Val {
}
function value_into_str(into: Val, s: pointer) -> Val {
let v = value_into(into, 4)
v.txt = s
v.txt = intern(s)
return v
}
function value_into_bool(into: Val, b: bool) -> Val {
@ -235,7 +237,7 @@ function value_into_strs(into: Val, xs: []string) -> Val {
let l = value_into_list(into, len(xs))
for i in 0 .. len(xs) {
let v = value_item(l, i, 4)
v.txt = xs[i]
v.txt = intern(xs[i])
}
return l
}