From f993c4a36a819dc05d5aa1e57aa3ff716ea95bc7 Mon Sep 17 00:00:00 2001 From: Orkuncakilkaya Date: Mon, 28 Sep 2026 20:10:59 +0300 Subject: [PATCH] r3d/runtime: allocs, keeps and births at 0 on this side render3d: shadow_fit, water_reflection_pass, layer_partition_lods and the GPU cull's scratch are made with the state; v3_dist is scalar; the pushes into lists sized at start-up, the caps probe, the table growth, the loads and the constructors declared with their bounds (one statement a line); the renderer's name made once with the device; the two error messages given back; the dead lupine models removed. runtime: a component's text is held interned in its value cell (one copy per distinct text), so the getter's own text goes with its frame instead of being kept by ludic.ui's model - 80 of the 83 keeps. Co-Authored-By: Claude Opus 5.5 --- packages/ludic.render3d/FEATURES.md | 2 +- packages/ludic.render3d/actor.ludic | 2 + packages/ludic.render3d/drawstats.ludic | 1 + packages/ludic.render3d/env.ludic | 19 ++- packages/ludic.render3d/fmath.ludic | 11 +- packages/ludic.render3d/gpu.ludic | 6 +- packages/ludic.render3d/gpu_vk.ludic | 16 ++- packages/ludic.render3d/gpu_vk_draw.ludic | 10 +- packages/ludic.render3d/gpu_vk_res.ludic | 10 +- packages/ludic.render3d/programs.ludic | 1 + packages/ludic.render3d/quat.ludic | 1 + packages/ludic.render3d/scatter.ludic | 158 ++-------------------- packages/ludic.render3d/shadow.ludic | 17 +-- packages/ludic.render3d/skin.ludic | 1 + packages/ludic.render3d/stream.ludic | 3 + packages/ludic.render3d/texture.ludic | 2 + packages/ludic.render3d/water.ludic | 12 +- runtime/native/value.ludic | 10 +- 18 files changed, 105 insertions(+), 177 deletions(-) diff --git a/packages/ludic.render3d/FEATURES.md b/packages/ludic.render3d/FEATURES.md index b0eaeddc..d4e6b1ce 100644 --- a/packages/ludic.render3d/FEATURES.md +++ b/packages/ludic.render3d/FEATURES.md @@ -54,7 +54,7 @@ Geometry, terrain and world 29. GPU instancing with divisor attributes and per-instance transforms — `scatter.ludic`, `shaders/model.vert` 30. Baked impostors (16-view atlas with coverage and normals, hull-normal blending, LOD switch) — `scatter.ludic` `impostor_bake`, `shaders/impostor.*` 31. Baked branch-card trees: wedge slices of a scanned crown (sectors × height bands) on radial cards over the real trunk mesh — `scatter.ludic` `impostor_bake_wedges`, `model_branch_cards` -32. Card atlases baked from scanned grass clumps and a dense procedural lupine, instanced as crossed cards — `scatter.ludic` `layer_cards`, `model_lupine_dense` +32. Card atlases baked from scanned grass clumps and a dense procedural lupine, instanced as crossed cards — `scatter.ludic` `layer_cards` 33. Procedural grass blades with wind, dryness and base occlusion; distant blades widen, lie flat to the ground normal and sink into the carpet at the ring's edge — `scatter.ludic` `model_blade`, `shaders/model.vert`, `shaders/model.frag` `BLADE` 34. Wind animation (height-weighted sway, per-instance phase) — `shaders/model.vert` `WIND` 35. Chunk-streamed ground cover with distance bands (blades in four rings, densest underfoot), deterministic per-chunk generation, nearest-first amortised over frames with an instance budget — `stream.ludic` diff --git a/packages/ludic.render3d/actor.ludic b/packages/ludic.render3d/actor.ludic index 0d53a1e9..199be570 100644 --- a/packages/ludic.render3d/actor.ludic +++ b/packages/ludic.render3d/actor.ludic @@ -119,6 +119,7 @@ function actor_remove(render3d_st: mut Render3dState, a: Actor) -> void { function actor_keep(render3d_st: mut Render3dState, a: Actor) -> void { if a == null { return } a.row = len(render3d_st.ac_actors) + @alloc_ok("within the stage's room made at start-up (@max 2048)") push(render3d_st.ac_actors, a) } # colour one named part of the model (a material name from the file) @@ -470,6 +471,7 @@ function actor_release(render3d_st: mut Render3dState, a: Actor) -> void { a.model = null; a.skin = null; a.ocol = null a.id = 0 if render3d_st.ac_spare == null { return } # made in actor_init, with its room + @alloc_ok("within the spares' room made at start-up (@max 2048)") push(render3d_st.ac_spare, a) } diff --git a/packages/ludic.render3d/drawstats.ludic b/packages/ludic.render3d/drawstats.ludic index 756674a5..5058c97e 100644 --- a/packages/ludic.render3d/drawstats.ludic +++ b/packages/ludic.render3d/drawstats.ludic @@ -119,6 +119,7 @@ function ds_kind_name(k: int) -> pointer { return "indirect" } # a per-frame mean with one decimal +@alloc_ok("the draw-count report: printed once, at the end of a run that asked for it") function ds_per(render3d_st: Render3dState, v: long) -> string { let x = (v * 10) / render3d_st.ds_frames return `{string(x / 10)}.{string(x % 10)}` diff --git a/packages/ludic.render3d/env.ludic b/packages/ludic.render3d/env.ludic index 760fe135..8dcd805f 100644 --- a/packages/ludic.render3d/env.ludic +++ b/packages/ludic.render3d/env.ludic @@ -166,6 +166,7 @@ export state Render3dState { gvk_family: int = -1 gvk_mp: bytes = null # VkPhysicalDeviceMemoryProperties gvk_device_name: string = "" + gvk_renderer_name: string = "" # " (Vulkan)", made once with the device gvk_why: string = "" # why the device did not come up, for the fallback notice gvk_max_aniso: int = 0x3F800000 # the device's anisotropy limit as float bits; 1.0 when it has none gvk_want_surface: bool = false # set before gvk_init when the renderer will present to a window @@ -692,7 +693,23 @@ export state Render3dState { stream_debug_n: int = 0 stream_scratch: floats = null gvk_pipe_bufs: words = null # a pipeline's vertex buffers, while it is made - gsl_up: floats = floats(3) # the camera's up for DLSS's constants + gsl_up: floats = floats(3) + sh_fit_cview: floats = floats(3) # shadow_fit's scratch + sh_fit_corners: floats = floats(24) + sh_fit_center: floats = floats(3) + sh_fit_eye: floats = floats(3) + sh_fit_up: floats = floats(3) + sh_fit_out: floats = floats(16) + water_eye: floats = floats(3) # water_reflection_pass's scratch + water_mirror: floats = floats(16) + water_mv: floats = floats(16) + sc_lod_counts: words = words(64) # layer_partition_lods' scratch (SC_LOD_ROOM) + sc_lod_start: words = words(64) + sc_lod_fill: words = words(64) + sc_gpu_rec: words = words(29 * 5) # layer_gpu_prepare's (SC_RECS * 5) and layer_gpu_cull's + sc_gpu_zeros: words = words(5) + sc_gpu_pr: words = words(36) + sc_gpu_bufs: words = words(4) # the camera's up for DLSS's constants ov_nine_buf: floats = floats(16) # ov_nine's corners and uvs gvk_ac: pointer = null # the VkAllocationCallbacks every create and destroy is given (R3D_ALLOC_VK=1) gvk_prime_b: words = null # buffers made since the frame began, read once so MoltenVK makes their diff --git a/packages/ludic.render3d/fmath.ludic b/packages/ludic.render3d/fmath.ludic index 54962cd6..11238a2d 100644 --- a/packages/ludic.render3d/fmath.ludic +++ b/packages/ludic.render3d/fmath.ludic @@ -21,6 +21,7 @@ const F_PI: int = 0x40490FDB # the n 4x4 matrices laid end to end in `m`, a view of each. Views are made once, where their # buffer is: a view is a small allocation, and one per matrix per frame is one nobody frees. +@alloc_ok("views over a skeleton's matrices: made once per skeleton, with it") function m4_views(m: floats, n: int) -> [][]float { let out = new [][]float var i = 0 @@ -40,6 +41,7 @@ function f_rad(deg: float) -> float { return deg * (PI / 180.0) } function f_fx(x: float) -> fixed { return fixed(x) } # ---- vectors ---------------------------------------------------------------- +@alloc_ok("a constructor: what it makes is its caller's (records made at start-up or once per figure)") function v3_new(x: float, y: float, z: float) -> floats { let v = floats(3) v[0] = x @@ -78,14 +80,15 @@ function v3_normalize(o: floats, a: floats) -> void { v3_scale(o, a, inv) } function v3_dist(a: floats, b: floats) -> float { - let t = floats(3) - v3_sub(t, a, b) - let d = v3_len(t) - free(t) + let dx = a[0] - b[0] + let dy = a[1] - b[1] + let dz = a[2] - b[2] + let d = Math.sqrt(dx * dx + dy * dy + dz * dz) return d } # ---- matrices (column-major, m[col*4 + row]) ------------------------------------ +@alloc_ok("a constructor: what it makes is its caller's (records made at start-up or once per figure)") function m4_new() -> floats { let m = floats(16); m4_identity(m); return m } function m4_identity(m: floats) -> void { for i in 0 .. 16 { m[i] = 0.0 } diff --git a/packages/ludic.render3d/gpu.ludic b/packages/ludic.render3d/gpu.ludic index 07ea1439..c37a508b 100644 --- a/packages/ludic.render3d/gpu.ludic +++ b/packages/ludic.render3d/gpu.ludic @@ -163,7 +163,7 @@ function gpu_query_result(render3d_st: mut Render3dState, id: int, out: words) - # ---- the context -------------------------------------------------------------------- function gpu_open(render3d_st: mut Render3dState, w: int, h: int, title: string) -> bool { return gvk_open(render3d_st, w, h, title) } function gpu_vsync(render3d_st: mut Render3dState, on: int) -> void { render3d_st.gvk_vsync = on != 0; if render3d_st.gvk_swap != 0 { render3d_st.gvk_swap_stale = true } } -function gpu_renderer_name(render3d_st: Render3dState) -> string { return `{render3d_st.gvk_device_name} (Vulkan)` } +function gpu_renderer_name(render3d_st: Render3dState) -> string { return render3d_st.gvk_renderer_name } function gpu_resize_check(render3d_st: mut Render3dState) -> bool { return gvk_resize_check(render3d_st) } # the finished frame: presented to the window, or (headless) the GPU's work finished function gpu_present(render3d_st: mut Render3dState) -> void { gvk_present(render3d_st) } @@ -225,6 +225,7 @@ function gpu_ext_in(props: bytes, n: int, want: string) -> bool { # R3D_CAPS=rtx50|rtx40|rtx30|amd|intel|none pretends to be a Windows machine with that GPU, # so the settings screen can be shot and tested anywhere +@alloc_ok("a developer switch (R3D_CAPS): the caps faked once at start-up") function gpu_caps_fake(render3d_st: mut Render3dState, kind: string) -> void { render3d_st.gpu_cap_windows = true if kind == "none" { return } @@ -237,6 +238,7 @@ function gpu_caps_fake(render3d_st: mut Render3dState, kind: string) -> void { render3d_st.gpu_cap_device = `test NVIDIA GeForce RTX {kind[3 .. 5]}` } +@alloc_ok("start-up: the machine's capabilities, probed once (or on a settings page asking)") function gpu_caps_probe(render3d_st: mut Render3dState) -> void { if render3d_st.gpu_cap_probed { return } render3d_st.gpu_cap_probed = true @@ -518,6 +520,7 @@ const GPU_TX_W: int = 12 # per handle: kind, w, h, layers, ifmt, min, m function gpu_gl_target(kind: int) -> int { if kind == GPU_TEX2D_ARRAY { return GL_TEXTURE_2D_ARRAY }; return GL_TEXTURE_2D } # the record for a handle, growing the table when a new name is larger than it +@alloc_ok("a texture handle past the 4096 made at start-up: the table doubles to hold it (ids are reused, so it stops)") function gpu_tx_at(render3d_st: mut Render3dState, tex: int) -> int { if tex <= 0 { return -1 } if tex >= render3d_st.gpu_tx_cap { @@ -642,6 +645,7 @@ function gpu_bind_sampler(render3d_st: mut Render3dState, prog: int, name: strin # a pass drawing into an incomplete target says which one; unset, it calls nothing. const GPU_FB_W: int = 8 # per handle: colour 0, colour 1, depth texture, depth layer + 1, colour rb, depth rb, samples, colour layer + 1 +@alloc_ok("a framebuffer handle past the 1024 made at start-up: the table doubles to hold it (bounded by targets)") function gpu_fb_at(render3d_st: mut Render3dState, fb: int) -> int { if fb <= 0 { return -1 } if fb >= render3d_st.gpu_fb_cap { diff --git a/packages/ludic.render3d/gpu_vk.ludic b/packages/ludic.render3d/gpu_vk.ludic index d777bd29..21d4c5f7 100644 --- a/packages/ludic.render3d/gpu_vk.ludic +++ b/packages/ludic.render3d/gpu_vk.ludic @@ -131,6 +131,7 @@ function gvk_init(render3d_st: mut Render3dState) -> bool { render3d_st.gvk_pd = Vk.get_ptr(devs, pick * 8) Vk.get_physical_device_properties(render3d_st.gvk_pd, props) render3d_st.gvk_device_name = string(Vk.at(props, VkPhysicalDeviceProperties_deviceName)) + render3d_st.gvk_renderer_name = `{render3d_st.gvk_device_name} (Vulkan)` Vk.put_i32(cnt, 0, 0) Vk.get_physical_device_queue_family_properties(render3d_st.gvk_pd, cnt, null) @@ -363,7 +364,12 @@ function gvk_mem_raw(render3d_st: mut Render3dState, t: int, size: int, host: bo let out = gvk_tmp(render3d_st, 8) render3d_st.gvk_mk_mem += 1 let r = Vk.allocate_memory(render3d_st.gvk_dev, mai, render3d_st.gvk_ac, out) - if r != VK_SUCCESS { gvk_note(render3d_st, `r3d: vulkan: vkAllocateMemory of {size} bytes failed (VkResult {r}, {render3d_st.gvk_n_allocs} allocations live)`); return zero } + if r != VK_SUCCESS { + let msg = `r3d: vulkan: vkAllocateMemory of {size} bytes failed (VkResult {r}, {render3d_st.gvk_n_allocs} allocations live)` + gvk_note(render3d_st, msg) + free(msg) + return zero + } render3d_st.gvk_n_allocs += 1 let mem = gvk_handle(out) out_map[0] = null @@ -378,7 +384,12 @@ function gvk_mem_raw(render3d_st: mut Render3dState, t: int, size: int, host: bo function gvk_fr_put(render3d_st: mut Render3dState, b: int, off: int, len_: int) -> void { if len_ <= 0 { return } for i in 0 .. len(render3d_st.gvk_fr_blk) { if render3d_st.gvk_fr_len[i] == 0 { render3d_st.gvk_fr_blk[i] = b; render3d_st.gvk_fr_off[i] = off; render3d_st.gvk_fr_len[i] = len_; return } } - push(render3d_st.gvk_fr_blk, b); push(render3d_st.gvk_fr_off, off); push(render3d_st.gvk_fr_len, len_) + @alloc_ok("within the free ranges' room made at start-up (@max 4096)") + push(render3d_st.gvk_fr_blk, b) + @alloc_ok("within the free ranges' room made at start-up (@max 4096)") + push(render3d_st.gvk_fr_off, off) + @alloc_ok("within the free ranges' room made at start-up (@max 4096)") + push(render3d_st.gvk_fr_len, len_) } # Memory for a resource from its requirements (a VkMemoryRequirements): an allocation id, 0 if none. @@ -510,6 +521,7 @@ function gvk_mem_free(render3d_st: mut Render3dState, a: int) -> void { } } render3d_st.gvk_al_len[a] = 0; render3d_st.gvk_al_mem[a] = zero; render3d_st.gvk_al_map[a] = null; render3d_st.gvk_al_blk[a] = -1 + @alloc_ok("within the spare ids' room made at start-up (@max 4096)") push(render3d_st.gvk_al_spare, a) } function gvk_mem_id(x: long) -> int { return int(x) } diff --git a/packages/ludic.render3d/gpu_vk_draw.ludic b/packages/ludic.render3d/gpu_vk_draw.ludic index 99410eea..c1a5f08e 100644 --- a/packages/ludic.render3d/gpu_vk_draw.ludic +++ b/packages/ludic.render3d/gpu_vk_draw.ludic @@ -341,6 +341,7 @@ function gvk_input_format(t: string) -> int { return VK_FORMAT_R32G32B32A32_SFLOAT } # 64 KB of zeros, the buffer every unfed input reads (one vec4 per instance, up to 4096) +@alloc_ok("the one buffer of zeros every unfed input reads: made once") function gvk_zero_vbuf_get(render3d_st: mut Render3dState) -> int { if render3d_st.gvk_zero_vbuf == 0 { render3d_st.gvk_zero_vbuf = gvk_buf_new(render3d_st) @@ -604,7 +605,12 @@ function gvk_pipeline(render3d_st: mut Render3dState, p: int, m: Mesh, st: GvkSt # the create infos are read by the create and go with it: a pipeline first met in play kept all of them free(stages); free(attrs); free(bnds); free(vin); free(ias); free(vps); free(rs); free(ms); free(ds) free(cba); free(cbs); free(dyn_states); free(dys); free(formats); free(prci); free(gpci); free(out) - if r != VK_SUCCESS { gvk_fail(render3d_st, `vkCreateGraphicsPipelines for {v.vs} + {v.fs}`, r); return zero } + if r != VK_SUCCESS { + let what = `vkCreateGraphicsPipelines for {v.vs} + {v.fs}` + gvk_fail(render3d_st, what, r) + free(what) + return zero + } # R3D_VK_PROF names each pipeline as it is made: one made during play is a stall a warm-up missed if gvk_prof(render3d_st) { render3d_st.gvk_n_pipe_new += 1; print(`r3d: vulkan pipeline {len(render3d_st.gvk_pipe) + 1}: {v.vs} + {v.fs}, {n_color} colour format {color_fmt}, depth {depth_fmt}, {samples}x`) } push(render3d_st.gvk_pipe_keys, key) @@ -832,9 +838,11 @@ function gvk_draw_set(render3d_st: mut Render3dState, p: int, tx: words, tx_w: i let zero: long = 0 let v = render3d_st.gvk_prog_var[p] gvk_uniform_blocks(render3d_st, p) + @alloc_ok("filled to 4096 programs at start-up (@max 4096)") while len(render3d_st.gvk_ub_frame) <= p { push(render3d_st.gvk_ub_frame, 0); push(render3d_st.gvk_ub_offv, 0); push(render3d_st.gvk_ub_offf, 0) } let nt = len(v.t_name) # the key: every texture as this draw resolves it, and its sampler + @alloc_ok("filled to 130 at start-up (@max 130: 64 textures a draw)") while len(render3d_st.gvk_sc_tmp) < nt * 2 + 2 { push(render3d_st.gvk_sc_tmp, zero) } for t in 0 .. nt { var tex = render3d_st.gvk_prog_tex[p][t] diff --git a/packages/ludic.render3d/gpu_vk_res.ludic b/packages/ludic.render3d/gpu_vk_res.ludic index 57072420..ec403195 100644 --- a/packages/ludic.render3d/gpu_vk_res.ludic +++ b/packages/ludic.render3d/gpu_vk_res.ludic @@ -62,6 +62,7 @@ function gvk_channel_bytes(ifmt: int) -> int { function gvk_tex_give_back(render3d_st: mut Render3dState, tex: int) -> void { if tex <= 0 or tex >= len(render3d_st.gvk_tex_image) or render3d_st.gvk_tex_image[tex] == 0 { return } gvk_tex_release(render3d_st, tex) + @alloc_ok("within the spare ids' room made at start-up (@max 4096)") push(render3d_st.gvk_tex_spare, tex) } @alloc_ok("made once per resource and kept for its life (a texture, program, sampler, view, layout or memory block is created when first asked for)") @@ -672,8 +673,13 @@ function gvk_buf_release(render3d_st: mut Render3dState, b: int) -> void { let zero: long = 0 if gvk_buf_busy(render3d_st, b) { # a draw recorded this frame (or the frame in flight) still reads it: destroy it once the frame has been submitted - push(render3d_st.gvk_retired_buf, render3d_st.gvk_buf[b]); push(render3d_st.gvk_retired_mem, render3d_st.gvk_buf_mem[b]) + @alloc_ok("within the retire lists' room made at start-up (@max 4096)") + push(render3d_st.gvk_retired_buf, render3d_st.gvk_buf[b]) + @alloc_ok("within the retire lists' room made at start-up (@max 4096)") + push(render3d_st.gvk_retired_mem, render3d_st.gvk_buf_mem[b]) + @alloc_ok("within the retire lists' room made at start-up (@max 4096)") while len(render3d_st.gvk_retired_frame) < len(render3d_st.gvk_retired_buf) - 1 { push(render3d_st.gvk_retired_frame, 0) } + @alloc_ok("within the retire lists' room made at start-up (@max 4096)") push(render3d_st.gvk_retired_frame, render3d_st.gvk_buf_used[b]) } else { render3d_st.gvk_mk_x_buf += 1 @@ -695,6 +701,7 @@ function gvk_retire_flush(render3d_st: mut Render3dState) -> void { gvk_retire_u # the frame being recorded read stays function gvk_retire_upto(render3d_st: mut Render3dState, done: int) -> void { if render3d_st.gvk_retired_buf == null { return } + @alloc_ok("within the retire lists' room made at start-up (@max 4096)") while len(render3d_st.gvk_retired_frame) < len(render3d_st.gvk_retired_buf) { push(render3d_st.gvk_retired_frame, 0) } # compacted in place: three fresh lists a frame were never given back var w = 0 @@ -720,6 +727,7 @@ function gvk_retire_upto(render3d_st: mut Render3dState, done: int) -> void { # A buffer a compute pass writes: a draw later in the same frame reads what the GPU put there, # so it is never swapped for fresh storage because it was used this frame (gvk_buf_reserve). function gvk_buf_gpu_owned(render3d_st: mut Render3dState, b: int) -> void { + @alloc_ok("within the room made at start-up (@max 16384 buffers)") while len(render3d_st.gvk_buf_gpu) <= b { push(render3d_st.gvk_buf_gpu, 0) } render3d_st.gvk_buf_gpu[b] = 1 } diff --git a/packages/ludic.render3d/programs.ludic b/packages/ludic.render3d/programs.ludic index 7acf1eb9..f4db2a03 100644 --- a/packages/ludic.render3d/programs.ludic +++ b/packages/ludic.render3d/programs.ludic @@ -16,6 +16,7 @@ function r3d_set_paths(render3d_st: mut Render3dState, root: string, assets: str # The install root, as the compiler computes it: $LUDIC_HOME, else the directory of the # `ludic` on PATH. A game running from its own tree finds this package there. +@alloc_ok("once a run: the install root, asked only when the shaders are not beside the program") function r3d_home() -> string { let env = Os.env("LUDIC_HOME") if env != null and env != "" { return env } diff --git a/packages/ludic.render3d/quat.ludic b/packages/ludic.render3d/quat.ludic index 4b771e92..fcb6cb16 100644 --- a/packages/ludic.render3d/quat.ludic +++ b/packages/ludic.render3d/quat.ludic @@ -4,6 +4,7 @@ # major matrices, o may alias its inputs unless stated. # ============================================================================ +@alloc_ok("a constructor: what it makes is its caller's (records made at start-up or once per figure)") function q_new() -> floats { let q = floats(4); q_identity(q); return q } function q_identity(q: floats) -> void { q[0] = 0.0; q[1] = 0.0; q[2] = 0.0; q[3] = 1.0 } function q_set(q: words, x: int, y: int, z: int, w: int) -> void { q[0] = x; q[1] = y; q[2] = z; q[3] = w } diff --git a/packages/ludic.render3d/scatter.ludic b/packages/ludic.render3d/scatter.ludic index be8d06fd..48d70f6a 100644 --- a/packages/ludic.render3d/scatter.ludic +++ b/packages/ludic.render3d/scatter.ludic @@ -144,145 +144,8 @@ function layer_cards(render3d_st: mut Render3dState, scan: Model, cap: int, wind return l } -# A lupine spike (1 m tall): a stem of two crossed quads (uv.x in [0,1]) and -# seven tiers of crossed floret quads (uv.x in [1,2]), coloured in the shader. -function model_lupine(render3d_st: mut Render3dState) -> Model { - let model = new Model - model.prims = new []Prim - let pr = new Prim - let m = gpu_mesh_new(render3d_st) - # quads: stem x2 + tiers 12 x 2 + 3 leaves = 29 quads - let nq = 29 - let v = gl_floats(nq * 4 * 8) - let idx = words(nq * 6) - var k = 0 - var qi = 0 - for q in 0 .. nq { - var w = 0.012; var y0 = 0.0; var y1 = 0.62; var ukind = 0.0 - var ang = 0.0 - if q >= 2 and q < 26 { - let tier = (q - 2) / 2 - let t = float(tier) / 12.0 - w = 0.05 * (1.1 - t) - y0 = 0.27 + t * 0.36 - y1 = y0 + 0.045 - ukind = 1.0 - ang = float(tier) / 12.0 * 2.1 - if (q & 1) == 1 { ang = ang + PI * 0.5 } - } else if q >= 26 { - # a rosette of three leaves near the ground - w = 0.09; y0 = 0.02; y1 = 0.2; ukind = 2.0 - ang = float(q - 26) / 3.0 * (2.0 * PI) - } else { - if (q & 1) == 1 { ang = ang + PI * 0.5 } - } - let cx = Math.cos(ang) * w; let cz = Math.sin(ang) * w - for c in 0 .. 4 { - var sx = -1.0; var sy = y0; var u = 0.0 - if c == 1 or c == 2 { sx = 1.0; u = 1.0 } - if c == 2 or c == 3 { sy = y1 } - gl_put_bits(v, k, float_bits(cx * sx)); gl_put_bits(v, k + 1, float_bits(sy)); gl_put_bits(v, k + 2, float_bits(cz * sx)) - gl_put_bits(v, k + 3, float_bits(-cz)); gl_put_bits(v, k + 4, float_bits(0.2)); gl_put_bits(v, k + 5, float_bits(cx)) - gl_put_bits(v, k + 6, float_bits(ukind + u)) - var vv = sy - if ukind == 1.0 { vv = (sy - 0.27) / 0.4 } - if ukind == 2.0 { vv = (sy - 0.02) / 0.18 } - gl_put_bits(v, k + 7, float_bits(vv)) - k += 8 - } - let b = q * 4 - idx[qi] = b; idx[qi + 1] = b + 1; idx[qi + 2] = b + 2; idx[qi + 3] = b; idx[qi + 4] = b + 2; idx[qi + 5] = b + 3 - qi += 6 - } - gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(nq * 4 * 8), GPU_STATIC) - sc_model_layout(render3d_st, m) - free(v) - gpu_mesh_indices(render3d_st, m, data_of(idx), nq * 6 * 4, 4) - free(idx) - m.count = nq * 6 - gpu_mesh_done(render3d_st, m) - pr.mesh = m - if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) } - pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white - push(model.prims, pr) - model.radius = 0.08; model.height = 0.65; model.tris = nq * 2 - return model -} - -# A dense lupine for baking into a card: a stem, ~220 small floret quads in a -# tapering spiral (uv.x in [1,2]) and five leaves (uv.x in [2,3]). -function model_lupine_dense(render3d_st: mut Render3dState) -> Model { - let model = new Model - model.prims = new []Prim - let pr = new Prim - let m = gpu_mesh_new(render3d_st) - let nfl = 220 - let nq = 2 + nfl + 5 - let v = gl_floats(nq * 4 * 8) - let idx = words(nq * 6) - var k = 0 - var qi = 0 - seed(5) - for q in 0 .. nq { - var w = 0.008; var y0 = 0.0; var y1 = 0.66; var ukind = 0.0 - var ang = 0.0; var ox = 0.0; var oz = 0.0; var tilt = 0.0 - if q >= 2 and q < 2 + nfl { - let t = float(q - 2) / float(nfl) - let yy = 0.28 + t * 0.4 - ang = float(q) * 2.39996 # golden angle spiral - let rad = 0.055 * (1.05 - t) - ox = Math.cos(ang) * rad; oz = Math.sin(ang) * rad - w = 0.028 * (1.1 - t * 0.5) - y0 = yy - 0.016; y1 = yy + 0.016 - ukind = 1.0 - tilt = 0.6 - } else if q >= 2 + nfl { - w = 0.05; y0 = 0.03; y1 = 0.16; ukind = 2.0 - ang = float(q - 2 - nfl) / 5.0 * (2.0 * PI) - ox = Math.cos(ang) * 0.05; oz = Math.sin(ang) * 0.05 - } else { - if (q & 1) == 1 { ang = PI * 0.5 } - } - # the quad faces outward (its normal along the spiral radius), leaning out by `tilt` - let nx = Math.cos(ang); let nz = Math.sin(ang) - let tx = -nz; let tz = nx # tangent (quad width direction) - for c in 0 .. 4 { - var sx = -1.0; var sy = y0; var u = 0.0 - if c == 1 or c == 2 { sx = 1.0; u = 1.0 } - if c == 2 or c == 3 { sy = y1 } - var lean = 0.0 - if c == 2 or c == 3 { lean = tilt * w } - gl_put_bits(v, k, float_bits(ox + tx * (sx * w) + nx * lean)) - gl_put_bits(v, k + 1, float_bits(sy)) - gl_put_bits(v, k + 2, float_bits(oz + tz * (sx * w) + nz * lean)) - gl_put_bits(v, k + 3, float_bits(nx)); gl_put_bits(v, k + 4, float_bits(0.35)); gl_put_bits(v, k + 5, float_bits(nz)) - gl_put_bits(v, k + 6, float_bits(ukind + u)) - var vv = sy - if ukind == 1.0 { vv = (sy - 0.27) / 0.42 } - if ukind == 2.0 { vv = (sy - 0.03) / 0.19 } - gl_put_bits(v, k + 7, float_bits(vv)) - k += 8 - } - let b = q * 4 - idx[qi] = b; idx[qi + 1] = b + 1; idx[qi + 2] = b + 2; idx[qi + 3] = b; idx[qi + 4] = b + 2; idx[qi + 5] = b + 3 - qi += 6 - } - gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(nq * 4 * 8), GPU_STATIC) - sc_model_layout(render3d_st, m) - free(v) - gpu_mesh_indices(render3d_st, m, data_of(idx), nq * 6 * 4, 4) - free(idx) - m.count = nq * 6 - gpu_mesh_done(render3d_st, m) - pr.mesh = m - if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) } - pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white - push(model.prims, pr) - model.radius = 0.11; model.height = 0.68; model.tris = nq * 2 - return model -} - # A procedural grass blade (1 m tall, 5 cm wide, curved): 5 rows of 2 vertices. +@alloc_ok("a model: made once by the program that asks for it, and kept with its layer") function model_blade(render3d_st: mut Render3dState) -> Model { let model = new Model model.prims = new []Prim @@ -553,6 +416,7 @@ function scatter_begin_frame(render3d_st: mut Render3dState) -> void { # wrote - one draw per material covering every level. Nothing is partitioned or uploaded on the CPU # when the view moves. PC camp benchmark: 2791 -> 2657 draws, 4.3 -> 4.1 s for 400 frames. const SC_REC_W: int = 20 # a VkDrawIndexedIndirectCommand +const SC_LOD_ROOM: int = 64 # a layer's LODs plus near and far, most (sc_lod_* are made this size) const SC_RECS: int = 29 # 16 level x prim, 1 impostor, 12 shadow LOD (scatter_cull.comp) function layer_gpu_eligible(render3d_st: Render3dState, l: Layer) -> bool { @@ -661,7 +525,7 @@ function layer_gpu_prepare(render3d_st: mut Render3dState, l: Layer) -> bool { gpu_buffer_upload(render3d_st, l.g_dst, (n + 1) * cap * INST_FLOATS * 4, null, GPU_DYNAMIC) if l.g_arena == null { layer_arena_build(render3d_st, l) } let n_mat = len(l.g_arena) - let rec = words(SC_RECS * 5) + let rec = render3d_st.sc_gpu_rec for i in 0 .. SC_RECS * 5 { rec[i] = 0 } # material j, level k: that level's range of the merged mesh, its instances from bucket k for j in 0 .. n_mat { @@ -681,10 +545,9 @@ function layer_gpu_prepare(render3d_st: mut Render3dState, l: Layer) -> bool { } } gpu_buffer_upload(render3d_st, l.g_cmds, SC_RECS * SC_REC_W, data_of(rec), GPU_DYNAMIC) - let zeros = words(5) + let zeros = render3d_st.sc_gpu_zeros for i in 0 .. 5 { zeros[i] = 0 } gpu_buffer_upload(render3d_st, l.g_counts, 20, data_of(zeros), GPU_DYNAMIC) - free(rec); free(zeros) # the card casts every instance, as on the CPU path (layer_grid_build uploads this there) l.n_sh = l.count gpu_buffer_upload(render3d_st, l.sh_buf, l.count * INST_FLOATS * 4, data_of(l.inst), GPU_STATIC) @@ -695,7 +558,7 @@ function layer_gpu_prepare(render3d_st: mut Render3dState, l: Layer) -> bool { # the dispatch for the view as it stands: frustum, camera, distances, the layer's shape function layer_gpu_cull(render3d_st: mut Render3dState, l: Layer) -> void { - let pr = words(36) + let pr = render3d_st.sc_gpu_pr for i in 0 .. 36 { pr[i] = 0 } if render3d_st.cam_planes != null { for i in 0 .. 16 { pr[i] = float_bits(render3d_st.cam_planes[i]) } } pr[16] = float_bits(render3d_st.cam_pos[0]); pr[17] = float_bits(render3d_st.cam_pos[1]); pr[18] = float_bits(render3d_st.cam_pos[2]); pr[19] = float_bits(l.cull) @@ -703,10 +566,9 @@ function layer_gpu_cull(render3d_st: mut Render3dState, l: Layer) -> void { pr[28] = l.count; pr[29] = l.count; pr[30] = l.n_lods; pr[31] = 1 # as layer_grid_gather pads a cell: the tallest instance, plus a margin pr[32] = float_bits(l.lods[0].height * 2.0); pr[33] = float_bits(4.0) - let bufs = words(4) + let bufs = render3d_st.sc_gpu_bufs bufs[0] = l.g_src; bufs[1] = l.g_dst; bufs[2] = l.g_cmds; bufs[3] = l.g_counts gpu_dispatch(render3d_st, render3d_st.sc_cull_prog, data_of(pr), 144, bufs, 1) - free(pr); free(bufs) } # Sort a static layer's instances into square cells (call once, after placement; a @@ -789,7 +651,8 @@ function layer_grid_gather(render3d_st: Render3dState, l: Layer) -> void { # the impostor bucket last, and upload one buffer per level. function layer_partition_lods(render3d_st: mut Render3dState, l: Layer, src: floats, total: int) -> void { let n = l.n_lods - let counts = words(n + 2) + if n + 2 > SC_LOD_ROOM { return } # past the room made with the state + let counts = render3d_st.sc_lod_counts for k in 0 .. n + 2 { counts[k] = 0 } let cull2 = l.cull * l.cull let open = l.lod_dist[n - 1] == 0.0 # the last level runs out to the cull distance @@ -810,10 +673,10 @@ function layer_partition_lods(render3d_st: mut Render3dState, l: Layer, src: flo counts[lv] += 1 } # prefix offsets (in instances) per bucket - let start = words(n + 2) + let start = render3d_st.sc_lod_start var acc = 0 for k in 0 .. n + 2 { start[k] = acc; acc += counts[k] } - let fill = words(n + 2) + let fill = render3d_st.sc_lod_fill for k in 0 .. n + 2 { fill[k] = start[k] } let tmp = l.scratch for i in 0 .. total { @@ -841,7 +704,6 @@ function layer_partition_lods(render3d_st: mut Render3dState, l: Layer, src: flo l.n_sh = total if total > 0 { gpu_buffer_upload(render3d_st, l.sh_buf, total * INST_FLOATS * 4, data_of(src), GPU_DYNAMIC) } } - free(counts); free(start); free(fill) } # split the instances by distance to the camera (only when the view changed) diff --git a/packages/ludic.render3d/shadow.ludic b/packages/ludic.render3d/shadow.ludic index ad81e970..8c1e2d92 100644 --- a/packages/ludic.render3d/shadow.ludic +++ b/packages/ludic.render3d/shadow.ludic @@ -80,8 +80,10 @@ function shadow_fit(render3d_st: mut Render3dState, c: int, near: float, far: fl # map resampled every frame: that is the crawl and flicker seen while moving. m4_perspective(render3d_st.sh_tmp_proj, render3d_st.cam_fov, render3d_st.cam_aspect, near, far) m4_inverse(render3d_st.sh_tmp_inv, render3d_st.sh_tmp_proj) # NDC -> view space - let cview = v3_new(0.0, 0.0, 0.0) - let corners = floats(24) + # the fit's scratch is the state's, made once (a cascade fitted every frame made six of them) + let cview = render3d_st.sh_fit_cview + v3_set(cview, 0.0, 0.0, 0.0) + let corners = render3d_st.sh_fit_corners for i in 0 .. 8 { var x = -1.0; var y = -1.0; var z = -1.0 if (i & 1) != 0 { x = 1.0 } @@ -102,16 +104,16 @@ function shadow_fit(render3d_st: mut Render3dState, c: int, near: float, far: fl radius = radius * 1.05 # the slice centre back into world space m4_inverse(render3d_st.sh_tmp_vp, render3d_st.cam_view) - let center = floats(3) + let center = render3d_st.sh_fit_center m4_xform_point(center, render3d_st.sh_tmp_vp, cview[0], cview[1], cview[2]) - free(cview) # light view: from far along the sun direction, looking at the centre - let eye = floats(3) + let eye = render3d_st.sh_fit_eye # casters up to ~900 m toward the sun (a mountain across the valley), and the # slice itself behind the centre: a tight depth range keeps the bias small let back = radius + 900.0 v3_madd(eye, center, render3d_st.sun_dir, back) - let up = v3_new(0.0, 1.0, 0.0) + let up = render3d_st.sh_fit_up + v3_set(up, 0.0, 1.0, 0.0) m4_look_at(render3d_st.sh_tmp_view, eye, center, up) # snap the ortho window to the shadow texel grid let texel = radius * 2.0 / float(render3d_st.shadow_res) @@ -123,10 +125,9 @@ function shadow_fit(render3d_st: mut Render3dState, c: int, near: float, far: fl m4_ortho(render3d_st.sh_tmp_proj, nr + ox, radius + ox, nr + oy, radius + oy, 1.0, zfar) render3d_st.sh_range[c] = zfar - 1.0 render3d_st.sh_texel[c] = texel - let out = floats(16) + let out = render3d_st.sh_fit_out m4_mul(out, render3d_st.sh_tmp_proj, render3d_st.sh_tmp_view) for i in 0 .. 16 { render3d_st.sh_vp[c * 16 + i] = out[i] } - free(out); free(eye); free(up); free(center); free(corners) } function shadow_cascade_vp(render3d_st: Render3dState, c: int) -> floats { return render3d_st.sh_vp_v[c] } diff --git a/packages/ludic.render3d/skin.ludic b/packages/ludic.render3d/skin.ludic index da91ee1d..470ed845 100644 --- a/packages/ludic.render3d/skin.ludic +++ b/packages/ludic.render3d/skin.ludic @@ -214,6 +214,7 @@ function skin_bind(render3d_st: mut Render3dState, sk: Skin, prog: int) -> void # the same skeleton posed on its own: shares the rest data, owns the pose and the matrices @creates(SkinClone) +@alloc_ok("one per figure placed (a person, an animal): actor_release frees it with the actor") function skin_clone(src: Skin) -> Skin { let sk = new Skin sk.n_nodes = src.n_nodes; sk.par = src.par; sk.walk = src.walk diff --git a/packages/ludic.render3d/stream.ludic b/packages/ludic.render3d/stream.ludic index 9e3cba36..99f8e42f 100644 --- a/packages/ludic.render3d/stream.ludic +++ b/packages/ludic.render3d/stream.ludic @@ -193,6 +193,7 @@ function stream_evict(render3d_st: mut Render3dState, s: Stream) -> void { w += 1 } else { c.off = -1 + @alloc_ok("within the stream's record pool, all made with the stream (STREAM_MAX_CHUNKS)") push(s.spare, c) } i += 1 @@ -263,6 +264,7 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float, if need > 0 { mem_copy(mem_off(data_of(s.arena), s.top * 4), data_of(render3d_st.stream_scratch), need * 4) } c.off = s.top s.top += need + @alloc_ok("within the stream's record pool, all made with the stream (STREAM_MAX_CHUNKS)") push(s.chunks, c); s.keys[s.n] = key; stream_remember(s, key, s.n); s.n += 1 c.used = render3d_st.stream_walk_no } else { loose = true } @@ -287,6 +289,7 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float, l.count += c.count render3d_st.stream_us_gather = render3d_st.stream_us_gather + (gl_now_us() - tg) } + @alloc_ok("within the stream's record pool, all made with the stream (STREAM_MAX_CHUNKS)") if loose and c != render3d_st.stream_loose { push(s.spare, c) } } } diff --git a/packages/ludic.render3d/texture.ludic b/packages/ludic.render3d/texture.ludic index 8d21cfb3..e7a7f60c 100644 --- a/packages/ludic.render3d/texture.ludic +++ b/packages/ludic.render3d/texture.ludic @@ -36,6 +36,7 @@ function r3d_set_anisotropy(render3d_st: mut Render3dState, level: int) -> void if keep > 0 { gpu_tex_bind(render3d_st, GPU_TEX2D, keep) } } +@alloc_ok("a load: a file read whole, kept or freed by its caller") function r3d_read_file(render3d_st: mut Render3dState, path: pointer) -> pointer { let f = file_open(path, "rb") if f == null { return null } @@ -483,6 +484,7 @@ function tex_dump(render3d_st: mut Render3dState, tex: int, w: int, h: int, path # A binary PPM (P6, what Gl.screenshot writes) as an RGB8 texture, box-filtered down by # `shrink` (a photo thumbnail); 0 when the file is missing. +@alloc_ok("a load: an image read from disk") function tex_load_ppm(render3d_st: mut Render3dState, path: pointer, shrink: int) -> int { let d = r3d_read_file(render3d_st, path) if d == null { return 0 } diff --git a/packages/ludic.render3d/water.ludic b/packages/ludic.render3d/water.ludic index 2cbda7f8..8aa2a41b 100644 --- a/packages/ludic.render3d/water.ludic +++ b/packages/ludic.render3d/water.ludic @@ -36,15 +36,16 @@ function water_reflection_pass(render3d_st: mut Render3dState) -> void { # the mirrored camera: view' = view * R, R reflecting y about the surface (y' = 2L - y). # R has determinant -1, so the winding flips (front faces culled below) and the image # lands exactly where the main camera's pixels expect the reflection. - let eye = v3_new(sx, 2.0 * render3d_st.water_level - sy, sz) - let refl = m4_new() + # the pass's scratch is the state's, made once + let eye = render3d_st.water_eye + v3_set(eye, sx, 2.0 * render3d_st.water_level - sy, sz) + let refl = render3d_st.water_mirror + m4_identity(refl) refl[5] = -1.0 refl[13] = 2.0 * render3d_st.water_level - let mv = floats(16) + let mv = render3d_st.water_mv m4_mul(mv, render3d_st.water_saved, refl) m4_copy(render3d_st.cam_view, mv) - free(mv); free(refl) - let fwd = words(3); let up = words(3); let at = words(3) m4_mul(render3d_st.cam_vp, render3d_st.cam_proj, render3d_st.cam_view) m4_inverse(render3d_st.cam_inv_vp, render3d_st.cam_vp) v3_copy(render3d_st.cam_pos, eye) @@ -79,7 +80,6 @@ function water_reflection_pass(render3d_st: mut Render3dState) -> void { m4_copy(render3d_st.cam_vp, render3d_st.water_saved_vp) m4_copy(render3d_st.cam_inv_vp, render3d_st.water_saved_ivp) v3_set(render3d_st.cam_pos, sx, sy, sz) - free(eye); free(fwd); free(up); free(at) gpu_fb_bind(render3d_st, 0) } diff --git a/runtime/native/value.ludic b/runtime/native/value.ludic index 86a2d2bf..2140d677 100644 --- a/runtime/native/value.ludic +++ b/runtime/native/value.ludic @@ -96,9 +96,11 @@ function value_set_float(o: Val, key: pointer, x: float) -> void { let v = value_slot(o, key, 7) value_num_set(v, float_bits(x)) } +# a component's text held interned: one copy per distinct text, so the caller's own (a template built +# every frame) is given back with its frame rather than kept by the model function value_set_str(o: Val, key: pointer, s: pointer) -> void { let v = value_slot(o, key, 4) - v.txt = s + v.txt = intern(s) } function value_set_bool(o: Val, key: pointer, b: bool) -> void { let v = value_slot(o, key, 3) @@ -142,7 +144,7 @@ function value_set_strs(o: Val, key: pointer, xs: []string) -> void { let l = value_list_fit(o, key, len(xs)) for i in 0 .. len(xs) { let v = value_item(l, i, 4) - v.txt = xs[i] + v.txt = intern(xs[i]) } } function value_set_bools(o: Val, key: pointer, xs: []bool) -> void { @@ -197,7 +199,7 @@ function value_into_float(into: Val, x: float) -> Val { } function value_into_str(into: Val, s: pointer) -> Val { let v = value_into(into, 4) - v.txt = s + v.txt = intern(s) return v } function value_into_bool(into: Val, b: bool) -> Val { @@ -235,7 +237,7 @@ function value_into_strs(into: Val, xs: []string) -> Val { let l = value_into_list(into, len(xs)) for i in 0 .. len(xs) { let v = value_item(l, i, 4) - v.txt = xs[i] + v.txt = intern(xs[i]) } return l }