Merge branch 'r3d/no-cmd-pool' into lang/leaks2

This commit is contained in:
Orkun ÇAKILKAYA 2026-09-28 13:41:23 +03:00
commit f8fe157352
13 changed files with 30721 additions and 30490 deletions

View file

@ -0,0 +1,10 @@
bump: patch
type: fix
**A frame that draws more than any before no longer grows the heap on the Mac.** MoltenVK's command
pooling kept every command object a frame had ever recorded - about 650 bytes for each draw beyond the
busiest frame so far, for as long as the game ran. render3d turns it off
(`MVK_CONFIG_USE_COMMAND_POOLING=0`, unless the environment already says otherwise) before the first
Vulkan call; the objects are made and freed with their command buffer, at no measured cost.
`examples/rendering/steady.ludic` ramps a frame from 20 to 200 actors and fails on what pooling left.
A buffer written through its mapping is also read once at the start of the next frame's commands,
so MoltenVK makes its Metal buffer then rather than at its first draw, however much later that is.

View file

@ -0,0 +1,5 @@
bump: patch
type: fix
**`Os.platform()` and `Os.arch()` allocate nothing after the first call.** Each call malloc'd an
8 KB `uname` buffer and let it go, and a game asks the platform every frame in places (a launcher's
wait, an update panel, a renderer's present). The buffer is made once and kept.

9
changes/stream-arena.md Normal file
View file

@ -0,0 +1,9 @@
bump: patch
type: fix
**Streaming new ground allocates nothing.** A streamed layer's cache made a record and a copy of
its instances for every chunk the camera found, so the heap grew for as long as there was ground
nobody had stood near - a boat drifting across the lake grew it every frame it crossed a cell. Each
stream now makes its whole cache when it is created: a record per chunk it can keep and an arena of
twice what its layer draws, compacted in place when it evicts. Replacing the world
(`stream_clear_all`) now gives every record, list and stream back. `examples/rendering/steady.ludic`
crosses 300 new cells with eviction running: 156 KB before, 0 after, every kept chunk still its own.

View file

@ -12,6 +12,7 @@ program StringTemps {
if a + "/" + b + "/" + string(i) != `{a}/{b}/{i}` { bad += 1 } if a + "/" + b + "/" + string(i) != `{a}/{b}/{i}` { bad += 1 }
let p = "lake/camp.png" # a literal: the slices are what is made here let p = "lake/camp.png" # a literal: the slices are what is made here
if p[len(p) - 4 .. len(p)] != ".png" or p[0 .. len(p) - 4] + ".dds" != `lake/{b[0 .. 0]}camp.dds` { bad += 1 } if p[len(p) - 4 .. len(p)] != ".png" or p[0 .. len(p) - 4] + ".dds" != `lake/{b[0 .. 0]}camp.dds` { bad += 1 }
if Os.platform() == "" or Os.arch() == "" { bad += 1 } # asked every round: nothing each time
if `{i}-{i * 2}-{long(i) * 1000000000}` != string(i) + "-" + string(i * 2) + "-" + string(long(i) * 1000000000) { bad += 1 } if `{i}-{i * 2}-{long(i) * 1000000000}` != string(i) + "-" + string(i * 2) + "-" + string(long(i) * 1000000000) { bad += 1 }
return bad return bad
} }

View file

@ -11,9 +11,13 @@ program Steady {
property Marker { on: int = 1 } property Marker { on: int = 1 }
model Anchor { Marker } model Anchor { Marker }
function scene_draw(render3d_st: mut Render3dState) -> void { } function scene_draw(render3d_st: mut Render3dState) -> void { actor_draw(render3d_st) }
function scene_draw_casters(render3d_st: mut Render3dState, light_vp: floats) -> void { } function scene_draw_casters(render3d_st: mut Render3dState, light_vp: floats) -> void { }
function stream_fill(s: Stream, cx: int, cz: int, band: int) -> void { } # a chunk of 20 to 51 instances, different for every cell
function stream_fill(render3d_st: mut Render3dState, s: Stream, cx: int, cz: int, band: int) -> void {
let n = 20 + ((cx * 7 + cz * 13 + band) & 31)
for i in 0 .. n { stream_emit(render3d_st, s, float(cx) * 32.0 + float(i), 0.0, float(cz) * 32.0, 1.0, 0.0, 0.5, 0.0) }
}
# The heap once the device is idle and MoltenVK's completion handlers, which release a finished # The heap once the device is idle and MoltenVK's completion handlers, which release a finished
# command buffer on their own thread, have caught up: two reads 20 ms apart that agree. Read at # command buffer on their own thread, have caught up: two reads 20 ms apart that agree. Read at
@ -70,6 +74,58 @@ program Steady {
return settled(render3d_st) - before return settled(render3d_st) - before
} }
# bytes gained while a frame draws more and more: 20 actors to 200, 30 frames at each count.
# MoltenVK's command pooling kept every command a frame had recorded (~650 bytes a draw more).
function ramp_rounds(render3d_st: mut Render3dState, m: Model) -> long {
let acts = new []Actor
for i in 0 .. 200 {
let a = actor_new(render3d_st, m)
a.cull = 0.0
a.visible = false
actor_place(a, float(i % 20) - 10.0, 0.0, -float(i / 20), 0.0)
push(acts, a)
}
for i in 0 .. 20 { acts[i].visible = true }
frame_rounds(render3d_st, 30)
let before = settled(render3d_st)
for k in 2 .. 11 {
for i in 0 .. k * 20 { acts[i].visible = true }
frame_rounds(render3d_st, 30)
}
let grew = settled(render3d_st) - before
for i in 0 .. 200 { actor_release(render3d_st, acts[i]) }
return grew
}
# bytes gained while the camera crosses new ground for 300 cells: a streamed layer's cache holds
# at most 256 chunks, so it fills, evicts and compacts on the way. Each chunk once brought its
# own record and copy, made as the ground was found.
function stream_rounds(render3d_st: mut Render3dState, m: Model) -> long {
render3d_st.STREAM_MAX_CHUNKS = 256
let l = layer_new(render3d_st, m, 20000, false, 0.0, 0.0, 400.0)
let st = stream_new(render3d_st, l, 32.0, 96.0, 32.0, 64.0, 80.0, 96.0)
cam_set(render3d_st, 0.0, 3.0, 0.0, 0.0, -10.0)
frame_rounds(render3d_st, 10)
let before = settled(render3d_st)
for step in 1 .. 301 {
cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0)
frame_rounds(render3d_st, 2)
}
let grew = settled(render3d_st) - before
# every chunk kept, after all that evicting and compacting, still holds its own cell's instances
var wrong = 0
for i in 0 .. st.n {
let c = st.chunks[i]
let cx = c.key / 4 / 8192 - 4096
if c.count < 20 or float_from_bits(st.arena[c.off]) != float(cx) * 32.0 { wrong += 1 }
}
if wrong > 0 or st.n == 0 or l.count == 0 {
print(`steady: FAILED - {wrong} of {st.n} kept chunks hold another cell's instances ({l.count} drawn)`)
return 1000000
}
return grew
}
# bytes gained over n parses of a glTF document, each freed whole (Json.free_all): strings too # bytes gained over n parses of a glTF document, each freed whole (Json.free_all): strings too
function parse_rounds(render3d_st: Render3dState, text: string, n: int) -> long { function parse_rounds(render3d_st: Render3dState, text: string, n: int) -> long {
let before = settled(render3d_st) let before = settled(render3d_st)
@ -107,10 +163,12 @@ program Steady {
let am = gltf_load(render3d_st, "packages/ludic.lab/plate", "plate.gltf", "plate") let am = gltf_load(render3d_st, "packages/ludic.lab/plate", "plate.gltf", "plate")
actor_rounds(render3d_st, am, 20) actor_rounds(render3d_st, am, 20)
let grew_a = actor_rounds(render3d_st, am, 2000) let grew_a = actor_rounds(render3d_st, am, 2000)
let grew_r = ramp_rounds(render3d_st, am)
let grew_s = stream_rounds(render3d_st, am)
let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf") let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf")
parse_rounds(render3d_st, text, 20) parse_rounds(render3d_st, text, 20)
let grew_p = parse_rounds(render3d_st, text, 200) let grew_p = parse_rounds(render3d_st, text, 200)
print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000`) print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 300 new cells {grew_s}`)
# a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator) # a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator)
var ok = grew_b < 16384 var ok = grew_b < 16384
if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") } if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") }
@ -132,6 +190,14 @@ program Steady {
ok = false ok = false
print("steady: FAILED - an actor placed and released leaves memory behind") print("steady: FAILED - an actor placed and released leaves memory behind")
} }
if grew_r >= 16384 {
ok = false
print("steady: FAILED - a frame drawing more than before leaves memory behind")
}
if grew_s >= 4096 {
ok = false
print("steady: FAILED - streaming new ground leaves memory behind")
}
if ok { print("STEADY OK") } else { print("STEADY FAILED") } if ok { print("STEADY OK") } else { print("STEADY FAILED") }
quit() quit()
} }

View file

@ -691,6 +691,12 @@ export state Render3dState {
stream_walks: int = 0 # streams that walked their whole ring this frame stream_walks: int = 0 # streams that walked their whole ring this frame
stream_debug_n: int = 0 stream_debug_n: int = 0
stream_scratch: floats = null stream_scratch: floats = null
gvk_prime_b: words = null # buffers made since the frame began, read once so MoltenVK makes their
gvk_prime_h: []long = null # Metal buffers now (gvk_prime), and their handles
gvk_prime_n: int = 0
gvk_prime_dst: long = 0 # the 64 bytes those reads land in
gvk_prime_mem: long = 0
stream_loose: Chunk = null # the record for a chunk no stream's cache can keep this walk
stream_deadline: long = 0 stream_deadline: long = 0
gsl_on: bool = false # slInit succeeded: the interposer is the loader, the plugins are in gsl_on: bool = false # slInit succeeded: the interposer is the loader, the plugins are in
gsl_dlss_ok: bool = false # what this adapter can run, from slIsFeatureSupported gsl_dlss_ok: bool = false # what this adapter can run, from slIsFeatureSupported

View file

@ -48,8 +48,18 @@ function gvk_surface_ext() -> string {
return VK_KHR_WIN32_SURFACE_EXTENSION_NAME return VK_KHR_WIN32_SURFACE_EXTENSION_NAME
} }
# MoltenVK's command pooling keeps every command object a frame ever recorded and never gives one
# back, so the heap grew each time a frame drew more than any before (~650 bytes a draw, for as long
# as the game ran). Off, the objects are made and freed with their command buffer, at no measured
# cost (3.7 ms either way over 520 actors). Read when the library loads, so before any Vk call.
function gvk_no_command_pooling() -> void {
if Os.platform() != "macos" or Os.has_env("MVK_CONFIG_USE_COMMAND_POOLING") { return }
Os.set_env("MVK_CONFIG_USE_COMMAND_POOLING", "0")
}
function gvk_init(render3d_st: mut Render3dState) -> bool { function gvk_init(render3d_st: mut Render3dState) -> bool {
if render3d_st.gvk_ready { return true } if render3d_st.gvk_ready { return true }
gvk_no_command_pooling()
gsl_boot(render3d_st) gsl_boot(render3d_st)
if Vk.open() == 0 { render3d_st.gvk_why = "no Vulkan loader"; return false } if Vk.open() == 0 { render3d_st.gvk_why = "no Vulkan loader"; return false }
gsl_init(render3d_st) gsl_init(render3d_st)
@ -638,6 +648,7 @@ function gvk_shutdown(render3d_st: mut Render3dState) -> void {
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_fence, 0), null) Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_fence, 0), null)
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_frame_fence, 0), null) Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_frame_fence, 0), null)
Vk.destroy_command_pool(render3d_st.gvk_dev, render3d_st.gvk_pool, null) Vk.destroy_command_pool(render3d_st.gvk_dev, render3d_st.gvk_pool, null)
if render3d_st.gvk_prime_dst != 0 { Vk.destroy_buffer(render3d_st.gvk_dev, render3d_st.gvk_prime_dst, null); render3d_st.gvk_prime_dst = 0 }
Vk.destroy_device(render3d_st.gvk_dev, null) Vk.destroy_device(render3d_st.gvk_dev, null)
Vk.destroy_instance(render3d_st.gvk_inst, null) Vk.destroy_instance(render3d_st.gvk_inst, null)
render3d_st.gvk_ready = false render3d_st.gvk_ready = false

View file

@ -1013,6 +1013,7 @@ function gvk_frame_cb(render3d_st: mut Render3dState) -> pointer {
if render3d_st.gvk_frame_pending and (render3d_st.gvk_frame_pending_no & 1) == (render3d_st.gvk_frame_no & 1) { gvk_frame_wait(render3d_st) } if render3d_st.gvk_frame_pending and (render3d_st.gvk_frame_pending_no & 1) == (render3d_st.gvk_frame_no & 1) { gvk_frame_wait(render3d_st) }
gvk_frame_reset(render3d_st) gvk_frame_reset(render3d_st)
render3d_st.gvk_cb = gvk_once_begin(render3d_st) render3d_st.gvk_cb = gvk_once_begin(render3d_st)
gvk_prime_flush(render3d_st, render3d_st.gvk_cb)
} }
return render3d_st.gvk_cb return render3d_st.gvk_cb
} }

View file

@ -753,6 +753,68 @@ function gvk_buf_reserve(render3d_st: mut Render3dState, b: int, n: int) -> bool
r = Vk.bind_buffer_memory(render3d_st.gvk_dev, buf, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma)) r = Vk.bind_buffer_memory(render3d_st.gvk_dev, buf, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma))
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkBindBufferMemory", r) } if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkBindBufferMemory", r) }
render3d_st.gvk_buf[b] = buf; render3d_st.gvk_buf_mem[b] = mem; render3d_st.gvk_buf_size[b] = size; render3d_st.gvk_buf_map[b] = gvk_mem_ptr(render3d_st, ma) render3d_st.gvk_buf[b] = buf; render3d_st.gvk_buf_mem[b] = mem; render3d_st.gvk_buf_size[b] = size; render3d_st.gvk_buf_map[b] = gvk_mem_ptr(render3d_st, ma)
gvk_prime(render3d_st, b, buf)
return true
}
# MoltenVK makes a buffer's Metal buffer the first time a command uses it, so one written through
# its mapping and first drawn a thousand frames later allocated then, in play. Each new buffer is
# read once (4 bytes, copied out) at the start of the next frame's commands instead.
const GVK_PRIME_MAX: int = 16384
function gvk_prime(render3d_st: mut Render3dState, b: int, buf: long) -> void {
if Os.platform() != "macos" { return }
if render3d_st.gvk_prime_b == null {
render3d_st.gvk_prime_b = words(GVK_PRIME_MAX)
render3d_st.gvk_prime_h = new []long
let hs = render3d_st.gvk_prime_h
let none: long = 0
for i in 0 .. GVK_PRIME_MAX { push(hs, none) }
}
if render3d_st.gvk_prime_n >= GVK_PRIME_MAX { return }
render3d_st.gvk_prime_b[render3d_st.gvk_prime_n] = b
render3d_st.gvk_prime_h[render3d_st.gvk_prime_n] = buf
render3d_st.gvk_prime_n += 1
}
function gvk_prime_flush(render3d_st: mut Render3dState, cb: pointer) -> void {
if render3d_st.gvk_prime_n == 0 { return }
if render3d_st.gvk_prime_dst == 0 and not gvk_prime_target(render3d_st) {
render3d_st.gvk_prime_n = 0
return
}
let region = gvk_tmp(render3d_st, VkBufferCopy_sizeof)
Vk.zero(region, VkBufferCopy_sizeof)
let four: long = 4
Vk.put_i64(region, VkBufferCopy_size, four)
for i in 0 .. render3d_st.gvk_prime_n {
let b = render3d_st.gvk_prime_b[i]
let h = render3d_st.gvk_prime_h[i]
# released and made again, or gone, since: that one is not this buffer any more
if b > 0 and b < len(render3d_st.gvk_buf) and render3d_st.gvk_buf[b] == h { Vk.cmd_copy_buffer(cb, h, render3d_st.gvk_prime_dst, 1, region) }
}
render3d_st.gvk_prime_n = 0
}
function gvk_prime_target(render3d_st: mut Render3dState) -> bool {
let size: long = 64
let bci = gvk_tmp(render3d_st, VkBufferCreateInfo_sizeof)
Vk.zero(bci, VkBufferCreateInfo_sizeof)
Vk.put_i32(bci, VkBufferCreateInfo_sType, VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO)
Vk.put_i64(bci, VkBufferCreateInfo_size, size)
Vk.put_i32(bci, VkBufferCreateInfo_usage, VK_BUFFER_USAGE_TRANSFER_DST_BIT)
Vk.put_i32(bci, VkBufferCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE)
let out = gvk_tmp(render3d_st, 8)
if Vk.create_buffer(render3d_st.gvk_dev, bci, null, out) != VK_SUCCESS { return false }
let buf = gvk_handle(out)
let req = gvk_tmp(render3d_st, VkMemoryRequirements_sizeof)
Vk.get_buffer_memory_requirements(render3d_st.gvk_dev, buf, req)
let ma = gvk_mem_new(render3d_st, req, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, false)
if ma == 0 {
Vk.destroy_buffer(render3d_st.gvk_dev, buf, null)
return false
}
Vk.bind_buffer_memory(render3d_st.gvk_dev, buf, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma))
render3d_st.gvk_prime_dst = buf
let m: long = ma
render3d_st.gvk_prime_mem = m
return true return true
} }

View file

@ -16,7 +16,7 @@
property Chunk { property Chunk {
key: int = 0, # packed (cx, cz, band) key: int = 0, # packed (cx, cz, band)
used: int = 0, # the walk that last wanted it (for eviction) used: int = 0, # the walk that last wanted it (for eviction)
data: words, # INST_FLOATS per instance off: int = -1, # where its instances start in the stream's arena (words); -1 not kept
count: int = 0, count: int = 0,
ymin: float = 0.0, # height range of its instances (float bits), for the frustum test ymin: float = 0.0, # height range of its instances (float bits), for the frustum test
ymax: float = 0.0 ymax: float = 0.0
@ -36,7 +36,10 @@ property Stream {
kind: int = 0, # the scene's generator selector for this stream kind: int = 0, # the scene's generator selector for this stream
min_band: int = 0, # bands below this belong to another (nearer) stream min_band: int = 0, # bands below this belong to another (nearer) stream
view_gen: int = -1, # sc_view_gen the layer was last gathered for (the view turned -> regather) view_gen: int = -1, # sc_view_gen the layer was last gathered for (the view turned -> regather)
htab: words # open-addressed key -> chunk index + 1 (0 = empty) htab: words, # open-addressed key -> chunk index + 1 (0 = empty)
arena: words, # every kept chunk's instances, in the order of `chunks`
top: int = 0, # words of the arena in use
spare: []Chunk # records not in use: all of them are made with the stream
} }
# microseconds spent per frame, split so the hitch can be attributed (R3D_PROF=1) # microseconds spent per frame, split so the hitch can be attributed (R3D_PROF=1)
@ -47,13 +50,24 @@ function stream_new(render3d_st: mut Render3dState, layer: Layer, size: float, r
if r3d_env_has(render3d_st, "R3D_STREAM_CAP") { render3d_st.STREAM_MAX_CHUNKS = Text.to_int(r3d_env(render3d_st, "R3D_STREAM_CAP")) } if r3d_env_has(render3d_st, "R3D_STREAM_CAP") { render3d_st.STREAM_MAX_CHUNKS = Text.to_int(r3d_env(render3d_st, "R3D_STREAM_CAP")) }
render3d_st.stream_no_evict = r3d_env_has(render3d_st, "R3D_NOEVICT") render3d_st.stream_no_evict = r3d_env_has(render3d_st, "R3D_NOEVICT")
} }
# the fill's scratch and the record for a chunk the cache cannot keep, both once for every stream
if render3d_st.stream_scratch == null { render3d_st.stream_scratch = floats(STREAM_CHUNK_MAX * INST_FLOATS) }
if render3d_st.stream_loose == null { render3d_st.stream_loose = new Chunk }
let s = new Stream let s = new Stream
s.layer = layer; s.size = size; s.reach = reach s.layer = layer; s.size = size; s.reach = reach
layer.streamed = true layer.streamed = true
layer.grounded = true layer.grounded = true
s.bands = floats(4) s.bands = floats(4)
s.bands[0] = b0; s.bands[1] = b1; s.bands[2] = b2; s.bands[3] = b3 s.bands[0] = b0; s.bands[1] = b1; s.bands[2] = b2; s.bands[3] = b3
# Everything the cache will ever hold is made here: a record per chunk it can keep and an arena
# of twice what the layer draws. Each chunk once had its own record and copy, made as the camera
# found new ground, so the heap grew for as long as there was ground nobody had stood near.
s.chunks = new []Chunk s.chunks = new []Chunk
s.spare = new []Chunk
let live = s.chunks
for i in 0 .. render3d_st.STREAM_MAX_CHUNKS { push(live, new Chunk) }
while len(live) > 0 { push(s.spare, List.pop(live)) }
s.arena = words(Math.max(layer.cap, 4096) * 2 * INST_FLOATS)
s.keys = words(render3d_st.STREAM_MAX_CHUNKS) s.keys = words(render3d_st.STREAM_MAX_CHUNKS)
s.htab = words(STREAM_HASH) s.htab = words(STREAM_HASH)
for i in 0 .. STREAM_HASH { s.htab[i] = 0 } for i in 0 .. STREAM_HASH { s.htab[i] = 0 }
@ -100,7 +114,6 @@ const STREAM_CHUNK_MAX: int = 262144
function stream_emit(render3d_st: mut Render3dState, s: Stream, x: float, y: float, z: float, scale: float, yaw: float, seed: float, wind: float) -> void { function stream_emit(render3d_st: mut Render3dState, s: Stream, x: float, y: float, z: float, scale: float, yaw: float, seed: float, wind: float) -> void {
let c = s.cur let c = s.cur
if c.count >= STREAM_CHUNK_MAX { return } if c.count >= STREAM_CHUNK_MAX { return }
if render3d_st.stream_scratch == null { render3d_st.stream_scratch = floats(STREAM_CHUNK_MAX * INST_FLOATS) }
if c.count == 0 { c.ymin = y; c.ymax = y } else { c.ymin = Math.min(c.ymin, y); c.ymax = Math.max(c.ymax, y) } if c.count == 0 { c.ymin = y; c.ymax = y } else { c.ymin = Math.min(c.ymin, y); c.ymax = Math.max(c.ymax, y) }
let o = c.count * INST_FLOATS let o = c.count * INST_FLOATS
render3d_st.stream_scratch[o] = x; render3d_st.stream_scratch[o + 1] = y; render3d_st.stream_scratch[o + 2] = z; render3d_st.stream_scratch[o + 3] = scale render3d_st.stream_scratch[o] = x; render3d_st.stream_scratch[o + 1] = y; render3d_st.stream_scratch[o + 2] = z; render3d_st.stream_scratch[o + 3] = scale
@ -164,19 +177,26 @@ function stream_evict(render3d_st: mut Render3dState, s: Stream) -> void {
# everything wanted by the walk in progress stays whatever the threshold says # everything wanted by the walk in progress stays whatever the threshold says
# compacted in place, and an evicted chunk goes whole: a new list per eviction and the chunks' # compacted in place, and an evicted chunk goes whole: a new list per eviction and the chunks'
# own records were never given back # own records were never given back
# the arena is compacted in the same pass: a kept chunk's instances only ever move down
var w = 0 var w = 0
var top = 0
var i = 0 var i = 0
while i < s.n { while i < s.n {
let c = s.chunks[i] let c = s.chunks[i]
if c.used >= t or c.used == render3d_st.stream_walk_no { if c.used >= t or c.used == render3d_st.stream_walk_no {
let n = c.count * INST_FLOATS
if c.off != top { for k in 0 .. n { s.arena[top + k] = s.arena[c.off + k] } }
c.off = top
top += n
s.chunks[w] = c s.chunks[w] = c
w += 1 w += 1
} else { } else {
if c.data != null { free(c.data) } c.off = -1
free(c) push(s.spare, c)
} }
i += 1 i += 1
} }
s.top = top
let ch = s.chunks let ch = s.chunks
while len(ch) > w { List.pop(ch) } while len(ch) > w { List.pop(ch) }
s.n = w s.n = w
@ -221,9 +241,12 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float,
# The cell underfoot and its neighbours are never deferred: they are what you # The cell underfoot and its neighbours are never deferred: they are what you
# are looking at, and a hole there is the grass vanishing as you walk into it. # are looking at, and a hole there is the grass vanishing as you walk into it.
let urgent = band == 0 and ring <= 1 let urgent = band == 0 and ring <= 1
var loose = false
if c == null and (first or urgent or gl_now_us() < render3d_st.stream_deadline) { if c == null and (first or urgent or gl_now_us() < render3d_st.stream_deadline) {
c = new Chunk if len(s.spare) == 0 and not render3d_st.stream_no_evict { stream_evict(render3d_st, s) }
c.key = key let sp = s.spare
if len(sp) > 0 { c = List.pop(sp) } else { c = render3d_st.stream_loose }
c.key = key; c.count = 0; c.off = -1; c.used = 0
s.cur = c s.cur = c
let t0 = gl_now_us() let t0 = gl_now_us()
r3d_stream_fill(render3d_st, s, cx, cz, band) r3d_stream_fill(render3d_st, s, cx, cz, band)
@ -231,15 +254,17 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float,
render3d_st.stream_us_gen = render3d_st.stream_us_gen + dt render3d_st.stream_us_gen = render3d_st.stream_us_gen + dt
prof_chunk(render3d_st, s.kind, band, c.count, dt) prof_chunk(render3d_st, s.kind, band, c.count, dt)
if render3d_st.r3d_debug and band == 0 and render3d_st.stream_debug_n < 40 { render3d_st.stream_debug_n += 1; print(`stream kind {s.kind} band {band} chunk {cx},{cz}: {c.count} instances`) } if render3d_st.r3d_debug and band == 0 and render3d_st.stream_debug_n < 40 { render3d_st.stream_debug_n += 1; print(`stream kind {s.kind} band {band} chunk {cx},{cz}: {c.count} instances`) }
if c.count > 0 { c.data = words(c.count * INST_FLOATS); mem_copy(c.data, render3d_st.stream_scratch, c.count * INST_FLOATS * 4) } let need = c.count * INST_FLOATS
if s.n >= render3d_st.STREAM_MAX_CHUNKS and not render3d_st.stream_no_evict { stream_evict(render3d_st, s) } if s.top + need > len(s.arena) and not render3d_st.stream_no_evict { stream_evict(render3d_st, s) }
# If the walk in progress wants more chunks than the cache can hold, there # If the walk in progress wants more than the cache can hold, there is nothing to
# is nothing to evict and this one is used and dropped, as every chunk used # evict and this one is drawn from the scratch and dropped, as every chunk used to be.
# to be. The cap has to exceed one walk's ring for the cache to work at all. if c != render3d_st.stream_loose and s.n < render3d_st.STREAM_MAX_CHUNKS and s.top + need <= len(s.arena) {
if s.n < render3d_st.STREAM_MAX_CHUNKS { if need > 0 { mem_copy(mem_off(data_of(s.arena), s.top * 4), data_of(render3d_st.stream_scratch), need * 4) }
c.off = s.top
s.top += need
push(s.chunks, c); s.keys[s.n] = key; stream_remember(s, key, s.n); s.n += 1 push(s.chunks, c); s.keys[s.n] = key; stream_remember(s, key, s.n); s.n += 1
c.used = render3d_st.stream_walk_no c.used = render3d_st.stream_walk_no
} } else { loose = true }
prof_gen_add(render3d_st, c.count + 512) prof_gen_add(render3d_st, c.count + 512)
} }
@ -255,10 +280,13 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float,
if c != null and c.count > 0 and l.count + c.count <= l.cap and stream_chunk_visible(render3d_st, s, cx, cz, c) { if c != null and c.count > 0 and l.count + c.count <= l.cap and stream_chunk_visible(render3d_st, s, cx, cz, c) {
let tg = gl_now_us() let tg = gl_now_us()
layer_room(l, l.count + c.count) layer_room(l, l.count + c.count)
mem_copy(mem_off(l.inst, l.count * INST_FLOATS * 4), data_of(c.data), c.count * INST_FLOATS * 4) var src = data_of(render3d_st.stream_scratch)
if c.off >= 0 { src = mem_off(data_of(s.arena), c.off * 4) }
mem_copy(mem_off(l.inst, l.count * INST_FLOATS * 4), src, c.count * INST_FLOATS * 4)
l.count += c.count l.count += c.count
render3d_st.stream_us_gather = render3d_st.stream_us_gather + (gl_now_us() - tg) render3d_st.stream_us_gather = render3d_st.stream_us_gather + (gl_now_us() - tg)
} }
if loose and c != render3d_st.stream_loose { push(s.spare, c) }
} }
} }
cx += 1 cx += 1
@ -313,10 +341,14 @@ function stream_clear_all(render3d_st: mut Render3dState) -> void {
if render3d_st.stream_all == null { return } if render3d_st.stream_all == null { return }
for i in 0 .. len(render3d_st.stream_all) { for i in 0 .. len(render3d_st.stream_all) {
let s = render3d_st.stream_all[i] let s = render3d_st.stream_all[i]
if s.chunks != null { for c in 0 .. len(s.chunks) { if s.chunks[c].data != null { free(s.chunks[c].data) } } } for c in 0 .. len(s.chunks) { free(s.chunks[c]) }
for c in 0 .. len(s.spare) { free(s.spare[c]) }
free(s.chunks); free(s.spare); free(s.arena)
if s.keys != null { free(s.keys) } if s.keys != null { free(s.keys) }
if s.htab != null { free(s.htab) } if s.htab != null { free(s.htab) }
if s.bands != null { free(s.bands) } if s.bands != null { free(s.bands) }
free(s)
} }
render3d_st.stream_all = null let all = render3d_st.stream_all
List.clear(all)
} }

View file

@ -235,9 +235,17 @@ function emit_os_prelude() -> void {
emith(" %d2 = getelementptr inbounds %LSlice, ptr %h, i32 0, i32 2\n store i32 %c, ptr %d2\n") emith(" %d2 = getelementptr inbounds %LSlice, ptr %h, i32 0, i32 2\n store i32 %c, ptr %d2\n")
emith(" ret ptr %h\n}\n") emith(" ret ptr %h\n}\n")
# uname once, into a buffer kept for the program's life: platform() and arch() are asked every
# frame by some callers, and each call once malloc'd 8 KB for it and let it go
emith("@L_os_uts = internal global ptr null\n")
emith("define ptr @lp_os_uts() {\n")
emith("entry:\n %c = load ptr, ptr @L_os_uts\n %have = icmp ne ptr %c, null\n br i1 %have, label %done, label %make\n")
emith("done:\n ret ptr %c\n")
emith("make:\n %buf = call ptr @malloc(i64 8192)\n call i32 @uname(ptr %buf)\n store ptr %buf, ptr @L_os_uts\n ret ptr %buf\n}\n")
# platform(): uname sysname (field 0, portable) mapped to a short id # platform(): uname sysname (field 0, portable) mapped to a short id
emith("define ptr @lp_os_platform() {\n") emith("define ptr @lp_os_platform() {\n")
emith("entry:\n %buf = call ptr @malloc(i64 8192)\n call i32 @uname(ptr %buf)\n") emith("entry:\n %buf = call ptr @lp_os_uts()\n")
emith(` %cd = call i32 @strncmp(ptr %buf, ptr {k_darw}, i64 6)\n %isd = icmp eq i32 %cd, 0\n br i1 %isd, label %mac, label %chkl\n`) emith(` %cd = call i32 @strncmp(ptr %buf, ptr {k_darw}, i64 6)\n %isd = icmp eq i32 %cd, 0\n br i1 %isd, label %mac, label %chkl\n`)
emith(`mac:\n ret ptr {k_macos}\n`) emith(`mac:\n ret ptr {k_macos}\n`)
emith(`chkl:\n %cl = call i32 @strncmp(ptr %buf, ptr {k_linux_k}, i64 5)\n %isl = icmp eq i32 %cl, 0\n br i1 %isl, label %lin, label %other\n`) emith(`chkl:\n %cl = call i32 @strncmp(ptr %buf, ptr {k_linux_k}, i64 5)\n %isl = icmp eq i32 %cl, 0\n br i1 %isl, label %lin, label %other\n`)
@ -248,7 +256,7 @@ function emit_os_prelude() -> void {
# bytes, so `machine` (index 4) sits at offset 1024. Documented BSD-layout # bytes, so `machine` (index 4) sits at offset 1024. Documented BSD-layout
# assumption (see the header note); other layouts are a follow-up. # assumption (see the header note); other layouts are a follow-up.
emith("define ptr @lp_os_arch() {\n") emith("define ptr @lp_os_arch() {\n")
emith("entry:\n %buf = call ptr @malloc(i64 8192)\n call i32 @uname(ptr %buf)\n") emith("entry:\n %buf = call ptr @lp_os_uts()\n")
emith(" %m = getelementptr i8, ptr %buf, i64 1024\n ret ptr %m\n}\n") emith(" %m = getelementptr i8, ptr %buf, i64 1024\n ret ptr %m\n}\n")
} }

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff