Merge branch 'r3d/no-cmd-pool' into lang/leaks2

This commit is contained in:
Orkun ÇAKILKAYA 2026-09-28 13:41:23 +03:00
commit f8fe157352
13 changed files with 30721 additions and 30490 deletions

View file

@ -691,6 +691,12 @@ export state Render3dState {
stream_walks: int = 0 # streams that walked their whole ring this frame
stream_debug_n: int = 0
stream_scratch: floats = null
gvk_prime_b: words = null # buffers made since the frame began, read once so MoltenVK makes their
gvk_prime_h: []long = null # Metal buffers now (gvk_prime), and their handles
gvk_prime_n: int = 0
gvk_prime_dst: long = 0 # the 64 bytes those reads land in
gvk_prime_mem: long = 0
stream_loose: Chunk = null # the record for a chunk no stream's cache can keep this walk
stream_deadline: long = 0
gsl_on: bool = false # slInit succeeded: the interposer is the loader, the plugins are in
gsl_dlss_ok: bool = false # what this adapter can run, from slIsFeatureSupported

View file

@ -48,8 +48,18 @@ function gvk_surface_ext() -> string {
return VK_KHR_WIN32_SURFACE_EXTENSION_NAME
}
# MoltenVK's command pooling keeps every command object a frame ever recorded and never gives one
# back, so the heap grew each time a frame drew more than any before (~650 bytes a draw, for as long
# as the game ran). Off, the objects are made and freed with their command buffer, at no measured
# cost (3.7 ms either way over 520 actors). Read when the library loads, so before any Vk call.
function gvk_no_command_pooling() -> void {
if Os.platform() != "macos" or Os.has_env("MVK_CONFIG_USE_COMMAND_POOLING") { return }
Os.set_env("MVK_CONFIG_USE_COMMAND_POOLING", "0")
}
function gvk_init(render3d_st: mut Render3dState) -> bool {
if render3d_st.gvk_ready { return true }
gvk_no_command_pooling()
gsl_boot(render3d_st)
if Vk.open() == 0 { render3d_st.gvk_why = "no Vulkan loader"; return false }
gsl_init(render3d_st)
@ -638,6 +648,7 @@ function gvk_shutdown(render3d_st: mut Render3dState) -> void {
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_fence, 0), null)
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_frame_fence, 0), null)
Vk.destroy_command_pool(render3d_st.gvk_dev, render3d_st.gvk_pool, null)
if render3d_st.gvk_prime_dst != 0 { Vk.destroy_buffer(render3d_st.gvk_dev, render3d_st.gvk_prime_dst, null); render3d_st.gvk_prime_dst = 0 }
Vk.destroy_device(render3d_st.gvk_dev, null)
Vk.destroy_instance(render3d_st.gvk_inst, null)
render3d_st.gvk_ready = false

View file

@ -1013,6 +1013,7 @@ function gvk_frame_cb(render3d_st: mut Render3dState) -> pointer {
if render3d_st.gvk_frame_pending and (render3d_st.gvk_frame_pending_no & 1) == (render3d_st.gvk_frame_no & 1) { gvk_frame_wait(render3d_st) }
gvk_frame_reset(render3d_st)
render3d_st.gvk_cb = gvk_once_begin(render3d_st)
gvk_prime_flush(render3d_st, render3d_st.gvk_cb)
}
return render3d_st.gvk_cb
}

View file

@ -753,6 +753,68 @@ function gvk_buf_reserve(render3d_st: mut Render3dState, b: int, n: int) -> bool
r = Vk.bind_buffer_memory(render3d_st.gvk_dev, buf, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma))
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkBindBufferMemory", r) }
render3d_st.gvk_buf[b] = buf; render3d_st.gvk_buf_mem[b] = mem; render3d_st.gvk_buf_size[b] = size; render3d_st.gvk_buf_map[b] = gvk_mem_ptr(render3d_st, ma)
gvk_prime(render3d_st, b, buf)
return true
}
# MoltenVK makes a buffer's Metal buffer the first time a command uses it, so one written through
# its mapping and first drawn a thousand frames later allocated then, in play. Each new buffer is
# read once (4 bytes, copied out) at the start of the next frame's commands instead.
const GVK_PRIME_MAX: int = 16384
function gvk_prime(render3d_st: mut Render3dState, b: int, buf: long) -> void {
if Os.platform() != "macos" { return }
if render3d_st.gvk_prime_b == null {
render3d_st.gvk_prime_b = words(GVK_PRIME_MAX)
render3d_st.gvk_prime_h = new []long
let hs = render3d_st.gvk_prime_h
let none: long = 0
for i in 0 .. GVK_PRIME_MAX { push(hs, none) }
}
if render3d_st.gvk_prime_n >= GVK_PRIME_MAX { return }
render3d_st.gvk_prime_b[render3d_st.gvk_prime_n] = b
render3d_st.gvk_prime_h[render3d_st.gvk_prime_n] = buf
render3d_st.gvk_prime_n += 1
}
function gvk_prime_flush(render3d_st: mut Render3dState, cb: pointer) -> void {
if render3d_st.gvk_prime_n == 0 { return }
if render3d_st.gvk_prime_dst == 0 and not gvk_prime_target(render3d_st) {
render3d_st.gvk_prime_n = 0
return
}
let region = gvk_tmp(render3d_st, VkBufferCopy_sizeof)
Vk.zero(region, VkBufferCopy_sizeof)
let four: long = 4
Vk.put_i64(region, VkBufferCopy_size, four)
for i in 0 .. render3d_st.gvk_prime_n {
let b = render3d_st.gvk_prime_b[i]
let h = render3d_st.gvk_prime_h[i]
# released and made again, or gone, since: that one is not this buffer any more
if b > 0 and b < len(render3d_st.gvk_buf) and render3d_st.gvk_buf[b] == h { Vk.cmd_copy_buffer(cb, h, render3d_st.gvk_prime_dst, 1, region) }
}
render3d_st.gvk_prime_n = 0
}
function gvk_prime_target(render3d_st: mut Render3dState) -> bool {
let size: long = 64
let bci = gvk_tmp(render3d_st, VkBufferCreateInfo_sizeof)
Vk.zero(bci, VkBufferCreateInfo_sizeof)
Vk.put_i32(bci, VkBufferCreateInfo_sType, VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO)
Vk.put_i64(bci, VkBufferCreateInfo_size, size)
Vk.put_i32(bci, VkBufferCreateInfo_usage, VK_BUFFER_USAGE_TRANSFER_DST_BIT)
Vk.put_i32(bci, VkBufferCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE)
let out = gvk_tmp(render3d_st, 8)
if Vk.create_buffer(render3d_st.gvk_dev, bci, null, out) != VK_SUCCESS { return false }
let buf = gvk_handle(out)
let req = gvk_tmp(render3d_st, VkMemoryRequirements_sizeof)
Vk.get_buffer_memory_requirements(render3d_st.gvk_dev, buf, req)
let ma = gvk_mem_new(render3d_st, req, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, false)
if ma == 0 {
Vk.destroy_buffer(render3d_st.gvk_dev, buf, null)
return false
}
Vk.bind_buffer_memory(render3d_st.gvk_dev, buf, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma))
render3d_st.gvk_prime_dst = buf
let m: long = ma
render3d_st.gvk_prime_mem = m
return true
}

View file

@ -16,7 +16,7 @@
property Chunk {
key: int = 0, # packed (cx, cz, band)
used: int = 0, # the walk that last wanted it (for eviction)
data: words, # INST_FLOATS per instance
off: int = -1, # where its instances start in the stream's arena (words); -1 not kept
count: int = 0,
ymin: float = 0.0, # height range of its instances (float bits), for the frustum test
ymax: float = 0.0
@ -36,7 +36,10 @@ property Stream {
kind: int = 0, # the scene's generator selector for this stream
min_band: int = 0, # bands below this belong to another (nearer) stream
view_gen: int = -1, # sc_view_gen the layer was last gathered for (the view turned -> regather)
htab: words # open-addressed key -> chunk index + 1 (0 = empty)
htab: words, # open-addressed key -> chunk index + 1 (0 = empty)
arena: words, # every kept chunk's instances, in the order of `chunks`
top: int = 0, # words of the arena in use
spare: []Chunk # records not in use: all of them are made with the stream
}
# microseconds spent per frame, split so the hitch can be attributed (R3D_PROF=1)
@ -47,13 +50,24 @@ function stream_new(render3d_st: mut Render3dState, layer: Layer, size: float, r
if r3d_env_has(render3d_st, "R3D_STREAM_CAP") { render3d_st.STREAM_MAX_CHUNKS = Text.to_int(r3d_env(render3d_st, "R3D_STREAM_CAP")) }
render3d_st.stream_no_evict = r3d_env_has(render3d_st, "R3D_NOEVICT")
}
# the fill's scratch and the record for a chunk the cache cannot keep, both once for every stream
if render3d_st.stream_scratch == null { render3d_st.stream_scratch = floats(STREAM_CHUNK_MAX * INST_FLOATS) }
if render3d_st.stream_loose == null { render3d_st.stream_loose = new Chunk }
let s = new Stream
s.layer = layer; s.size = size; s.reach = reach
layer.streamed = true
layer.grounded = true
s.bands = floats(4)
s.bands[0] = b0; s.bands[1] = b1; s.bands[2] = b2; s.bands[3] = b3
# Everything the cache will ever hold is made here: a record per chunk it can keep and an arena
# of twice what the layer draws. Each chunk once had its own record and copy, made as the camera
# found new ground, so the heap grew for as long as there was ground nobody had stood near.
s.chunks = new []Chunk
s.spare = new []Chunk
let live = s.chunks
for i in 0 .. render3d_st.STREAM_MAX_CHUNKS { push(live, new Chunk) }
while len(live) > 0 { push(s.spare, List.pop(live)) }
s.arena = words(Math.max(layer.cap, 4096) * 2 * INST_FLOATS)
s.keys = words(render3d_st.STREAM_MAX_CHUNKS)
s.htab = words(STREAM_HASH)
for i in 0 .. STREAM_HASH { s.htab[i] = 0 }
@ -100,7 +114,6 @@ const STREAM_CHUNK_MAX: int = 262144
function stream_emit(render3d_st: mut Render3dState, s: Stream, x: float, y: float, z: float, scale: float, yaw: float, seed: float, wind: float) -> void {
let c = s.cur
if c.count >= STREAM_CHUNK_MAX { return }
if render3d_st.stream_scratch == null { render3d_st.stream_scratch = floats(STREAM_CHUNK_MAX * INST_FLOATS) }
if c.count == 0 { c.ymin = y; c.ymax = y } else { c.ymin = Math.min(c.ymin, y); c.ymax = Math.max(c.ymax, y) }
let o = c.count * INST_FLOATS
render3d_st.stream_scratch[o] = x; render3d_st.stream_scratch[o + 1] = y; render3d_st.stream_scratch[o + 2] = z; render3d_st.stream_scratch[o + 3] = scale
@ -164,19 +177,26 @@ function stream_evict(render3d_st: mut Render3dState, s: Stream) -> void {
# everything wanted by the walk in progress stays whatever the threshold says
# compacted in place, and an evicted chunk goes whole: a new list per eviction and the chunks'
# own records were never given back
# the arena is compacted in the same pass: a kept chunk's instances only ever move down
var w = 0
var top = 0
var i = 0
while i < s.n {
let c = s.chunks[i]
if c.used >= t or c.used == render3d_st.stream_walk_no {
let n = c.count * INST_FLOATS
if c.off != top { for k in 0 .. n { s.arena[top + k] = s.arena[c.off + k] } }
c.off = top
top += n
s.chunks[w] = c
w += 1
} else {
if c.data != null { free(c.data) }
free(c)
c.off = -1
push(s.spare, c)
}
i += 1
}
s.top = top
let ch = s.chunks
while len(ch) > w { List.pop(ch) }
s.n = w
@ -221,9 +241,12 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float,
# The cell underfoot and its neighbours are never deferred: they are what you
# are looking at, and a hole there is the grass vanishing as you walk into it.
let urgent = band == 0 and ring <= 1
var loose = false
if c == null and (first or urgent or gl_now_us() < render3d_st.stream_deadline) {
c = new Chunk
c.key = key
if len(s.spare) == 0 and not render3d_st.stream_no_evict { stream_evict(render3d_st, s) }
let sp = s.spare
if len(sp) > 0 { c = List.pop(sp) } else { c = render3d_st.stream_loose }
c.key = key; c.count = 0; c.off = -1; c.used = 0
s.cur = c
let t0 = gl_now_us()
r3d_stream_fill(render3d_st, s, cx, cz, band)
@ -231,15 +254,17 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float,
render3d_st.stream_us_gen = render3d_st.stream_us_gen + dt
prof_chunk(render3d_st, s.kind, band, c.count, dt)
if render3d_st.r3d_debug and band == 0 and render3d_st.stream_debug_n < 40 { render3d_st.stream_debug_n += 1; print(`stream kind {s.kind} band {band} chunk {cx},{cz}: {c.count} instances`) }
if c.count > 0 { c.data = words(c.count * INST_FLOATS); mem_copy(c.data, render3d_st.stream_scratch, c.count * INST_FLOATS * 4) }
if s.n >= render3d_st.STREAM_MAX_CHUNKS and not render3d_st.stream_no_evict { stream_evict(render3d_st, s) }
# If the walk in progress wants more chunks than the cache can hold, there
# is nothing to evict and this one is used and dropped, as every chunk used
# to be. The cap has to exceed one walk's ring for the cache to work at all.
if s.n < render3d_st.STREAM_MAX_CHUNKS {
let need = c.count * INST_FLOATS
if s.top + need > len(s.arena) and not render3d_st.stream_no_evict { stream_evict(render3d_st, s) }
# If the walk in progress wants more than the cache can hold, there is nothing to
# evict and this one is drawn from the scratch and dropped, as every chunk used to be.
if c != render3d_st.stream_loose and s.n < render3d_st.STREAM_MAX_CHUNKS and s.top + need <= len(s.arena) {
if need > 0 { mem_copy(mem_off(data_of(s.arena), s.top * 4), data_of(render3d_st.stream_scratch), need * 4) }
c.off = s.top
s.top += need
push(s.chunks, c); s.keys[s.n] = key; stream_remember(s, key, s.n); s.n += 1
c.used = render3d_st.stream_walk_no
}
} else { loose = true }
prof_gen_add(render3d_st, c.count + 512)
}
@ -255,10 +280,13 @@ function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float,
if c != null and c.count > 0 and l.count + c.count <= l.cap and stream_chunk_visible(render3d_st, s, cx, cz, c) {
let tg = gl_now_us()
layer_room(l, l.count + c.count)
mem_copy(mem_off(l.inst, l.count * INST_FLOATS * 4), data_of(c.data), c.count * INST_FLOATS * 4)
var src = data_of(render3d_st.stream_scratch)
if c.off >= 0 { src = mem_off(data_of(s.arena), c.off * 4) }
mem_copy(mem_off(l.inst, l.count * INST_FLOATS * 4), src, c.count * INST_FLOATS * 4)
l.count += c.count
render3d_st.stream_us_gather = render3d_st.stream_us_gather + (gl_now_us() - tg)
}
if loose and c != render3d_st.stream_loose { push(s.spare, c) }
}
}
cx += 1
@ -313,10 +341,14 @@ function stream_clear_all(render3d_st: mut Render3dState) -> void {
if render3d_st.stream_all == null { return }
for i in 0 .. len(render3d_st.stream_all) {
let s = render3d_st.stream_all[i]
if s.chunks != null { for c in 0 .. len(s.chunks) { if s.chunks[c].data != null { free(s.chunks[c].data) } } }
for c in 0 .. len(s.chunks) { free(s.chunks[c]) }
for c in 0 .. len(s.spare) { free(s.spare[c]) }
free(s.chunks); free(s.spare); free(s.arena)
if s.keys != null { free(s.keys) }
if s.htab != null { free(s.htab) }
if s.bands != null { free(s.bands) }
free(s)
}
render3d_st.stream_all = null
let all = render3d_st.stream_all
List.clear(all)
}