Os.platform/arch uname once; render3d primes a new buffer's Metal buffer
lp_os_platform and lp_os_arch malloc'd 8 KB per call (the uname buffer) and kept none of it: string_temps now asks both every round, 327 MB over 20,000 before, 0 after. Reseeded. render3d: MoltenVK made a mapped buffer's MTLBuffer at its first bind (fn_gvk_draw +4 blocks in the boat window); gvk_buf_reserve queues it and the next frame's command buffer copies 4 bytes out of it. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
7e9fa4473c
commit
0141f99dd1
10 changed files with 30575 additions and 30470 deletions
|
|
@ -6,3 +6,5 @@ busiest frame so far, for as long as the game ran. render3d turns it off
|
|||
(`MVK_CONFIG_USE_COMMAND_POOLING=0`, unless the environment already says otherwise) before the first
|
||||
Vulkan call; the objects are made and freed with their command buffer, at no measured cost.
|
||||
`examples/rendering/steady.ludic` ramps a frame from 20 to 200 actors and fails on what pooling left.
|
||||
A buffer written through its mapping is also read once at the start of the next frame's commands,
|
||||
so MoltenVK makes its Metal buffer then rather than at its first draw, however much later that is.
|
||||
|
|
|
|||
5
changes/os-platform-once.md
Normal file
5
changes/os-platform-once.md
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
bump: patch
|
||||
type: fix
|
||||
**`Os.platform()` and `Os.arch()` allocate nothing after the first call.** Each call malloc'd an
|
||||
8 KB `uname` buffer and let it go, and a game asks the platform every frame in places (a launcher's
|
||||
wait, an update panel, a renderer's present). The buffer is made once and kept.
|
||||
|
|
@ -12,6 +12,7 @@ program StringTemps {
|
|||
if a + "/" + b + "/" + string(i) != `{a}/{b}/{i}` { bad += 1 }
|
||||
let p = "lake/camp.png" # a literal: the slices are what is made here
|
||||
if p[len(p) - 4 .. len(p)] != ".png" or p[0 .. len(p) - 4] + ".dds" != `lake/{b[0 .. 0]}camp.dds` { bad += 1 }
|
||||
if Os.platform() == "" or Os.arch() == "" { bad += 1 } # asked every round: nothing each time
|
||||
if `{i}-{i * 2}-{long(i) * 1000000000}` != string(i) + "-" + string(i * 2) + "-" + string(long(i) * 1000000000) { bad += 1 }
|
||||
return bad
|
||||
}
|
||||
|
|
|
|||
|
|
@ -691,6 +691,11 @@ export state Render3dState {
|
|||
stream_walks: int = 0 # streams that walked their whole ring this frame
|
||||
stream_debug_n: int = 0
|
||||
stream_scratch: floats = null
|
||||
gvk_prime_b: words = null # buffers made since the frame began, read once so MoltenVK makes their
|
||||
gvk_prime_h: []long = null # Metal buffers now (gvk_prime), and their handles
|
||||
gvk_prime_n: int = 0
|
||||
gvk_prime_dst: long = 0 # the 64 bytes those reads land in
|
||||
gvk_prime_mem: long = 0
|
||||
stream_loose: Chunk = null # the record for a chunk no stream's cache can keep this walk
|
||||
stream_deadline: long = 0
|
||||
gsl_on: bool = false # slInit succeeded: the interposer is the loader, the plugins are in
|
||||
|
|
|
|||
|
|
@ -648,6 +648,7 @@ function gvk_shutdown(render3d_st: mut Render3dState) -> void {
|
|||
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_fence, 0), null)
|
||||
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_frame_fence, 0), null)
|
||||
Vk.destroy_command_pool(render3d_st.gvk_dev, render3d_st.gvk_pool, null)
|
||||
if render3d_st.gvk_prime_dst != 0 { Vk.destroy_buffer(render3d_st.gvk_dev, render3d_st.gvk_prime_dst, null); render3d_st.gvk_prime_dst = 0 }
|
||||
Vk.destroy_device(render3d_st.gvk_dev, null)
|
||||
Vk.destroy_instance(render3d_st.gvk_inst, null)
|
||||
render3d_st.gvk_ready = false
|
||||
|
|
|
|||
|
|
@ -1013,6 +1013,7 @@ function gvk_frame_cb(render3d_st: mut Render3dState) -> pointer {
|
|||
if render3d_st.gvk_frame_pending and (render3d_st.gvk_frame_pending_no & 1) == (render3d_st.gvk_frame_no & 1) { gvk_frame_wait(render3d_st) }
|
||||
gvk_frame_reset(render3d_st)
|
||||
render3d_st.gvk_cb = gvk_once_begin(render3d_st)
|
||||
gvk_prime_flush(render3d_st, render3d_st.gvk_cb)
|
||||
}
|
||||
return render3d_st.gvk_cb
|
||||
}
|
||||
|
|
|
|||
|
|
@ -753,6 +753,68 @@ function gvk_buf_reserve(render3d_st: mut Render3dState, b: int, n: int) -> bool
|
|||
r = Vk.bind_buffer_memory(render3d_st.gvk_dev, buf, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma))
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkBindBufferMemory", r) }
|
||||
render3d_st.gvk_buf[b] = buf; render3d_st.gvk_buf_mem[b] = mem; render3d_st.gvk_buf_size[b] = size; render3d_st.gvk_buf_map[b] = gvk_mem_ptr(render3d_st, ma)
|
||||
gvk_prime(render3d_st, b, buf)
|
||||
return true
|
||||
}
|
||||
|
||||
# MoltenVK makes a buffer's Metal buffer the first time a command uses it, so one written through
|
||||
# its mapping and first drawn a thousand frames later allocated then, in play. Each new buffer is
|
||||
# read once (4 bytes, copied out) at the start of the next frame's commands instead.
|
||||
const GVK_PRIME_MAX: int = 16384
|
||||
function gvk_prime(render3d_st: mut Render3dState, b: int, buf: long) -> void {
|
||||
if Os.platform() != "macos" { return }
|
||||
if render3d_st.gvk_prime_b == null {
|
||||
render3d_st.gvk_prime_b = words(GVK_PRIME_MAX)
|
||||
render3d_st.gvk_prime_h = new []long
|
||||
let hs = render3d_st.gvk_prime_h
|
||||
let none: long = 0
|
||||
for i in 0 .. GVK_PRIME_MAX { push(hs, none) }
|
||||
}
|
||||
if render3d_st.gvk_prime_n >= GVK_PRIME_MAX { return }
|
||||
render3d_st.gvk_prime_b[render3d_st.gvk_prime_n] = b
|
||||
render3d_st.gvk_prime_h[render3d_st.gvk_prime_n] = buf
|
||||
render3d_st.gvk_prime_n += 1
|
||||
}
|
||||
function gvk_prime_flush(render3d_st: mut Render3dState, cb: pointer) -> void {
|
||||
if render3d_st.gvk_prime_n == 0 { return }
|
||||
if render3d_st.gvk_prime_dst == 0 and not gvk_prime_target(render3d_st) {
|
||||
render3d_st.gvk_prime_n = 0
|
||||
return
|
||||
}
|
||||
let region = gvk_tmp(render3d_st, VkBufferCopy_sizeof)
|
||||
Vk.zero(region, VkBufferCopy_sizeof)
|
||||
let four: long = 4
|
||||
Vk.put_i64(region, VkBufferCopy_size, four)
|
||||
for i in 0 .. render3d_st.gvk_prime_n {
|
||||
let b = render3d_st.gvk_prime_b[i]
|
||||
let h = render3d_st.gvk_prime_h[i]
|
||||
# released and made again, or gone, since: that one is not this buffer any more
|
||||
if b > 0 and b < len(render3d_st.gvk_buf) and render3d_st.gvk_buf[b] == h { Vk.cmd_copy_buffer(cb, h, render3d_st.gvk_prime_dst, 1, region) }
|
||||
}
|
||||
render3d_st.gvk_prime_n = 0
|
||||
}
|
||||
function gvk_prime_target(render3d_st: mut Render3dState) -> bool {
|
||||
let size: long = 64
|
||||
let bci = gvk_tmp(render3d_st, VkBufferCreateInfo_sizeof)
|
||||
Vk.zero(bci, VkBufferCreateInfo_sizeof)
|
||||
Vk.put_i32(bci, VkBufferCreateInfo_sType, VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO)
|
||||
Vk.put_i64(bci, VkBufferCreateInfo_size, size)
|
||||
Vk.put_i32(bci, VkBufferCreateInfo_usage, VK_BUFFER_USAGE_TRANSFER_DST_BIT)
|
||||
Vk.put_i32(bci, VkBufferCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE)
|
||||
let out = gvk_tmp(render3d_st, 8)
|
||||
if Vk.create_buffer(render3d_st.gvk_dev, bci, null, out) != VK_SUCCESS { return false }
|
||||
let buf = gvk_handle(out)
|
||||
let req = gvk_tmp(render3d_st, VkMemoryRequirements_sizeof)
|
||||
Vk.get_buffer_memory_requirements(render3d_st.gvk_dev, buf, req)
|
||||
let ma = gvk_mem_new(render3d_st, req, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, false)
|
||||
if ma == 0 {
|
||||
Vk.destroy_buffer(render3d_st.gvk_dev, buf, null)
|
||||
return false
|
||||
}
|
||||
Vk.bind_buffer_memory(render3d_st.gvk_dev, buf, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma))
|
||||
render3d_st.gvk_prime_dst = buf
|
||||
let m: long = ma
|
||||
render3d_st.gvk_prime_mem = m
|
||||
return true
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -235,9 +235,17 @@ function emit_os_prelude() -> void {
|
|||
emith(" %d2 = getelementptr inbounds %LSlice, ptr %h, i32 0, i32 2\n store i32 %c, ptr %d2\n")
|
||||
emith(" ret ptr %h\n}\n")
|
||||
|
||||
# uname once, into a buffer kept for the program's life: platform() and arch() are asked every
|
||||
# frame by some callers, and each call once malloc'd 8 KB for it and let it go
|
||||
emith("@L_os_uts = internal global ptr null\n")
|
||||
emith("define ptr @lp_os_uts() {\n")
|
||||
emith("entry:\n %c = load ptr, ptr @L_os_uts\n %have = icmp ne ptr %c, null\n br i1 %have, label %done, label %make\n")
|
||||
emith("done:\n ret ptr %c\n")
|
||||
emith("make:\n %buf = call ptr @malloc(i64 8192)\n call i32 @uname(ptr %buf)\n store ptr %buf, ptr @L_os_uts\n ret ptr %buf\n}\n")
|
||||
|
||||
# platform(): uname sysname (field 0, portable) mapped to a short id
|
||||
emith("define ptr @lp_os_platform() {\n")
|
||||
emith("entry:\n %buf = call ptr @malloc(i64 8192)\n call i32 @uname(ptr %buf)\n")
|
||||
emith("entry:\n %buf = call ptr @lp_os_uts()\n")
|
||||
emith(` %cd = call i32 @strncmp(ptr %buf, ptr {k_darw}, i64 6)\n %isd = icmp eq i32 %cd, 0\n br i1 %isd, label %mac, label %chkl\n`)
|
||||
emith(`mac:\n ret ptr {k_macos}\n`)
|
||||
emith(`chkl:\n %cl = call i32 @strncmp(ptr %buf, ptr {k_linux_k}, i64 5)\n %isl = icmp eq i32 %cl, 0\n br i1 %isl, label %lin, label %other\n`)
|
||||
|
|
@ -248,7 +256,7 @@ function emit_os_prelude() -> void {
|
|||
# bytes, so `machine` (index 4) sits at offset 1024. Documented BSD-layout
|
||||
# assumption (see the header note); other layouts are a follow-up.
|
||||
emith("define ptr @lp_os_arch() {\n")
|
||||
emith("entry:\n %buf = call ptr @malloc(i64 8192)\n call i32 @uname(ptr %buf)\n")
|
||||
emith("entry:\n %buf = call ptr @lp_os_uts()\n")
|
||||
emith(" %m = getelementptr i8, ptr %buf, i64 1024\n ret ptr %m\n}\n")
|
||||
}
|
||||
|
||||
|
|
|
|||
30478
selfhost/ludicc.seed.ll
30478
selfhost/ludicc.seed.ll
File diff suppressed because it is too large
Load diff
File diff suppressed because it is too large
Load diff
Loading…
Add table
Add a link
Reference in a new issue