perf(render3d): Vulkan device memory in blocks, and a shadow map that survives running out of it

- A block sub-allocator: 64 MB blocks per memory type, images and buffers kept apart, first fit with
  alignment, freed ranges merged and empty blocks given back; anything over 16 MB still gets its own
  allocation. The self-tests' second world holds 94 allocations instead of 8735 (the driver's limit
  refused a shadow map before). Camp bench unchanged, 105.3 fps.
- shadow_set_res keeps the size that worked when the card has no memory for the new one, and records
  it in shadow_refused, instead of ending with no shadow map; gpu_tex_ok says whether a texture has an
  image behind it.
- Image barriers skip an image that was never made (a failed allocation used to crash there), and
  R3D_VK_ERRLOG=<file> appends every Vulkan failure line by line, so a crash no longer takes the
  message with it.
- R3D_VK_PROF reports draws asked for and not made, so a layer missing from a frame is never silent.

Validation on (VK_INSTANCE_LAYERS): the game's self-tests 61 OK, 0 errors, on the RTX 3070 Ti.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-15 15:44:21 +03:00
parent 3706920e16
commit 989bdca736
5 changed files with 255 additions and 21 deletions

View file

@ -623,6 +623,12 @@ function gpu_tx_at(tex: int) -> int {
function gpu_bound(kind: int) -> int { if kind == GPU_TEX2D_ARRAY { return gpu_bound_array }; return gpu_bound_2d }
function gpu_tex_new() -> int { if gpu_kind == GPU_VK { return gvk_tex_new() }; return gl_texture() }
# does texture `tex` have an image behind it? Always on OpenGL; on Vulkan an image whose memory could
# not be had is never made, and a caller that can fall back (a smaller shadow map) asks here.
function gpu_tex_ok(tex: int) -> bool {
if gpu_kind != GPU_VK { return tex > 0 }
return tex > 0 and tex < len(gvk_tex_image) and gvk_tex_image[tex] != 0
}
# what GL has on each unit's 2D target, for R3D_GLCHECK: deleting a texture unbinds it everywhere
var gpu_unit_2d: words = null
var gpu_unit_cur: int = 0

View file

@ -22,9 +22,28 @@ var gvk_has_surface: bool = false # the instance and device can make a surfac
function gvk_fail(what: string, r: int) -> bool {
gvk_why = `{what} failed (VkResult {r})`
print(`r3d: vulkan: {gvk_why}`)
gvk_note(`r3d: vulkan: {gvk_why}`)
return false
}
# A Vulkan failure is printed, and with R3D_VK_ERRLOG=<file> also appended to that file, opened and
# closed for each line: stdout is buffered, and a crash straight after a failure (an image that was
# never made, then used) used to take the one line that explained it with it.
var gvk_errlog: string = null
var gvk_errlog_read: bool = false
function gvk_note(msg: string) -> void {
print(msg)
if not gvk_errlog_read {
gvk_errlog_read = true
if Os.has_env("R3D_VK_ERRLOG") { gvk_errlog = Os.env("R3D_VK_ERRLOG") }
}
if gvk_errlog == null { return }
let f = file_open(gvk_errlog, "ab")
if f == null { return }
let line = msg + "\n"
file_write(f, line, len(line))
file_close(f)
}
function gvk_handle(out: bytes) -> long { return Vk.get_i64(out, 0) }
function gvk_ext_in(props: bytes, n: int, want: string) -> bool {
@ -242,6 +261,193 @@ function gvk_alloc(req: bytes, want: int) -> long {
return gvk_handle(out)
}
# ---- sub-allocation ----------------------------------------------------------------------
# Device memory in blocks, resources carved out of them: one vkAllocateMemory per resource met the
# driver's limit in the self-tests' second world (8735 live allocations and a shadow map refused).
# Blocks are 64 MB of device-local memory or of host-visible, kept apart for images and for
# buffers (so the driver's granularity between the two never matters), first fit with alignment,
# and a freed range merges with its neighbours. A host-visible block is mapped once; a buffer's
# pointer is the block's plus its offset. Anything over 16 MB still gets an allocation of its own:
# bigger blocks held back video memory a second world then could not get (a 4096^2 shadow map and
# then its 2048^2 fallback were refused with 71 allocations live on an 8 GB card).
# An allocation is an id (0 is none), kept where a memory handle used to be.
const GVK_BLOCK_DEVICE: int = 67108864
const GVK_BLOCK_HOST: int = 67108864
const GVK_OWN_OVER: int = 16777216
var gvk_blk_mem: []long = null
var gvk_blk_kind: []int = null # memory type * 2, + 1 for images
var gvk_blk_size: []int = null
var gvk_blk_map: []pointer = null
var gvk_fr_blk: []int = null # free ranges: block, offset, length (0 = an unused slot)
var gvk_fr_off: []int = null
var gvk_fr_len: []int = null
var gvk_al_blk: []int = null # per allocation id: its block, or -1 for its own memory
var gvk_al_mem: []long = null # its own memory when it has one
var gvk_al_off: []int = null
var gvk_al_len: []int = null
var gvk_al_map: []pointer = null
var gvk_al_spare: []int = null # ids to reuse
function gvk_mem_init() -> void {
if gvk_al_blk != null { return }
let zero: long = 0
gvk_blk_mem = new []long; gvk_blk_kind = new []int; gvk_blk_size = new []int; gvk_blk_map = new []pointer
gvk_fr_blk = new []int; gvk_fr_off = new []int; gvk_fr_len = new []int
gvk_al_blk = new []int; gvk_al_mem = new []long; gvk_al_off = new []int; gvk_al_len = new []int
gvk_al_map = new []pointer; gvk_al_spare = new []int
push(gvk_al_blk, -1); push(gvk_al_mem, zero); push(gvk_al_off, 0); push(gvk_al_len, 0); push(gvk_al_map, null)
}
# one vkAllocateMemory of `size` bytes of memory type t, mapped when host-visible; 0 on failure
function gvk_mem_raw(t: int, size: int, host: bool, out_map: []pointer) -> long {
let zero: long = 0
let mai = bytes(VkMemoryAllocateInfo_sizeof)
Vk.zero(mai, VkMemoryAllocateInfo_sizeof)
Vk.put_i32(mai, VkMemoryAllocateInfo_sType, VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO)
let size_l: long = size
Vk.put_i64(mai, VkMemoryAllocateInfo_allocationSize, size_l)
Vk.put_i32(mai, VkMemoryAllocateInfo_memoryTypeIndex, t)
let out = bytes(8)
let r = Vk.allocate_memory(gvk_dev, mai, null, out)
if r != VK_SUCCESS { gvk_note(`r3d: vulkan: vkAllocateMemory of {size} bytes failed (VkResult {r}, {gvk_n_allocs} allocations live)`); return zero }
gvk_n_allocs += 1
let mem = gvk_handle(out)
out_map[0] = null
if host {
let pp = bytes(8)
if Vk.map_memory(gvk_dev, mem, zero, size_l, 0, pp) != VK_SUCCESS { gvk_note("r3d: vulkan: vkMapMemory failed"); return zero }
out_map[0] = Vk.get_ptr(pp, 0)
}
return mem
}
function gvk_fr_put(b: int, off: int, len_: int) -> void {
if len_ <= 0 { return }
for i in 0 .. len(gvk_fr_blk) { if gvk_fr_len[i] == 0 { gvk_fr_blk[i] = b; gvk_fr_off[i] = off; gvk_fr_len[i] = len_; return } }
push(gvk_fr_blk, b); push(gvk_fr_off, off); push(gvk_fr_len, len_)
}
# Memory for a resource from its requirements (a VkMemoryRequirements): an allocation id, 0 if none.
function gvk_mem_new(req: bytes, want: int, image: bool) -> int {
gvk_mem_init()
let allowed = Vk.get_i32(req, VkMemoryRequirements_memoryTypeBits)
var t = gvk_mem_type(allowed, want)
if t < 0 and want == VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT { t = gvk_mem_type(allowed, 0) }
if t < 0 { gvk_note(`r3d: vulkan: no memory type for properties {want}`); return 0 }
let host = (want & VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT) != 0
let size = Text.to_int(string(Vk.get_i64(req, VkMemoryRequirements_size)))
var align = Text.to_int(string(Vk.get_i64(req, VkMemoryRequirements_alignment)))
if align < 1 { align = 1 }
var blk = -1
var off = 0
let zero: long = 0
let mp = new []pointer
push(mp, null)
if size <= GVK_OWN_OVER {
var kind = t * 2
if image { kind += 1 }
var i = 0
while i < len(gvk_fr_blk) and blk < 0 {
let n = gvk_fr_len[i]
if n > 0 and gvk_blk_kind[gvk_fr_blk[i]] == kind {
let start = (gvk_fr_off[i] + align - 1) / align * align
let end = gvk_fr_off[i] + n
if start + size <= end {
blk = gvk_fr_blk[i]; off = start
let front = start - gvk_fr_off[i]
let back = end - (start + size)
if front > 0 { gvk_fr_len[i] = front } else { gvk_fr_len[i] = 0 }
gvk_fr_put(blk, start + size, back)
}
}
i += 1
}
if blk < 0 {
var bsize = GVK_BLOCK_DEVICE
if host { bsize = GVK_BLOCK_HOST }
let mem = gvk_mem_raw(t, bsize, host, mp)
if mem != 0 {
push(gvk_blk_mem, mem); push(gvk_blk_kind, kind); push(gvk_blk_size, bsize); push(gvk_blk_map, mp[0])
blk = len(gvk_blk_mem) - 1; off = 0
gvk_fr_put(blk, size, bsize - size)
}
}
}
var own: long = 0
var map: pointer = null
if blk < 0 {
own = gvk_mem_raw(t, size, host, mp)
if own == 0 { return 0 }
map = mp[0]
} else if gvk_blk_map[blk] != null {
map = mem_off(gvk_blk_map[blk], off)
}
var a = 0
if len(gvk_al_spare) > 0 {
a = gvk_al_spare[len(gvk_al_spare) - 1]
gvk_al_spare = gvk_list_drop_last(gvk_al_spare)
} else {
push(gvk_al_blk, -1); push(gvk_al_mem, zero); push(gvk_al_off, 0); push(gvk_al_len, 0); push(gvk_al_map, null)
a = len(gvk_al_blk) - 1
}
gvk_al_blk[a] = blk; gvk_al_mem[a] = own; gvk_al_off[a] = off; gvk_al_len[a] = size; gvk_al_map[a] = map
return a
}
function gvk_list_drop_last(l: []int) -> []int {
let out = new []int
for i in 0 .. len(l) - 1 { push(out, l[i]) }
return out
}
function gvk_mem_handle(a: int) -> long {
if gvk_al_blk[a] < 0 { return gvk_al_mem[a] }
return gvk_blk_mem[gvk_al_blk[a]]
}
function gvk_mem_offset(a: int) -> long {
let o: long = gvk_al_off[a]
return o
}
function gvk_mem_ptr(a: int) -> pointer { return gvk_al_map[a] }
# give an allocation back: its own memory is freed, a range returns to its block and merges
function gvk_mem_free(a: int) -> void {
if gvk_al_blk == null or a <= 0 or a >= len(gvk_al_blk) or gvk_al_len[a] == 0 { return }
let b = gvk_al_blk[a]
let zero: long = 0
if b < 0 {
if gvk_al_map[a] != null { Vk.unmap_memory(gvk_dev, gvk_al_mem[a]) }
Vk.free_memory(gvk_dev, gvk_al_mem[a], null)
gvk_n_allocs -= 1
} else {
var off = gvk_al_off[a]
var n = gvk_al_len[a]
# merge with a free range that ends where this starts, and one that starts where this ends
for i in 0 .. len(gvk_fr_blk) {
if gvk_fr_len[i] > 0 and gvk_fr_blk[i] == b and gvk_fr_off[i] + gvk_fr_len[i] == off {
off = gvk_fr_off[i]; n += gvk_fr_len[i]; gvk_fr_len[i] = 0
}
}
for i in 0 .. len(gvk_fr_blk) {
if gvk_fr_len[i] > 0 and gvk_fr_blk[i] == b and gvk_fr_off[i] == off + n {
n += gvk_fr_len[i]; gvk_fr_len[i] = 0
}
}
if off == 0 and n == gvk_blk_size[b] {
# the block is empty again: give it back, or a world swapped out keeps its memory for good
if gvk_blk_map[b] != null { Vk.unmap_memory(gvk_dev, gvk_blk_mem[b]) }
Vk.free_memory(gvk_dev, gvk_blk_mem[b], null)
gvk_n_allocs -= 1
gvk_blk_mem[b] = zero; gvk_blk_map[b] = null; gvk_blk_kind[b] = -1; gvk_blk_size[b] = 0
} else {
gvk_fr_put(b, off, n)
}
}
gvk_al_len[a] = 0; gvk_al_mem[a] = zero; gvk_al_map[a] = null; gvk_al_blk[a] = -1
push(gvk_al_spare, a)
}
function gvk_mem_id(x: long) -> int { return Text.to_int(string(x)) }
# ---- one-shot commands --------------------------------------------------------------------
# Uploads, bakes and read-backs record into a command buffer, submit it and wait. The frame
# itself does not go through here.

View file

@ -985,6 +985,8 @@ function gvk_view_of(tex: int, layer1: int) -> long {
}
# the layer range a barrier for an attachment covers: the whole image unless one layer is drawn
function gvk_att_barrier(cb: pointer, tex: int, layer1: int, depth: bool, old_layout: int, new_layout: int) -> void {
# an attachment whose image was never made (no memory for it) has nothing to transition
if tex <= 0 or tex >= len(gvk_tex_image) or gvk_tex_image[tex] == 0 { return }
let b = bytes(VkImageMemoryBarrier_sizeof)
Vk.zero(b, VkImageMemoryBarrier_sizeof)
Vk.put_i32(b, VkImageMemoryBarrier_sType, VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER)
@ -1415,6 +1417,7 @@ function gvk_draw_indirect_now(m: Mesh, cmds: int, offset: int, n: int, count_bu
}
function gvk_draw_now(m: Mesh, first: int, count: int, instances: int) -> void {
if gvk_prof() { gvk_n_asked += 1 }
gvk_draw(gpu_prog_cur, m, gvk_state_now(), first, count, instances, gpu_tx, GPU_TX_W, gpu_tx_cap, gpu_fb, gpu_fb_at(gvk_fb_cur))
}
@ -1664,6 +1667,7 @@ var gvk_us_pipe: long = 0
var gvk_us_set: long = 0
var gvk_us_draw: long = 0
var gvk_n_draws: int = 0
var gvk_n_asked: int = 0 # draws that reached gvk_draw_now: every one the renderer asked for
var gvk_n_flush: int = 0
var gvk_prof_frames: int = 0
function gvk_prof() -> bool {
@ -1679,9 +1683,14 @@ function gvk_prof_frame() -> void {
let set = Text.to_int(string(gvk_us_set)) / f
let draw = Text.to_int(string(gvk_us_draw)) / f
print(`r3d: vulkan per frame: {gvk_n_draws / f} draws, {gvk_n_flush / f} flushes; pipelines {pipe / 1000}.{(pipe / 100) % 10} ms, sets {set / 1000}.{(set / 100) % 10} ms, inside draws {draw / 1000}.{(draw / 100) % 10} ms`)
# every draw asked for should have been made: a gap is a layer missing from the frame, which is how
# MoltenVK's refused skinned pipelines went unseen (the drawstats session found it by counting)
if gvk_n_asked != gvk_n_draws {
print(`r3d: vulkan: {(gvk_n_asked - gvk_n_draws) / f} draws a frame were asked for and not made ({gvk_n_asked / f} asked, {gvk_n_draws / f} made)`)
}
let zero: long = 0
gvk_us_pipe = zero; gvk_us_set = zero; gvk_us_draw = zero
gvk_n_draws = 0; gvk_n_flush = 0; gvk_prof_frames = 0
gvk_n_draws = 0; gvk_n_asked = 0; gvk_n_flush = 0; gvk_prof_frames = 0
}
# ---- the pipeline cache, by integers -----------------------------------------------------------

View file

@ -100,6 +100,8 @@ function gvk_layout_access(layout: int) -> int {
}
# levels [base, base + n) of every layer of an image, from one layout to another
function gvk_barrier(cb: pointer, image: long, depth: bool, base: int, n: int, layers: int, old_layout: int, new_layout: int) -> void {
# an image that was never made (its memory could not be had) has nothing to transition
if image == 0 { return }
let b = bytes(VkImageMemoryBarrier_sizeof)
Vk.zero(b, VkImageMemoryBarrier_sizeof)
Vk.put_i32(b, VkImageMemoryBarrier_sType, VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER)
@ -183,7 +185,8 @@ function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layer
let image = gvk_handle(out)
let req = bytes(VkMemoryRequirements_sizeof)
Vk.get_image_memory_requirements(gvk_dev, image, req)
let mem = gvk_alloc(req, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT)
let ma = gvk_mem_new(req, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, true)
let mem: long = ma # the allocation id (gvk_mem_new), kept where the memory was
let zero: long = 0
# no memory for it (one allocation per resource meets the driver's allocation limit long before
# the card is full): say so instead of binding a null allocation, which the driver may accept
@ -191,7 +194,7 @@ function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layer
Vk.destroy_image(gvk_dev, image, null)
return gvk_fail(`no device memory for a {w}x{h}x{layers} image ({gvk_n_allocs} allocations live)`, VK_ERROR_OUT_OF_DEVICE_MEMORY)
}
r = Vk.bind_image_memory(gvk_dev, image, mem, zero)
r = Vk.bind_image_memory(gvk_dev, image, gvk_mem_handle(ma), gvk_mem_offset(ma))
if r != VK_SUCCESS { return gvk_fail("vkBindImageMemory", r) }
let vci = bytes(VkImageViewCreateInfo_sizeof)
Vk.zero(vci, VkImageViewCreateInfo_sizeof)
@ -331,8 +334,7 @@ function gvk_tex_grow_mips(tex: int, w: int, h: int) -> bool {
let ok = gvk_once_end(cb)
Vk.destroy_image_view(gvk_dev, old_view, null)
Vk.destroy_image(gvk_dev, old_image, null)
Vk.free_memory(gvk_dev, old_mem, null)
gvk_n_allocs -= 1
gvk_mem_free(gvk_mem_id(old_mem))
return ok
}
@ -428,8 +430,7 @@ function gvk_tex_release(tex: int) -> void {
let zero: long = 0
Vk.destroy_image_view(gvk_dev, gvk_tex_view[tex], null)
Vk.destroy_image(gvk_dev, gvk_tex_image[tex], null)
Vk.free_memory(gvk_dev, gvk_tex_mem[tex], null)
gvk_n_allocs -= 1
gvk_mem_free(gvk_mem_id(gvk_tex_mem[tex]))
gvk_tex_image[tex] = zero; gvk_tex_view[tex] = zero; gvk_tex_mem[tex] = zero
gvk_tex_levels[tex] = 0; gvk_tex_layers[tex] = 0; gvk_tex_vkfmt[tex] = 0
gvk_tex_gen[tex] = gvk_tex_gen[tex] + 1
@ -547,10 +548,8 @@ function gvk_buf_release(b: int) -> void {
# a draw recorded this frame still reads it: destroy it once the frame has been submitted
push(gvk_retired_buf, gvk_buf[b]); push(gvk_retired_mem, gvk_buf_mem[b])
} else {
Vk.unmap_memory(gvk_dev, gvk_buf_mem[b])
Vk.destroy_buffer(gvk_dev, gvk_buf[b], null)
Vk.free_memory(gvk_dev, gvk_buf_mem[b], null)
gvk_n_allocs -= 1
gvk_mem_free(gvk_mem_id(gvk_buf_mem[b]))
}
gvk_buf[b] = zero; gvk_buf_mem[b] = zero; gvk_buf_size[b] = 0; gvk_buf_map[b] = null; gvk_buf_used[b] = 0
}
@ -558,10 +557,8 @@ function gvk_buf_release(b: int) -> void {
function gvk_retire_flush() -> void {
if gvk_retired_buf == null { return }
for i in 0 .. len(gvk_retired_buf) {
Vk.unmap_memory(gvk_dev, gvk_retired_mem[i])
Vk.destroy_buffer(gvk_dev, gvk_retired_buf[i], null)
Vk.free_memory(gvk_dev, gvk_retired_mem[i], null)
gvk_n_allocs -= 1
gvk_mem_free(gvk_mem_id(gvk_retired_mem[i]))
}
gvk_retired_buf = new []long
gvk_retired_mem = new []long
@ -599,15 +596,13 @@ function gvk_buf_reserve(b: int, n: int) -> bool {
let buf = gvk_handle(out)
let req = bytes(VkMemoryRequirements_sizeof)
Vk.get_buffer_memory_requirements(gvk_dev, buf, req)
let mem = gvk_alloc(req, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT)
let ma = gvk_mem_new(req, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, false)
let zero: long = 0
if mem == 0 { Vk.destroy_buffer(gvk_dev, buf, null); return false }
r = Vk.bind_buffer_memory(gvk_dev, buf, mem, zero)
if ma == 0 { Vk.destroy_buffer(gvk_dev, buf, null); return false }
let mem: long = ma
r = Vk.bind_buffer_memory(gvk_dev, buf, gvk_mem_handle(ma), gvk_mem_offset(ma))
if r != VK_SUCCESS { return gvk_fail("vkBindBufferMemory", r) }
let pp = bytes(8)
r = Vk.map_memory(gvk_dev, mem, zero, size_l, 0, pp)
if r != VK_SUCCESS { return gvk_fail("vkMapMemory", r) }
gvk_buf[b] = buf; gvk_buf_mem[b] = mem; gvk_buf_size[b] = size; gvk_buf_map[b] = Vk.get_ptr(pp, 0)
gvk_buf[b] = buf; gvk_buf_mem[b] = mem; gvk_buf_size[b] = size; gvk_buf_map[b] = gvk_mem_ptr(ma)
return true
}

View file

@ -6,6 +6,7 @@
# the size of each cascade's depth layer; shadow_set_res changes it at run time
var shadow_res: int = 2048
var shadow_refused: int = 0 # the last size the graphics card had no memory for (0: none)
const SHADOW_CASCADES: int = 5
var sh_tex: int = 0
@ -42,10 +43,27 @@ function shadow_init() -> void {
# texel size from the map itself, so nothing else has to follow.
function shadow_set_res(r: int) -> void {
if r < 256 or r == shadow_res { return }
let was = shadow_res
shadow_res = r
if sh_tex == 0 { return }
gpu_tex_free(sh_tex)
shadow_make_tex()
# not enough video memory for that size: go back to the one that worked, and smaller again if even
# that is refused now, rather than ending with no shadow map at all
if not gpu_tex_ok(sh_tex) {
shadow_refused = r
print(`r3d: shadows: no memory for {r} x {r} cascades; keeping {was}`)
var size = was
shadow_res = size
gpu_tex_free(sh_tex)
shadow_make_tex()
while not gpu_tex_ok(sh_tex) and size > 512 {
size = size / 2
shadow_res = size
gpu_tex_free(sh_tex)
shadow_make_tex()
}
}
}
function shadow_make_tex() -> void {