perf(render3d): Vulkan buffer uploads and per-frame mipmaps no longer wait on the GPU

- A buffer re-uploaded after a draw this frame read it gets fresh storage, as OpenGL orphans
  one; storage moved off or freed while the frame still reads it is destroyed after the frame's
  submit. Draws mark the buffers they bind. No flush for buffer work.
- Mipmaps asked for mid-frame (the exposure measure, every frame) are recorded into the open frame
  after its pass; growing a chain the first time keeps its one-shot path.
- R3D_VK_PROF prints, every 120 frames, draws and flushes per frame and the milliseconds spent
  finding pipelines, filling descriptor sets and inside draws.

The gain was small - the camp view headless at 1920x1080 on the RTX 3070 Ti went from 21.3 to
21.7 fps (OpenGL: 114 fps) - so these flushes were not what holds the frame; the profile is how
the rest is found. The frame is unchanged and validation-clean; OpenGL frames byte-identical at the
five viewpoints with 59 self-tests passing.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-15 13:17:07 +03:00
parent 1b7408c6a1
commit 6942c36853
3 changed files with 100 additions and 18 deletions

View file

@ -826,15 +826,21 @@ function gvk_set_view(cb: pointer) -> void {
# first / count select vertices or indices; count 0 means the mesh's own count. instances >= 1.
function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instances: int, tx: words, tx_w: int, tx_cap: int, rec: words, rec_o: int) -> void {
if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) or gvk_prog_var[p] == null { return }
let prof = gvk_prof()
var t0: long = 0
if prof { t0 = gl_now_us() }
gvk_pass_begin(rec, rec_o)
let cb = gvk_cb
let samples = 1
let pipe = gvk_pipeline(p, m, st, gvk_pass_ncolor, gvk_pass_cfmt, gvk_pass_dfmt, samples)
if pipe == 0 { return }
var t1: long = 0
if prof { t1 = gl_now_us(); gvk_us_pipe = gvk_us_pipe + (t1 - t0) }
Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, pipe)
gvk_set_view(cb)
let set = gvk_draw_set(p, tx, tx_w, tx_cap)
if set == 0 { return }
if prof { gvk_us_set = gvk_us_set + (gl_now_us() - t1) }
let sets = bytes(8)
Vk.put_i64(sets, 0, set)
Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, gvk_prog_layout[p], 0, 1, sets, 0, null)
@ -857,6 +863,7 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc
if not known and nbd < GPU_MAX_VBUFS {
seen[nbd] = m.attrs[o]
Vk.put_i64(bufs, nbd * 8, gvk_buf[m.attrs[o]])
gvk_buf_used[m.attrs[o]] = gvk_frame_no
nbd += 1
}
}
@ -869,10 +876,12 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc
var itype = VK_INDEX_TYPE_UINT32
if m.itype == GL_UNSIGNED_SHORT { itype = VK_INDEX_TYPE_UINT16 }
Vk.cmd_bind_index_buffer(cb, gvk_buf[m.ebo], zero, itype)
gvk_buf_used[m.ebo] = gvk_frame_no
Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0)
} else {
Vk.cmd_draw(cb, n, instances, first, 0)
}
if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) }
}
# The frame so far, submitted and waited for, so work that submits on its own - an upload, a
@ -880,20 +889,26 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc
# order OpenGL would have done them. Costs a submit per such call while the backend comes up.
function gvk_flush() -> void {
if gvk_cb == null { return }
gvk_n_flush += 1
gvk_pass_end()
gvk_once_end(gvk_cb)
gvk_cb = null
gvk_frame_no += 1
gvk_retire_flush()
}
# The finished frame: submitted and waited for. With a window, the screen image is blitted into
# the swapchain's next image first - flipped, since the screen image keeps OpenGL's bottom-up rows
# - and that image is presented.
function gvk_present() -> void {
if gvk_swap != 0 { gvk_present_window(); return }
gvk_prof_frame()
if gvk_swap != 0 { gvk_present_window(); gvk_frame_no += 1; gvk_retire_flush(); return }
if gvk_cb == null { return }
gvk_pass_end()
gvk_once_end(gvk_cb)
gvk_cb = null
gvk_frame_no += 1
gvk_retire_flush()
}
# The screen as a binary PPM, top row first. The frame so far is finished first, then read back;
@ -1253,3 +1268,40 @@ function gvk_resize_check() -> bool {
gvk_swap_make(w, h)
return true
}
# Mipmaps for a texture the renderer asks for mid-frame (the exposure measure reads the HDR scene's
# smallest level every frame): recorded into the open frame after its pass, so they cost no submit.
# A chain that has to grow first still goes through its one-shot path.
function gvk_mips_now(tex: int, w: int, h: int) -> void {
if gvk_cb == null or gvk_tex_levels[tex] <= 1 { gvk_flush(); gvk_tex_mips(tex, w, h); return }
gvk_pass_end()
gvk_tex_mips_into(gvk_cb, tex, w, h)
}
# ---- R3D_VK_PROF: where a Vulkan frame's CPU time goes ---------------------------------------------
# Every 120 frames: draws and flushes per frame, and milliseconds per frame spent finding pipelines,
# filling descriptor sets, and inside draws altogether. Off, it costs one flag test a draw.
var gvk_prof_state: int = -1
var gvk_us_pipe: long = 0
var gvk_us_set: long = 0
var gvk_us_draw: long = 0
var gvk_n_draws: int = 0
var gvk_n_flush: int = 0
var gvk_prof_frames: int = 0
function gvk_prof() -> bool {
if gvk_prof_state < 0 { gvk_prof_state = 0; if Os.has_env("R3D_VK_PROF") { gvk_prof_state = 1 } }
return gvk_prof_state == 1
}
function gvk_prof_frame() -> void {
if not gvk_prof() { return }
gvk_prof_frames += 1
if gvk_prof_frames < 120 { return }
let f = gvk_prof_frames
let pipe = Text.to_int(string(gvk_us_pipe)) / f
let set = Text.to_int(string(gvk_us_set)) / f
let draw = Text.to_int(string(gvk_us_draw)) / f
print(`r3d: vulkan per frame: {gvk_n_draws / f} draws, {gvk_n_flush / f} flushes; pipelines {pipe / 1000}.{(pipe / 100) % 10} ms, sets {set / 1000}.{(set / 100) % 10} ms, inside draws {draw / 1000}.{(draw / 100) % 10} ms`)
let zero: long = 0
gvk_us_pipe = zero; gvk_us_set = zero; gvk_us_draw = zero
gvk_n_draws = 0; gvk_n_flush = 0; gvk_prof_frames = 0
}