perf(render3d): Vulkan buffer uploads and per-frame mipmaps no longer wait on the GPU
- A buffer re-uploaded after a draw this frame read it gets fresh storage, as OpenGL orphans one; storage moved off or freed while the frame still reads it is destroyed after the frame's submit. Draws mark the buffers they bind. No flush for buffer work. - Mipmaps asked for mid-frame (the exposure measure, every frame) are recorded into the open frame after its pass; growing a chain the first time keeps its one-shot path. - R3D_VK_PROF prints, every 120 frames, draws and flushes per frame and the milliseconds spent finding pipelines, filling descriptor sets and inside draws. The gain was small - the camp view headless at 1920x1080 on the RTX 3070 Ti went from 21.3 to 21.7 fps (OpenGL: 114 fps) - so these flushes were not what holds the frame; the profile is how the rest is found. The frame is unchanged and validation-clean; OpenGL frames byte-identical at the five viewpoints with 59 self-tests passing. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
1b7408c6a1
commit
6942c36853
3 changed files with 100 additions and 18 deletions
|
|
@ -432,7 +432,7 @@ function gpu_mesh_new() -> Mesh {
|
|||
# a vertex buffer for the mesh being built (data may be null: storage only); returns it
|
||||
function gpu_mesh_vertices(m: Mesh, data: pointer, nbytes: int, usage: int) -> int {
|
||||
var b = 0
|
||||
if gpu_kind == GPU_VK { gvk_flush(); b = gvk_buf_new(); gvk_buf_upload(b, nbytes, data) }
|
||||
if gpu_kind == GPU_VK { b = gvk_buf_new(); gvk_buf_upload(b, nbytes, data) }
|
||||
else {
|
||||
b = gl_buffer()
|
||||
gl_bind_buffer(GL_ARRAY_BUFFER, b)
|
||||
|
|
@ -464,7 +464,7 @@ function gpu_mesh_attr(m: Mesh, index: int, comps: int, type: int, stride: int,
|
|||
function gpu_mesh_indices(m: Mesh, data: pointer, nbytes: int, index_bytes: int) -> void {
|
||||
m.itype = GL_UNSIGNED_INT
|
||||
if index_bytes == 2 { m.itype = GL_UNSIGNED_SHORT }
|
||||
if gpu_kind == GPU_VK { gvk_flush(); m.ebo = gvk_buf_new(); gvk_buf_upload(m.ebo, nbytes, data); return }
|
||||
if gpu_kind == GPU_VK { m.ebo = gvk_buf_new(); gvk_buf_upload(m.ebo, nbytes, data); return }
|
||||
m.ebo = gl_buffer()
|
||||
gl_bind_buffer(GL_ELEMENT_ARRAY_BUFFER, m.ebo)
|
||||
gl_buffer_data(GL_ELEMENT_ARRAY_BUFFER, nbytes, data, GL_STATIC_DRAW)
|
||||
|
|
@ -495,12 +495,12 @@ function gpu_mesh_attr_inst(m: Mesh, index: int, comps: int, type: int, stride:
|
|||
# a buffer on its own (instances, a stream): made, filled whole, freed
|
||||
function gpu_buffer_new() -> int { if gpu_kind == GPU_VK { return gvk_buf_new() }; return gl_buffer() }
|
||||
function gpu_buffer_upload(buf: int, nbytes: int, data: pointer, usage: int) -> void {
|
||||
if gpu_kind == GPU_VK { gvk_flush(); gvk_buf_upload(buf, nbytes, data); return }
|
||||
if gpu_kind == GPU_VK { gvk_buf_upload(buf, nbytes, data); return }
|
||||
gl_bind_buffer(GL_ARRAY_BUFFER, buf)
|
||||
gl_buffer_data(GL_ARRAY_BUFFER, nbytes, data, gpu_gl_usage(usage))
|
||||
}
|
||||
function gpu_buffer_free(buf: int) -> void {
|
||||
if gpu_kind == GPU_VK { gvk_flush(); if buf > 0 { gvk_buf_release(buf) }; return }
|
||||
if gpu_kind == GPU_VK { if buf > 0 { gvk_buf_release(buf) }; return }
|
||||
if buf == 0 { return }
|
||||
let ids = gpu_tmp()
|
||||
ids[0] = buf
|
||||
|
|
@ -542,7 +542,7 @@ function gpu_draw_range(m: Mesh, first: int, count: int) -> void {
|
|||
}
|
||||
|
||||
function gpu_mesh_free(m: Mesh) -> void {
|
||||
if gpu_kind == GPU_VK { gvk_flush(); gvk_mesh_free(m); return }
|
||||
if gpu_kind == GPU_VK { gvk_mesh_free(m); return }
|
||||
if m == null { return }
|
||||
let ids = gpu_tmp()
|
||||
if m.vbufs != null {
|
||||
|
|
@ -645,10 +645,9 @@ function gpu_tex_paramf(kind: int, pname: int, value: fixed) -> void {
|
|||
function gpu_tex_border(kind: int, rgba: pointer) -> void { if gpu_kind == GPU_VK { return }; gl_tex_parameterfv(gpu_gl_target(kind), GL_TEXTURE_BORDER_COLOR, rgba) }
|
||||
function gpu_tex_mips(kind: int) -> void {
|
||||
if gpu_kind == GPU_VK {
|
||||
gvk_flush()
|
||||
let mt = gpu_bound(kind)
|
||||
let mo = gpu_tx_at(mt)
|
||||
if mo >= 0 { gvk_tex_mips(mt, gpu_tx[mo + 1], gpu_tx[mo + 2]) }
|
||||
if mo >= 0 { gvk_mips_now(mt, gpu_tx[mo + 1], gpu_tx[mo + 2]) }
|
||||
} else { gl_generate_mipmap(gpu_gl_target(kind)) }
|
||||
gpu_glcheck_after(`mipmaps for texture {gpu_bound(kind)}`)
|
||||
let o = gpu_tx_at(gpu_bound(kind))
|
||||
|
|
|
|||
|
|
@ -826,15 +826,21 @@ function gvk_set_view(cb: pointer) -> void {
|
|||
# first / count select vertices or indices; count 0 means the mesh's own count. instances >= 1.
|
||||
function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instances: int, tx: words, tx_w: int, tx_cap: int, rec: words, rec_o: int) -> void {
|
||||
if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) or gvk_prog_var[p] == null { return }
|
||||
let prof = gvk_prof()
|
||||
var t0: long = 0
|
||||
if prof { t0 = gl_now_us() }
|
||||
gvk_pass_begin(rec, rec_o)
|
||||
let cb = gvk_cb
|
||||
let samples = 1
|
||||
let pipe = gvk_pipeline(p, m, st, gvk_pass_ncolor, gvk_pass_cfmt, gvk_pass_dfmt, samples)
|
||||
if pipe == 0 { return }
|
||||
var t1: long = 0
|
||||
if prof { t1 = gl_now_us(); gvk_us_pipe = gvk_us_pipe + (t1 - t0) }
|
||||
Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, pipe)
|
||||
gvk_set_view(cb)
|
||||
let set = gvk_draw_set(p, tx, tx_w, tx_cap)
|
||||
if set == 0 { return }
|
||||
if prof { gvk_us_set = gvk_us_set + (gl_now_us() - t1) }
|
||||
let sets = bytes(8)
|
||||
Vk.put_i64(sets, 0, set)
|
||||
Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, gvk_prog_layout[p], 0, 1, sets, 0, null)
|
||||
|
|
@ -857,6 +863,7 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc
|
|||
if not known and nbd < GPU_MAX_VBUFS {
|
||||
seen[nbd] = m.attrs[o]
|
||||
Vk.put_i64(bufs, nbd * 8, gvk_buf[m.attrs[o]])
|
||||
gvk_buf_used[m.attrs[o]] = gvk_frame_no
|
||||
nbd += 1
|
||||
}
|
||||
}
|
||||
|
|
@ -869,10 +876,12 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc
|
|||
var itype = VK_INDEX_TYPE_UINT32
|
||||
if m.itype == GL_UNSIGNED_SHORT { itype = VK_INDEX_TYPE_UINT16 }
|
||||
Vk.cmd_bind_index_buffer(cb, gvk_buf[m.ebo], zero, itype)
|
||||
gvk_buf_used[m.ebo] = gvk_frame_no
|
||||
Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0)
|
||||
} else {
|
||||
Vk.cmd_draw(cb, n, instances, first, 0)
|
||||
}
|
||||
if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) }
|
||||
}
|
||||
|
||||
# The frame so far, submitted and waited for, so work that submits on its own - an upload, a
|
||||
|
|
@ -880,20 +889,26 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc
|
|||
# order OpenGL would have done them. Costs a submit per such call while the backend comes up.
|
||||
function gvk_flush() -> void {
|
||||
if gvk_cb == null { return }
|
||||
gvk_n_flush += 1
|
||||
gvk_pass_end()
|
||||
gvk_once_end(gvk_cb)
|
||||
gvk_cb = null
|
||||
gvk_frame_no += 1
|
||||
gvk_retire_flush()
|
||||
}
|
||||
|
||||
# The finished frame: submitted and waited for. With a window, the screen image is blitted into
|
||||
# the swapchain's next image first - flipped, since the screen image keeps OpenGL's bottom-up rows
|
||||
# - and that image is presented.
|
||||
function gvk_present() -> void {
|
||||
if gvk_swap != 0 { gvk_present_window(); return }
|
||||
gvk_prof_frame()
|
||||
if gvk_swap != 0 { gvk_present_window(); gvk_frame_no += 1; gvk_retire_flush(); return }
|
||||
if gvk_cb == null { return }
|
||||
gvk_pass_end()
|
||||
gvk_once_end(gvk_cb)
|
||||
gvk_cb = null
|
||||
gvk_frame_no += 1
|
||||
gvk_retire_flush()
|
||||
}
|
||||
|
||||
# The screen as a binary PPM, top row first. The frame so far is finished first, then read back;
|
||||
|
|
@ -1253,3 +1268,40 @@ function gvk_resize_check() -> bool {
|
|||
gvk_swap_make(w, h)
|
||||
return true
|
||||
}
|
||||
|
||||
# Mipmaps for a texture the renderer asks for mid-frame (the exposure measure reads the HDR scene's
|
||||
# smallest level every frame): recorded into the open frame after its pass, so they cost no submit.
|
||||
# A chain that has to grow first still goes through its one-shot path.
|
||||
function gvk_mips_now(tex: int, w: int, h: int) -> void {
|
||||
if gvk_cb == null or gvk_tex_levels[tex] <= 1 { gvk_flush(); gvk_tex_mips(tex, w, h); return }
|
||||
gvk_pass_end()
|
||||
gvk_tex_mips_into(gvk_cb, tex, w, h)
|
||||
}
|
||||
|
||||
# ---- R3D_VK_PROF: where a Vulkan frame's CPU time goes ---------------------------------------------
|
||||
# Every 120 frames: draws and flushes per frame, and milliseconds per frame spent finding pipelines,
|
||||
# filling descriptor sets, and inside draws altogether. Off, it costs one flag test a draw.
|
||||
var gvk_prof_state: int = -1
|
||||
var gvk_us_pipe: long = 0
|
||||
var gvk_us_set: long = 0
|
||||
var gvk_us_draw: long = 0
|
||||
var gvk_n_draws: int = 0
|
||||
var gvk_n_flush: int = 0
|
||||
var gvk_prof_frames: int = 0
|
||||
function gvk_prof() -> bool {
|
||||
if gvk_prof_state < 0 { gvk_prof_state = 0; if Os.has_env("R3D_VK_PROF") { gvk_prof_state = 1 } }
|
||||
return gvk_prof_state == 1
|
||||
}
|
||||
function gvk_prof_frame() -> void {
|
||||
if not gvk_prof() { return }
|
||||
gvk_prof_frames += 1
|
||||
if gvk_prof_frames < 120 { return }
|
||||
let f = gvk_prof_frames
|
||||
let pipe = Text.to_int(string(gvk_us_pipe)) / f
|
||||
let set = Text.to_int(string(gvk_us_set)) / f
|
||||
let draw = Text.to_int(string(gvk_us_draw)) / f
|
||||
print(`r3d: vulkan per frame: {gvk_n_draws / f} draws, {gvk_n_flush / f} flushes; pipelines {pipe / 1000}.{(pipe / 100) % 10} ms, sets {set / 1000}.{(set / 100) % 10} ms, inside draws {draw / 1000}.{(draw / 100) % 10} ms`)
|
||||
let zero: long = 0
|
||||
gvk_us_pipe = zero; gvk_us_set = zero; gvk_us_draw = zero
|
||||
gvk_n_draws = 0; gvk_n_flush = 0; gvk_prof_frames = 0
|
||||
}
|
||||
|
|
|
|||
|
|
@ -331,9 +331,16 @@ function gvk_tex_mips(tex: int, w: int, h: int) -> bool {
|
|||
}
|
||||
let levels = gvk_tex_levels[tex]
|
||||
if levels <= 1 { return true }
|
||||
let cb = gvk_once_begin()
|
||||
gvk_tex_mips_into(cb, tex, w, h)
|
||||
return gvk_once_end(cb)
|
||||
}
|
||||
# the chain's blits and barriers recorded into cb (the frame's own, when one is open)
|
||||
function gvk_tex_mips_into(cb: pointer, tex: int, w: int, h: int) -> void {
|
||||
let levels = gvk_tex_levels[tex]
|
||||
if levels <= 1 { return }
|
||||
let image = gvk_tex_image[tex]
|
||||
let layers = gvk_tex_layers[tex]
|
||||
let cb = gvk_once_begin()
|
||||
var sw = w
|
||||
var sh = h
|
||||
for lv in 1 .. levels {
|
||||
|
|
@ -363,7 +370,6 @@ function gvk_tex_mips(tex: int, w: int, h: int) -> bool {
|
|||
sw = dw
|
||||
sh = dh
|
||||
}
|
||||
return gvk_once_end(cb)
|
||||
}
|
||||
|
||||
# Level 0 of layer 0 back to the CPU, laid out the OpenGL way when the layouts agree (the terrain
|
||||
|
|
@ -491,33 +497,58 @@ var gvk_buf: []long = null
|
|||
var gvk_buf_mem: []long = null
|
||||
var gvk_buf_size: []int = null
|
||||
var gvk_buf_map: []pointer = null # host-visible buffers stay mapped for their whole life
|
||||
var gvk_buf_used: []int = null # the frame (gvk_frame_no) a draw last bound the buffer in
|
||||
var gvk_frame_no: int = 1
|
||||
# Storage a buffer was moved off, or freed, while the frame that used it has not been submitted:
|
||||
# destroyed at gvk_retire_flush, after the frame's work is done.
|
||||
var gvk_retired_buf: []long = null
|
||||
var gvk_retired_mem: []long = null
|
||||
|
||||
function gvk_buf_new() -> int {
|
||||
let zero: long = 0
|
||||
if gvk_buf == null {
|
||||
gvk_buf = new []long; gvk_buf_mem = new []long; gvk_buf_size = new []int; gvk_buf_map = new []pointer
|
||||
gvk_buf_used = new []int; gvk_retired_buf = new []long; gvk_retired_mem = new []long
|
||||
# handle 0 is "no buffer", as it is on OpenGL
|
||||
push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null)
|
||||
push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null); push(gvk_buf_used, 0)
|
||||
}
|
||||
push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null)
|
||||
push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null); push(gvk_buf_used, 0)
|
||||
return len(gvk_buf) - 1
|
||||
}
|
||||
|
||||
function gvk_buf_release(b: int) -> void {
|
||||
if gvk_buf[b] == 0 { return }
|
||||
let zero: long = 0
|
||||
Vk.unmap_memory(gvk_dev, gvk_buf_mem[b])
|
||||
Vk.destroy_buffer(gvk_dev, gvk_buf[b], null)
|
||||
Vk.free_memory(gvk_dev, gvk_buf_mem[b], null)
|
||||
gvk_n_allocs -= 1
|
||||
gvk_buf[b] = zero; gvk_buf_mem[b] = zero; gvk_buf_size[b] = 0; gvk_buf_map[b] = null
|
||||
if gvk_buf_used[b] == gvk_frame_no {
|
||||
# a draw recorded this frame still reads it: destroy it once the frame has been submitted
|
||||
push(gvk_retired_buf, gvk_buf[b]); push(gvk_retired_mem, gvk_buf_mem[b])
|
||||
} else {
|
||||
Vk.unmap_memory(gvk_dev, gvk_buf_mem[b])
|
||||
Vk.destroy_buffer(gvk_dev, gvk_buf[b], null)
|
||||
Vk.free_memory(gvk_dev, gvk_buf_mem[b], null)
|
||||
gvk_n_allocs -= 1
|
||||
}
|
||||
gvk_buf[b] = zero; gvk_buf_mem[b] = zero; gvk_buf_size[b] = 0; gvk_buf_map[b] = null; gvk_buf_used[b] = 0
|
||||
}
|
||||
# the frame's submitted work is done: storage retired during it can go
|
||||
function gvk_retire_flush() -> void {
|
||||
if gvk_retired_buf == null { return }
|
||||
for i in 0 .. len(gvk_retired_buf) {
|
||||
Vk.unmap_memory(gvk_dev, gvk_retired_mem[i])
|
||||
Vk.destroy_buffer(gvk_dev, gvk_retired_buf[i], null)
|
||||
Vk.free_memory(gvk_dev, gvk_retired_mem[i], null)
|
||||
gvk_n_allocs -= 1
|
||||
}
|
||||
gvk_retired_buf = new []long
|
||||
gvk_retired_mem = new []long
|
||||
}
|
||||
|
||||
# Room for at least n bytes behind handle b; a buffer that is already big enough is kept, so a
|
||||
# stream re-filled every frame allocates once.
|
||||
function gvk_buf_reserve(b: int, n: int) -> bool {
|
||||
if b <= 0 or b >= len(gvk_buf) { return false }
|
||||
if gvk_buf[b] != 0 and gvk_buf_size[b] >= n { return true }
|
||||
# already big enough, and no draw this frame reads what is there: fill it in place
|
||||
if gvk_buf[b] != 0 and gvk_buf_size[b] >= n and gvk_buf_used[b] != gvk_frame_no { return true }
|
||||
gvk_buf_release(b)
|
||||
var size = n
|
||||
if size < 64 { size = 64 }
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue