diff --git a/packages/ludic.render3d/gpu.ludic b/packages/ludic.render3d/gpu.ludic index e75b0bb0..6633aa6f 100644 --- a/packages/ludic.render3d/gpu.ludic +++ b/packages/ludic.render3d/gpu.ludic @@ -432,7 +432,7 @@ function gpu_mesh_new() -> Mesh { # a vertex buffer for the mesh being built (data may be null: storage only); returns it function gpu_mesh_vertices(m: Mesh, data: pointer, nbytes: int, usage: int) -> int { var b = 0 - if gpu_kind == GPU_VK { gvk_flush(); b = gvk_buf_new(); gvk_buf_upload(b, nbytes, data) } + if gpu_kind == GPU_VK { b = gvk_buf_new(); gvk_buf_upload(b, nbytes, data) } else { b = gl_buffer() gl_bind_buffer(GL_ARRAY_BUFFER, b) @@ -464,7 +464,7 @@ function gpu_mesh_attr(m: Mesh, index: int, comps: int, type: int, stride: int, function gpu_mesh_indices(m: Mesh, data: pointer, nbytes: int, index_bytes: int) -> void { m.itype = GL_UNSIGNED_INT if index_bytes == 2 { m.itype = GL_UNSIGNED_SHORT } - if gpu_kind == GPU_VK { gvk_flush(); m.ebo = gvk_buf_new(); gvk_buf_upload(m.ebo, nbytes, data); return } + if gpu_kind == GPU_VK { m.ebo = gvk_buf_new(); gvk_buf_upload(m.ebo, nbytes, data); return } m.ebo = gl_buffer() gl_bind_buffer(GL_ELEMENT_ARRAY_BUFFER, m.ebo) gl_buffer_data(GL_ELEMENT_ARRAY_BUFFER, nbytes, data, GL_STATIC_DRAW) @@ -495,12 +495,12 @@ function gpu_mesh_attr_inst(m: Mesh, index: int, comps: int, type: int, stride: # a buffer on its own (instances, a stream): made, filled whole, freed function gpu_buffer_new() -> int { if gpu_kind == GPU_VK { return gvk_buf_new() }; return gl_buffer() } function gpu_buffer_upload(buf: int, nbytes: int, data: pointer, usage: int) -> void { - if gpu_kind == GPU_VK { gvk_flush(); gvk_buf_upload(buf, nbytes, data); return } + if gpu_kind == GPU_VK { gvk_buf_upload(buf, nbytes, data); return } gl_bind_buffer(GL_ARRAY_BUFFER, buf) gl_buffer_data(GL_ARRAY_BUFFER, nbytes, data, gpu_gl_usage(usage)) } function gpu_buffer_free(buf: int) -> void { - if gpu_kind == GPU_VK { gvk_flush(); if buf > 0 { gvk_buf_release(buf) }; return } + if gpu_kind == GPU_VK { if buf > 0 { gvk_buf_release(buf) }; return } if buf == 0 { return } let ids = gpu_tmp() ids[0] = buf @@ -542,7 +542,7 @@ function gpu_draw_range(m: Mesh, first: int, count: int) -> void { } function gpu_mesh_free(m: Mesh) -> void { - if gpu_kind == GPU_VK { gvk_flush(); gvk_mesh_free(m); return } + if gpu_kind == GPU_VK { gvk_mesh_free(m); return } if m == null { return } let ids = gpu_tmp() if m.vbufs != null { @@ -645,10 +645,9 @@ function gpu_tex_paramf(kind: int, pname: int, value: fixed) -> void { function gpu_tex_border(kind: int, rgba: pointer) -> void { if gpu_kind == GPU_VK { return }; gl_tex_parameterfv(gpu_gl_target(kind), GL_TEXTURE_BORDER_COLOR, rgba) } function gpu_tex_mips(kind: int) -> void { if gpu_kind == GPU_VK { - gvk_flush() let mt = gpu_bound(kind) let mo = gpu_tx_at(mt) - if mo >= 0 { gvk_tex_mips(mt, gpu_tx[mo + 1], gpu_tx[mo + 2]) } + if mo >= 0 { gvk_mips_now(mt, gpu_tx[mo + 1], gpu_tx[mo + 2]) } } else { gl_generate_mipmap(gpu_gl_target(kind)) } gpu_glcheck_after(`mipmaps for texture {gpu_bound(kind)}`) let o = gpu_tx_at(gpu_bound(kind)) diff --git a/packages/ludic.render3d/gpu_vk_draw.ludic b/packages/ludic.render3d/gpu_vk_draw.ludic index 28b4d332..288370e8 100644 --- a/packages/ludic.render3d/gpu_vk_draw.ludic +++ b/packages/ludic.render3d/gpu_vk_draw.ludic @@ -826,15 +826,21 @@ function gvk_set_view(cb: pointer) -> void { # first / count select vertices or indices; count 0 means the mesh's own count. instances >= 1. function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instances: int, tx: words, tx_w: int, tx_cap: int, rec: words, rec_o: int) -> void { if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) or gvk_prog_var[p] == null { return } + let prof = gvk_prof() + var t0: long = 0 + if prof { t0 = gl_now_us() } gvk_pass_begin(rec, rec_o) let cb = gvk_cb let samples = 1 let pipe = gvk_pipeline(p, m, st, gvk_pass_ncolor, gvk_pass_cfmt, gvk_pass_dfmt, samples) if pipe == 0 { return } + var t1: long = 0 + if prof { t1 = gl_now_us(); gvk_us_pipe = gvk_us_pipe + (t1 - t0) } Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, pipe) gvk_set_view(cb) let set = gvk_draw_set(p, tx, tx_w, tx_cap) if set == 0 { return } + if prof { gvk_us_set = gvk_us_set + (gl_now_us() - t1) } let sets = bytes(8) Vk.put_i64(sets, 0, set) Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, gvk_prog_layout[p], 0, 1, sets, 0, null) @@ -857,6 +863,7 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc if not known and nbd < GPU_MAX_VBUFS { seen[nbd] = m.attrs[o] Vk.put_i64(bufs, nbd * 8, gvk_buf[m.attrs[o]]) + gvk_buf_used[m.attrs[o]] = gvk_frame_no nbd += 1 } } @@ -869,10 +876,12 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc var itype = VK_INDEX_TYPE_UINT32 if m.itype == GL_UNSIGNED_SHORT { itype = VK_INDEX_TYPE_UINT16 } Vk.cmd_bind_index_buffer(cb, gvk_buf[m.ebo], zero, itype) + gvk_buf_used[m.ebo] = gvk_frame_no Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0) } else { Vk.cmd_draw(cb, n, instances, first, 0) } + if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) } } # The frame so far, submitted and waited for, so work that submits on its own - an upload, a @@ -880,20 +889,26 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc # order OpenGL would have done them. Costs a submit per such call while the backend comes up. function gvk_flush() -> void { if gvk_cb == null { return } + gvk_n_flush += 1 gvk_pass_end() gvk_once_end(gvk_cb) gvk_cb = null + gvk_frame_no += 1 + gvk_retire_flush() } # The finished frame: submitted and waited for. With a window, the screen image is blitted into # the swapchain's next image first - flipped, since the screen image keeps OpenGL's bottom-up rows # - and that image is presented. function gvk_present() -> void { - if gvk_swap != 0 { gvk_present_window(); return } + gvk_prof_frame() + if gvk_swap != 0 { gvk_present_window(); gvk_frame_no += 1; gvk_retire_flush(); return } if gvk_cb == null { return } gvk_pass_end() gvk_once_end(gvk_cb) gvk_cb = null + gvk_frame_no += 1 + gvk_retire_flush() } # The screen as a binary PPM, top row first. The frame so far is finished first, then read back; @@ -1253,3 +1268,40 @@ function gvk_resize_check() -> bool { gvk_swap_make(w, h) return true } + +# Mipmaps for a texture the renderer asks for mid-frame (the exposure measure reads the HDR scene's +# smallest level every frame): recorded into the open frame after its pass, so they cost no submit. +# A chain that has to grow first still goes through its one-shot path. +function gvk_mips_now(tex: int, w: int, h: int) -> void { + if gvk_cb == null or gvk_tex_levels[tex] <= 1 { gvk_flush(); gvk_tex_mips(tex, w, h); return } + gvk_pass_end() + gvk_tex_mips_into(gvk_cb, tex, w, h) +} + +# ---- R3D_VK_PROF: where a Vulkan frame's CPU time goes --------------------------------------------- +# Every 120 frames: draws and flushes per frame, and milliseconds per frame spent finding pipelines, +# filling descriptor sets, and inside draws altogether. Off, it costs one flag test a draw. +var gvk_prof_state: int = -1 +var gvk_us_pipe: long = 0 +var gvk_us_set: long = 0 +var gvk_us_draw: long = 0 +var gvk_n_draws: int = 0 +var gvk_n_flush: int = 0 +var gvk_prof_frames: int = 0 +function gvk_prof() -> bool { + if gvk_prof_state < 0 { gvk_prof_state = 0; if Os.has_env("R3D_VK_PROF") { gvk_prof_state = 1 } } + return gvk_prof_state == 1 +} +function gvk_prof_frame() -> void { + if not gvk_prof() { return } + gvk_prof_frames += 1 + if gvk_prof_frames < 120 { return } + let f = gvk_prof_frames + let pipe = Text.to_int(string(gvk_us_pipe)) / f + let set = Text.to_int(string(gvk_us_set)) / f + let draw = Text.to_int(string(gvk_us_draw)) / f + print(`r3d: vulkan per frame: {gvk_n_draws / f} draws, {gvk_n_flush / f} flushes; pipelines {pipe / 1000}.{(pipe / 100) % 10} ms, sets {set / 1000}.{(set / 100) % 10} ms, inside draws {draw / 1000}.{(draw / 100) % 10} ms`) + let zero: long = 0 + gvk_us_pipe = zero; gvk_us_set = zero; gvk_us_draw = zero + gvk_n_draws = 0; gvk_n_flush = 0; gvk_prof_frames = 0 +} diff --git a/packages/ludic.render3d/gpu_vk_res.ludic b/packages/ludic.render3d/gpu_vk_res.ludic index 15a41612..6e7534e7 100644 --- a/packages/ludic.render3d/gpu_vk_res.ludic +++ b/packages/ludic.render3d/gpu_vk_res.ludic @@ -331,9 +331,16 @@ function gvk_tex_mips(tex: int, w: int, h: int) -> bool { } let levels = gvk_tex_levels[tex] if levels <= 1 { return true } + let cb = gvk_once_begin() + gvk_tex_mips_into(cb, tex, w, h) + return gvk_once_end(cb) +} +# the chain's blits and barriers recorded into cb (the frame's own, when one is open) +function gvk_tex_mips_into(cb: pointer, tex: int, w: int, h: int) -> void { + let levels = gvk_tex_levels[tex] + if levels <= 1 { return } let image = gvk_tex_image[tex] let layers = gvk_tex_layers[tex] - let cb = gvk_once_begin() var sw = w var sh = h for lv in 1 .. levels { @@ -363,7 +370,6 @@ function gvk_tex_mips(tex: int, w: int, h: int) -> bool { sw = dw sh = dh } - return gvk_once_end(cb) } # Level 0 of layer 0 back to the CPU, laid out the OpenGL way when the layouts agree (the terrain @@ -491,33 +497,58 @@ var gvk_buf: []long = null var gvk_buf_mem: []long = null var gvk_buf_size: []int = null var gvk_buf_map: []pointer = null # host-visible buffers stay mapped for their whole life +var gvk_buf_used: []int = null # the frame (gvk_frame_no) a draw last bound the buffer in +var gvk_frame_no: int = 1 +# Storage a buffer was moved off, or freed, while the frame that used it has not been submitted: +# destroyed at gvk_retire_flush, after the frame's work is done. +var gvk_retired_buf: []long = null +var gvk_retired_mem: []long = null function gvk_buf_new() -> int { let zero: long = 0 if gvk_buf == null { gvk_buf = new []long; gvk_buf_mem = new []long; gvk_buf_size = new []int; gvk_buf_map = new []pointer + gvk_buf_used = new []int; gvk_retired_buf = new []long; gvk_retired_mem = new []long # handle 0 is "no buffer", as it is on OpenGL - push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null) + push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null); push(gvk_buf_used, 0) } - push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null) + push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null); push(gvk_buf_used, 0) return len(gvk_buf) - 1 } function gvk_buf_release(b: int) -> void { if gvk_buf[b] == 0 { return } let zero: long = 0 - Vk.unmap_memory(gvk_dev, gvk_buf_mem[b]) - Vk.destroy_buffer(gvk_dev, gvk_buf[b], null) - Vk.free_memory(gvk_dev, gvk_buf_mem[b], null) - gvk_n_allocs -= 1 - gvk_buf[b] = zero; gvk_buf_mem[b] = zero; gvk_buf_size[b] = 0; gvk_buf_map[b] = null + if gvk_buf_used[b] == gvk_frame_no { + # a draw recorded this frame still reads it: destroy it once the frame has been submitted + push(gvk_retired_buf, gvk_buf[b]); push(gvk_retired_mem, gvk_buf_mem[b]) + } else { + Vk.unmap_memory(gvk_dev, gvk_buf_mem[b]) + Vk.destroy_buffer(gvk_dev, gvk_buf[b], null) + Vk.free_memory(gvk_dev, gvk_buf_mem[b], null) + gvk_n_allocs -= 1 + } + gvk_buf[b] = zero; gvk_buf_mem[b] = zero; gvk_buf_size[b] = 0; gvk_buf_map[b] = null; gvk_buf_used[b] = 0 +} +# the frame's submitted work is done: storage retired during it can go +function gvk_retire_flush() -> void { + if gvk_retired_buf == null { return } + for i in 0 .. len(gvk_retired_buf) { + Vk.unmap_memory(gvk_dev, gvk_retired_mem[i]) + Vk.destroy_buffer(gvk_dev, gvk_retired_buf[i], null) + Vk.free_memory(gvk_dev, gvk_retired_mem[i], null) + gvk_n_allocs -= 1 + } + gvk_retired_buf = new []long + gvk_retired_mem = new []long } # Room for at least n bytes behind handle b; a buffer that is already big enough is kept, so a # stream re-filled every frame allocates once. function gvk_buf_reserve(b: int, n: int) -> bool { if b <= 0 or b >= len(gvk_buf) { return false } - if gvk_buf[b] != 0 and gvk_buf_size[b] >= n { return true } + # already big enough, and no draw this frame reads what is there: fill it in place + if gvk_buf[b] != 0 and gvk_buf_size[b] >= n and gvk_buf_used[b] != gvk_frame_no { return true } gvk_buf_release(b) var size = n if size < 64 { size = 64 }