diff --git a/changes/moltenvk-no-command-pooling.md b/changes/moltenvk-no-command-pooling.md new file mode 100644 index 00000000..160b6e63 --- /dev/null +++ b/changes/moltenvk-no-command-pooling.md @@ -0,0 +1,8 @@ +bump: patch +type: fix +**A frame that draws more than any before no longer grows the heap on the Mac.** MoltenVK's command +pooling kept every command object a frame had ever recorded - about 650 bytes for each draw beyond the +busiest frame so far, for as long as the game ran. render3d turns it off +(`MVK_CONFIG_USE_COMMAND_POOLING=0`, unless the environment already says otherwise) before the first +Vulkan call; the objects are made and freed with their command buffer, at no measured cost. +`examples/rendering/steady.ludic` ramps a frame from 20 to 200 actors and fails on what pooling left. diff --git a/examples/rendering/steady.ludic b/examples/rendering/steady.ludic index e9c03857..747cfa73 100644 --- a/examples/rendering/steady.ludic +++ b/examples/rendering/steady.ludic @@ -11,7 +11,7 @@ program Steady { property Marker { on: int = 1 } model Anchor { Marker } - function scene_draw(render3d_st: mut Render3dState) -> void { } + function scene_draw(render3d_st: mut Render3dState) -> void { actor_draw(render3d_st) } function scene_draw_casters(render3d_st: mut Render3dState, light_vp: floats) -> void { } function stream_fill(s: Stream, cx: int, cz: int, band: int) -> void { } @@ -70,6 +70,29 @@ program Steady { return settled(render3d_st) - before } + # bytes gained while a frame draws more and more: 20 actors to 200, 30 frames at each count. + # MoltenVK's command pooling kept every command a frame had recorded (~650 bytes a draw more). + function ramp_rounds(render3d_st: mut Render3dState, m: Model) -> long { + let acts = new []Actor + for i in 0 .. 200 { + let a = actor_new(render3d_st, m) + a.cull = 0.0 + a.visible = false + actor_place(a, float(i % 20) - 10.0, 0.0, -float(i / 20), 0.0) + push(acts, a) + } + for i in 0 .. 20 { acts[i].visible = true } + frame_rounds(render3d_st, 30) + let before = settled(render3d_st) + for k in 2 .. 11 { + for i in 0 .. k * 20 { acts[i].visible = true } + frame_rounds(render3d_st, 30) + } + let grew = settled(render3d_st) - before + for i in 0 .. 200 { actor_release(render3d_st, acts[i]) } + return grew + } + # bytes gained over n parses of a glTF document, each freed whole (Json.free_all): strings too function parse_rounds(render3d_st: Render3dState, text: string, n: int) -> long { let before = settled(render3d_st) @@ -107,10 +130,11 @@ program Steady { let am = gltf_load(render3d_st, "packages/ludic.lab/plate", "plate.gltf", "plate") actor_rounds(render3d_st, am, 20) let grew_a = actor_rounds(render3d_st, am, 2000) + let grew_r = ramp_rounds(render3d_st, am) let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf") parse_rounds(render3d_st, text, 20) let grew_p = parse_rounds(render3d_st, text, 200) - print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000`) + print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}`) # a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator) var ok = grew_b < 16384 if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") } @@ -132,6 +156,10 @@ program Steady { ok = false print("steady: FAILED - an actor placed and released leaves memory behind") } + if grew_r >= 16384 { + ok = false + print("steady: FAILED - a frame drawing more than before leaves memory behind") + } if ok { print("STEADY OK") } else { print("STEADY FAILED") } quit() } diff --git a/packages/ludic.render3d/gpu_vk.ludic b/packages/ludic.render3d/gpu_vk.ludic index 35192776..ff18337f 100644 --- a/packages/ludic.render3d/gpu_vk.ludic +++ b/packages/ludic.render3d/gpu_vk.ludic @@ -48,8 +48,18 @@ function gvk_surface_ext() -> string { return VK_KHR_WIN32_SURFACE_EXTENSION_NAME } +# MoltenVK's command pooling keeps every command object a frame ever recorded and never gives one +# back, so the heap grew each time a frame drew more than any before (~650 bytes a draw, for as long +# as the game ran). Off, the objects are made and freed with their command buffer, at no measured +# cost (3.7 ms either way over 520 actors). Read when the library loads, so before any Vk call. +function gvk_no_command_pooling() -> void { + if Os.platform() != "macos" or Os.has_env("MVK_CONFIG_USE_COMMAND_POOLING") { return } + Os.set_env("MVK_CONFIG_USE_COMMAND_POOLING", "0") +} + function gvk_init(render3d_st: mut Render3dState) -> bool { if render3d_st.gvk_ready { return true } + gvk_no_command_pooling() gsl_boot(render3d_st) if Vk.open() == 0 { render3d_st.gvk_why = "no Vulkan loader"; return false } gsl_init(render3d_st)