From 1ab5ca14fbb6ce2008c5c3db678ce41ecd39198a Mon Sep 17 00:00:00 2001 From: Orkuncakilkaya Date: Wed, 30 Sep 2026 00:41:27 +0300 Subject: [PATCH] runtime (macOS): MoltenVK opened first, so a process holds one - the SDK's loader in /usr/local/lib loaded its own MoltenVK beside the linked one and the program drew through it; the loader only when a layer is asked for, pinned to ours (VK_DRIVER_FILES, lib/macos-arm64/MoltenVK_icd.json) unless a driver is named. steady's stream round: a 300-cell warm-up, then the least of three 150-cell windows Co-Authored-By: Claude Opus 5.5 --- changes/one-moltenvk.md | 11 ++ examples/rendering/steady.ludic | 35 ++++-- .../lib/macos-arm64/MoltenVK_icd.json | 3 + runtime/native/vk_mac.ll | 102 ++++++++++++++++-- 4 files changed, 134 insertions(+), 17 deletions(-) create mode 100644 changes/one-moltenvk.md create mode 100644 packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json diff --git a/changes/one-moltenvk.md b/changes/one-moltenvk.md new file mode 100644 index 00000000..6633afdb --- /dev/null +++ b/changes/one-moltenvk.md @@ -0,0 +1,11 @@ +bump: patch +type: fix +**One MoltenVK in a process.** On a Mac with the Vulkan SDK installed, the runtime opened the SDK's loader +(/usr/local/lib/libvulkan.1.dylib) before MoltenVK, and the loader loaded the SDK's own MoltenVK as its +driver - beside ludic.render3d's, which every program is linked against: two copies, each with its pools, +and the program drew through the SDK's ("MVKBlockObserver is implemented in both"). MoltenVK is opened +first now (dlopen hands back the linked copy), and the loader only when a layer is asked for +(VK_INSTANCE_LAYERS, VK_LOADER_LAYERS_ENABLE, R3D_VK_LOADER) - then pinned to ludic.render3d's MoltenVK +through VK_DRIVER_FILES and lib/macos-arm64/MoltenVK_icd.json, unless the caller named a driver. +examples/rendering/steady.ludic's stream round warms up over 300 cells and takes the least of three +windows of 150, as the frame round does: one sample of it read +75 KB on a run whose twin read -5 KB. diff --git a/examples/rendering/steady.ludic b/examples/rendering/steady.ludic index cd17b5c4..defdde0c 100644 --- a/examples/rendering/steady.ludic +++ b/examples/rendering/steady.ludic @@ -97,21 +97,36 @@ program Steady { return grew } - # bytes gained while the camera crosses new ground for 300 cells: a streamed layer's cache holds - # at most 256 chunks, so it fills, evicts and compacts on the way. Each chunk once brought its - # own record and copy, made as the ground was found. + # the camera over `cells` new cells from cell `from` on, two frames a cell; the heap's gain across them + function stream_window(render3d_st: mut Render3dState, from: int, cells: int) -> long { + let before = settled(render3d_st) + for step in from .. from + cells { + cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0) + for f in 0 .. 2 { + r3d_frame(render3d_st, float(step * 2 + f) / 60.0) + r3d_present(render3d_st) + } + } + return settled(render3d_st) - before + } + + # bytes gained while the camera crosses new ground: a streamed layer's cache holds at most 256 + # chunks, so over a warm-up of 300 cells it fills, evicts and compacts; then the least of three + # windows of 150 more. A leak grows in every window; a pool finding a new high-water mark, or the + # GPU's lag on a busy machine (one sample read +75 KB where its twin read -5 KB), does not repeat + # three times. Each chunk once brought its own record and copy, made as the ground was found. function stream_rounds(render3d_st: mut Render3dState, m: Model) -> long { render3d_st.STREAM_MAX_CHUNKS = 256 let l = layer_new(render3d_st, m, 20000, false, 0.0, 0.0, 400.0) let st = stream_new(render3d_st, l, 32.0, 96.0, 32.0, 64.0, 80.0, 96.0) cam_set(render3d_st, 0.0, 3.0, 0.0, 0.0, -10.0) frame_rounds(render3d_st, 10) - let before = settled(render3d_st) - for step in 1 .. 301 { - cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0) - frame_rounds(render3d_st, 2) + stream_window(render3d_st, 1, 300) + var grew = stream_window(render3d_st, 301, 150) + for w in 1 .. 3 { + let g = stream_window(render3d_st, 301 + w * 150, 150) + if g < grew { grew = g } } - let grew = settled(render3d_st) - before # every chunk kept, after all that evicting and compacting, still holds its own cell's instances var wrong = 0 for i in 0 .. st.n { @@ -180,7 +195,7 @@ program Steady { let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf") parse_rounds(render3d_st, text, 20) let grew_p = parse_rounds(render3d_st, text, 200) - print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 300 new cells {grew_s}`) + print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 150 new cells (the least of three windows) {grew_s}`) # a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator) var ok = grew_b < 16384 if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") } @@ -211,6 +226,8 @@ program Steady { ok = false print("steady: FAILED - Vulkan's counted host allocations grow with the frame") } + # 4 KB over 150 new cells, in the least of three windows: a chunk's own record or copy kept would be + # 20 to 51 instances' worth (over 150 bytes) a cell, some 25 KB a window if grew_s >= 4096 { ok = false print("steady: FAILED - streaming new ground leaves memory behind") diff --git a/packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json b/packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json new file mode 100644 index 00000000..5886a172 --- /dev/null +++ b/packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:31ed6de11d21d4084384699b2eb9a0c861dc1ae0262ad5282f9a3e252765ad71 +size 176 diff --git a/runtime/native/vk_mac.ll b/runtime/native/vk_mac.ll index 5f768280..abf43de2 100644 --- a/runtime/native/vk_mac.ll +++ b/runtime/native/vk_mac.ll @@ -39,20 +39,106 @@ entry: ret ptr %h } -; the candidates in order: a Vulkan loader first (the SDK's, so the validation layer can be -; stacked in development), then MoltenVK itself, which exports every vk* entry point and is what -; a bundle ships in Contents/Frameworks - no loader, no ICD json - or, found through the -; executable's rpath, what ludic.render3d carries (lib/macos-arm64) for every other build. -@lvk_paths = internal constant [8 x ptr] [ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3, ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2] +; The candidates in order. MoltenVK itself first - it exports every vk* entry point, it is what a bundle +; ships in Contents/Frameworks and what ludic.render3d carries (lib/macos-arm64), and the program is +; linked against it, so dlopen hands back the copy already loaded. A Vulkan loader (the SDK's, found +; in /usr/local/lib) loads its OWN MoltenVK as an ICD: two in one process, each keeping its pools, +; and the program drew through the SDK's. The loader comes first only when layers are asked for +; (VK_INSTANCE_LAYERS, VK_LOADER_LAYERS_ENABLE, R3D_VK_LOADER), and is then pinned to ours (lvk_pin_icd). +@lvk_direct = internal constant [8 x ptr] [ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2, ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3] +@lvk_layered = internal constant [8 x ptr] [ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3, ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2] + +declare i32 @dladdr(ptr, ptr) +declare i32 @setenv(ptr, ptr, i32) +declare ptr @strrchr(ptr, i32) +declare i32 @access(ptr, i32) +@.lvk_e1 = private unnamed_addr constant [19 x i8] c"VK_INSTANCE_LAYERS\00" +@.lvk_e2 = private unnamed_addr constant [24 x i8] c"VK_LOADER_LAYERS_ENABLE\00" +@.lvk_e3 = private unnamed_addr constant [14 x i8] c"R3D_VK_LOADER\00" +@.lvk_df = private unnamed_addr constant [16 x i8] c"VK_DRIVER_FILES\00" +@.lvk_if = private unnamed_addr constant [17 x i8] c"VK_ICD_FILENAMES\00" +@.lvk_gipa = private unnamed_addr constant [22 x i8] c"vkGetInstanceProcAddr\00" +@.lvk_icd = private unnamed_addr constant [23 x i8] c"%.*s/MoltenVK_icd.json\00" + +; 1 when a layer is asked for: the loader is what stacks one +define internal i32 @lvk_want_loader() { +entry: + %a = call ptr @getenv(ptr @.lvk_e1) + %b = call ptr @getenv(ptr @.lvk_e2) + %c = call ptr @getenv(ptr @.lvk_e3) + %ha = icmp ne ptr %a, null + %hb = icmp ne ptr %b, null + %hc = icmp ne ptr %c, null + %ab = or i1 %ha, %hb + %abc = or i1 %ab, %hc + %r = zext i1 %abc to i32 + ret i32 %r +} + +; The loader told to use our MoltenVK and no other: VK_DRIVER_FILES set to the ICD json beside the +; libMoltenVK.dylib the program is linked against (found through dladdr), unless the caller named a +; driver already, or there is no json there (a bundle ships none). +define internal void @lvk_pin_icd() { +entry: + %df = call ptr @getenv(ptr @.lvk_df) + %if = call ptr @getenv(ptr @.lvk_if) + %hdf = icmp ne ptr %df, null + %hif = icmp ne ptr %if, null + %named = or i1 %hdf, %hif + br i1 %named, label %out, label %find +find: + %h = call ptr @dlopen(ptr @.lvk_mr, i32 6) + %noh = icmp eq ptr %h, null + br i1 %noh, label %out, label %sym +sym: + %f = call ptr @dlsym(ptr %h, ptr @.lvk_gipa) + %nof = icmp eq ptr %f, null + br i1 %nof, label %out, label %where +where: + %info = alloca [4 x ptr] + %got = call i32 @dladdr(ptr %f, ptr %info) + %nogot = icmp eq i32 %got, 0 + br i1 %nogot, label %out, label %dir +dir: + %fname = load ptr, ptr %info + %slash = call ptr @strrchr(ptr %fname, i32 47) + %noslash = icmp eq ptr %slash, null + br i1 %noslash, label %out, label %json +json: + %pa = ptrtoint ptr %fname to i64 + %pb = ptrtoint ptr %slash to i64 + %len64 = sub i64 %pb, %pa + %len = trunc i64 %len64 to i32 + %buf = alloca [1024 x i8] + %w = call i32 (ptr, i64, ptr, ...) @snprintf(ptr %buf, i64 1024, ptr @.lvk_icd, i32 %len, ptr %fname) + %acc = call i32 @access(ptr %buf, i32 0) + %there = icmp eq i32 %acc, 0 + br i1 %there, label %pin, label %out +pin: + %s = call i32 @setenv(ptr @.lvk_df, ptr %buf, i32 0) + br label %out +out: + ret void +} define i32 @lvk_open() { entry: %have = load ptr, ptr @lvk_lib %open = icmp ne ptr %have, null - br i1 %open, label %ok, label %loop + br i1 %open, label %ok, label %pick +pick: + %wl = call i32 @lvk_want_loader() + %layered = icmp ne i32 %wl, 0 + br i1 %layered, label %pinned, label %start +pinned: + call void @lvk_pin_icd() + br label %start +start: + %paths = select i1 %layered, ptr @lvk_layered, ptr @lvk_direct + br label %loop loop: - %i = phi i32 [ 0, %entry ], [ %i1, %next ] - %slot = getelementptr [8 x ptr], ptr @lvk_paths, i32 0, i32 %i + %i = phi i32 [ 0, %start ], [ %i1, %next ] + %slot = getelementptr [8 x ptr], ptr %paths, i32 0, i32 %i %path = load ptr, ptr %slot %h = call ptr @lvk_try(ptr %path) %miss = icmp eq ptr %h, null