runtime (macOS): MoltenVK opened first, so a process holds one - the SDK's loader in /usr/local/lib loaded its own MoltenVK beside the linked one and the program drew through it; the loader only when a layer is asked for, pinned to ours (VK_DRIVER_FILES, lib/macos-arm64/MoltenVK_icd.json) unless a driver is named. steady's stream round: a 300-cell warm-up, then the least of three 150-cell windows
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
2e0a7f0031
commit
1ab5ca14fb
4 changed files with 134 additions and 17 deletions
11
changes/one-moltenvk.md
Normal file
11
changes/one-moltenvk.md
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
bump: patch
|
||||
type: fix
|
||||
**One MoltenVK in a process.** On a Mac with the Vulkan SDK installed, the runtime opened the SDK's loader
|
||||
(/usr/local/lib/libvulkan.1.dylib) before MoltenVK, and the loader loaded the SDK's own MoltenVK as its
|
||||
driver - beside ludic.render3d's, which every program is linked against: two copies, each with its pools,
|
||||
and the program drew through the SDK's ("MVKBlockObserver is implemented in both"). MoltenVK is opened
|
||||
first now (dlopen hands back the linked copy), and the loader only when a layer is asked for
|
||||
(VK_INSTANCE_LAYERS, VK_LOADER_LAYERS_ENABLE, R3D_VK_LOADER) - then pinned to ludic.render3d's MoltenVK
|
||||
through VK_DRIVER_FILES and lib/macos-arm64/MoltenVK_icd.json, unless the caller named a driver.
|
||||
examples/rendering/steady.ludic's stream round warms up over 300 cells and takes the least of three
|
||||
windows of 150, as the frame round does: one sample of it read +75 KB on a run whose twin read -5 KB.
|
||||
|
|
@ -97,21 +97,36 @@ program Steady {
|
|||
return grew
|
||||
}
|
||||
|
||||
# bytes gained while the camera crosses new ground for 300 cells: a streamed layer's cache holds
|
||||
# at most 256 chunks, so it fills, evicts and compacts on the way. Each chunk once brought its
|
||||
# own record and copy, made as the ground was found.
|
||||
# the camera over `cells` new cells from cell `from` on, two frames a cell; the heap's gain across them
|
||||
function stream_window(render3d_st: mut Render3dState, from: int, cells: int) -> long {
|
||||
let before = settled(render3d_st)
|
||||
for step in from .. from + cells {
|
||||
cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0)
|
||||
for f in 0 .. 2 {
|
||||
r3d_frame(render3d_st, float(step * 2 + f) / 60.0)
|
||||
r3d_present(render3d_st)
|
||||
}
|
||||
}
|
||||
return settled(render3d_st) - before
|
||||
}
|
||||
|
||||
# bytes gained while the camera crosses new ground: a streamed layer's cache holds at most 256
|
||||
# chunks, so over a warm-up of 300 cells it fills, evicts and compacts; then the least of three
|
||||
# windows of 150 more. A leak grows in every window; a pool finding a new high-water mark, or the
|
||||
# GPU's lag on a busy machine (one sample read +75 KB where its twin read -5 KB), does not repeat
|
||||
# three times. Each chunk once brought its own record and copy, made as the ground was found.
|
||||
function stream_rounds(render3d_st: mut Render3dState, m: Model) -> long {
|
||||
render3d_st.STREAM_MAX_CHUNKS = 256
|
||||
let l = layer_new(render3d_st, m, 20000, false, 0.0, 0.0, 400.0)
|
||||
let st = stream_new(render3d_st, l, 32.0, 96.0, 32.0, 64.0, 80.0, 96.0)
|
||||
cam_set(render3d_st, 0.0, 3.0, 0.0, 0.0, -10.0)
|
||||
frame_rounds(render3d_st, 10)
|
||||
let before = settled(render3d_st)
|
||||
for step in 1 .. 301 {
|
||||
cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0)
|
||||
frame_rounds(render3d_st, 2)
|
||||
stream_window(render3d_st, 1, 300)
|
||||
var grew = stream_window(render3d_st, 301, 150)
|
||||
for w in 1 .. 3 {
|
||||
let g = stream_window(render3d_st, 301 + w * 150, 150)
|
||||
if g < grew { grew = g }
|
||||
}
|
||||
let grew = settled(render3d_st) - before
|
||||
# every chunk kept, after all that evicting and compacting, still holds its own cell's instances
|
||||
var wrong = 0
|
||||
for i in 0 .. st.n {
|
||||
|
|
@ -180,7 +195,7 @@ program Steady {
|
|||
let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf")
|
||||
parse_rounds(render3d_st, text, 20)
|
||||
let grew_p = parse_rounds(render3d_st, text, 200)
|
||||
print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 300 new cells {grew_s}`)
|
||||
print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 150 new cells (the least of three windows) {grew_s}`)
|
||||
# a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator)
|
||||
var ok = grew_b < 16384
|
||||
if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") }
|
||||
|
|
@ -211,6 +226,8 @@ program Steady {
|
|||
ok = false
|
||||
print("steady: FAILED - Vulkan's counted host allocations grow with the frame")
|
||||
}
|
||||
# 4 KB over 150 new cells, in the least of three windows: a chunk's own record or copy kept would be
|
||||
# 20 to 51 instances' worth (over 150 bytes) a cell, some 25 KB a window
|
||||
if grew_s >= 4096 {
|
||||
ok = false
|
||||
print("steady: FAILED - streaming new ground leaves memory behind")
|
||||
|
|
|
|||
BIN
packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json
(Stored with Git LFS)
Normal file
BIN
packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json
(Stored with Git LFS)
Normal file
Binary file not shown.
|
|
@ -39,20 +39,106 @@ entry:
|
|||
ret ptr %h
|
||||
}
|
||||
|
||||
; the candidates in order: a Vulkan loader first (the SDK's, so the validation layer can be
|
||||
; stacked in development), then MoltenVK itself, which exports every vk* entry point and is what
|
||||
; a bundle ships in Contents/Frameworks - no loader, no ICD json - or, found through the
|
||||
; executable's rpath, what ludic.render3d carries (lib/macos-arm64) for every other build.
|
||||
@lvk_paths = internal constant [8 x ptr] [ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3, ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2]
|
||||
; The candidates in order. MoltenVK itself first - it exports every vk* entry point, it is what a bundle
|
||||
; ships in Contents/Frameworks and what ludic.render3d carries (lib/macos-arm64), and the program is
|
||||
; linked against it, so dlopen hands back the copy already loaded. A Vulkan loader (the SDK's, found
|
||||
; in /usr/local/lib) loads its OWN MoltenVK as an ICD: two in one process, each keeping its pools,
|
||||
; and the program drew through the SDK's. The loader comes first only when layers are asked for
|
||||
; (VK_INSTANCE_LAYERS, VK_LOADER_LAYERS_ENABLE, R3D_VK_LOADER), and is then pinned to ours (lvk_pin_icd).
|
||||
@lvk_direct = internal constant [8 x ptr] [ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2, ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3]
|
||||
@lvk_layered = internal constant [8 x ptr] [ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3, ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2]
|
||||
|
||||
declare i32 @dladdr(ptr, ptr)
|
||||
declare i32 @setenv(ptr, ptr, i32)
|
||||
declare ptr @strrchr(ptr, i32)
|
||||
declare i32 @access(ptr, i32)
|
||||
@.lvk_e1 = private unnamed_addr constant [19 x i8] c"VK_INSTANCE_LAYERS\00"
|
||||
@.lvk_e2 = private unnamed_addr constant [24 x i8] c"VK_LOADER_LAYERS_ENABLE\00"
|
||||
@.lvk_e3 = private unnamed_addr constant [14 x i8] c"R3D_VK_LOADER\00"
|
||||
@.lvk_df = private unnamed_addr constant [16 x i8] c"VK_DRIVER_FILES\00"
|
||||
@.lvk_if = private unnamed_addr constant [17 x i8] c"VK_ICD_FILENAMES\00"
|
||||
@.lvk_gipa = private unnamed_addr constant [22 x i8] c"vkGetInstanceProcAddr\00"
|
||||
@.lvk_icd = private unnamed_addr constant [23 x i8] c"%.*s/MoltenVK_icd.json\00"
|
||||
|
||||
; 1 when a layer is asked for: the loader is what stacks one
|
||||
define internal i32 @lvk_want_loader() {
|
||||
entry:
|
||||
%a = call ptr @getenv(ptr @.lvk_e1)
|
||||
%b = call ptr @getenv(ptr @.lvk_e2)
|
||||
%c = call ptr @getenv(ptr @.lvk_e3)
|
||||
%ha = icmp ne ptr %a, null
|
||||
%hb = icmp ne ptr %b, null
|
||||
%hc = icmp ne ptr %c, null
|
||||
%ab = or i1 %ha, %hb
|
||||
%abc = or i1 %ab, %hc
|
||||
%r = zext i1 %abc to i32
|
||||
ret i32 %r
|
||||
}
|
||||
|
||||
; The loader told to use our MoltenVK and no other: VK_DRIVER_FILES set to the ICD json beside the
|
||||
; libMoltenVK.dylib the program is linked against (found through dladdr), unless the caller named a
|
||||
; driver already, or there is no json there (a bundle ships none).
|
||||
define internal void @lvk_pin_icd() {
|
||||
entry:
|
||||
%df = call ptr @getenv(ptr @.lvk_df)
|
||||
%if = call ptr @getenv(ptr @.lvk_if)
|
||||
%hdf = icmp ne ptr %df, null
|
||||
%hif = icmp ne ptr %if, null
|
||||
%named = or i1 %hdf, %hif
|
||||
br i1 %named, label %out, label %find
|
||||
find:
|
||||
%h = call ptr @dlopen(ptr @.lvk_mr, i32 6)
|
||||
%noh = icmp eq ptr %h, null
|
||||
br i1 %noh, label %out, label %sym
|
||||
sym:
|
||||
%f = call ptr @dlsym(ptr %h, ptr @.lvk_gipa)
|
||||
%nof = icmp eq ptr %f, null
|
||||
br i1 %nof, label %out, label %where
|
||||
where:
|
||||
%info = alloca [4 x ptr]
|
||||
%got = call i32 @dladdr(ptr %f, ptr %info)
|
||||
%nogot = icmp eq i32 %got, 0
|
||||
br i1 %nogot, label %out, label %dir
|
||||
dir:
|
||||
%fname = load ptr, ptr %info
|
||||
%slash = call ptr @strrchr(ptr %fname, i32 47)
|
||||
%noslash = icmp eq ptr %slash, null
|
||||
br i1 %noslash, label %out, label %json
|
||||
json:
|
||||
%pa = ptrtoint ptr %fname to i64
|
||||
%pb = ptrtoint ptr %slash to i64
|
||||
%len64 = sub i64 %pb, %pa
|
||||
%len = trunc i64 %len64 to i32
|
||||
%buf = alloca [1024 x i8]
|
||||
%w = call i32 (ptr, i64, ptr, ...) @snprintf(ptr %buf, i64 1024, ptr @.lvk_icd, i32 %len, ptr %fname)
|
||||
%acc = call i32 @access(ptr %buf, i32 0)
|
||||
%there = icmp eq i32 %acc, 0
|
||||
br i1 %there, label %pin, label %out
|
||||
pin:
|
||||
%s = call i32 @setenv(ptr @.lvk_df, ptr %buf, i32 0)
|
||||
br label %out
|
||||
out:
|
||||
ret void
|
||||
}
|
||||
|
||||
define i32 @lvk_open() {
|
||||
entry:
|
||||
%have = load ptr, ptr @lvk_lib
|
||||
%open = icmp ne ptr %have, null
|
||||
br i1 %open, label %ok, label %loop
|
||||
br i1 %open, label %ok, label %pick
|
||||
pick:
|
||||
%wl = call i32 @lvk_want_loader()
|
||||
%layered = icmp ne i32 %wl, 0
|
||||
br i1 %layered, label %pinned, label %start
|
||||
pinned:
|
||||
call void @lvk_pin_icd()
|
||||
br label %start
|
||||
start:
|
||||
%paths = select i1 %layered, ptr @lvk_layered, ptr @lvk_direct
|
||||
br label %loop
|
||||
loop:
|
||||
%i = phi i32 [ 0, %entry ], [ %i1, %next ]
|
||||
%slot = getelementptr [8 x ptr], ptr @lvk_paths, i32 0, i32 %i
|
||||
%i = phi i32 [ 0, %start ], [ %i1, %next ]
|
||||
%slot = getelementptr [8 x ptr], ptr %paths, i32 0, i32 %i
|
||||
%path = load ptr, ptr %slot
|
||||
%h = call ptr @lvk_try(ptr %path)
|
||||
%miss = icmp eq ptr %h, null
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue