Merge commit '1ab5ca14' into lang/foundations

This commit is contained in:
Orkun ÇAKILKAYA 2026-09-30 00:43:24 +03:00
commit 47ba741f6d
4 changed files with 134 additions and 17 deletions

11
changes/one-moltenvk.md Normal file
View file

@ -0,0 +1,11 @@
bump: patch
type: fix
**One MoltenVK in a process.** On a Mac with the Vulkan SDK installed, the runtime opened the SDK's loader
(/usr/local/lib/libvulkan.1.dylib) before MoltenVK, and the loader loaded the SDK's own MoltenVK as its
driver - beside ludic.render3d's, which every program is linked against: two copies, each with its pools,
and the program drew through the SDK's ("MVKBlockObserver is implemented in both"). MoltenVK is opened
first now (dlopen hands back the linked copy), and the loader only when a layer is asked for
(VK_INSTANCE_LAYERS, VK_LOADER_LAYERS_ENABLE, R3D_VK_LOADER) - then pinned to ludic.render3d's MoltenVK
through VK_DRIVER_FILES and lib/macos-arm64/MoltenVK_icd.json, unless the caller named a driver.
examples/rendering/steady.ludic's stream round warms up over 300 cells and takes the least of three
windows of 150, as the frame round does: one sample of it read +75 KB on a run whose twin read -5 KB.

View file

@ -97,21 +97,36 @@ program Steady {
return grew return grew
} }
# bytes gained while the camera crosses new ground for 300 cells: a streamed layer's cache holds # the camera over `cells` new cells from cell `from` on, two frames a cell; the heap's gain across them
# at most 256 chunks, so it fills, evicts and compacts on the way. Each chunk once brought its function stream_window(render3d_st: mut Render3dState, from: int, cells: int) -> long {
# own record and copy, made as the ground was found. let before = settled(render3d_st)
for step in from .. from + cells {
cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0)
for f in 0 .. 2 {
r3d_frame(render3d_st, float(step * 2 + f) / 60.0)
r3d_present(render3d_st)
}
}
return settled(render3d_st) - before
}
# bytes gained while the camera crosses new ground: a streamed layer's cache holds at most 256
# chunks, so over a warm-up of 300 cells it fills, evicts and compacts; then the least of three
# windows of 150 more. A leak grows in every window; a pool finding a new high-water mark, or the
# GPU's lag on a busy machine (one sample read +75 KB where its twin read -5 KB), does not repeat
# three times. Each chunk once brought its own record and copy, made as the ground was found.
function stream_rounds(render3d_st: mut Render3dState, m: Model) -> long { function stream_rounds(render3d_st: mut Render3dState, m: Model) -> long {
render3d_st.STREAM_MAX_CHUNKS = 256 render3d_st.STREAM_MAX_CHUNKS = 256
let l = layer_new(render3d_st, m, 20000, false, 0.0, 0.0, 400.0) let l = layer_new(render3d_st, m, 20000, false, 0.0, 0.0, 400.0)
let st = stream_new(render3d_st, l, 32.0, 96.0, 32.0, 64.0, 80.0, 96.0) let st = stream_new(render3d_st, l, 32.0, 96.0, 32.0, 64.0, 80.0, 96.0)
cam_set(render3d_st, 0.0, 3.0, 0.0, 0.0, -10.0) cam_set(render3d_st, 0.0, 3.0, 0.0, 0.0, -10.0)
frame_rounds(render3d_st, 10) frame_rounds(render3d_st, 10)
let before = settled(render3d_st) stream_window(render3d_st, 1, 300)
for step in 1 .. 301 { var grew = stream_window(render3d_st, 301, 150)
cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0) for w in 1 .. 3 {
frame_rounds(render3d_st, 2) let g = stream_window(render3d_st, 301 + w * 150, 150)
if g < grew { grew = g }
} }
let grew = settled(render3d_st) - before
# every chunk kept, after all that evicting and compacting, still holds its own cell's instances # every chunk kept, after all that evicting and compacting, still holds its own cell's instances
var wrong = 0 var wrong = 0
for i in 0 .. st.n { for i in 0 .. st.n {
@ -180,7 +195,7 @@ program Steady {
let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf") let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf")
parse_rounds(render3d_st, text, 20) parse_rounds(render3d_st, text, 20)
let grew_p = parse_rounds(render3d_st, text, 200) let grew_p = parse_rounds(render3d_st, text, 200)
print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 300 new cells {grew_s}`) print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 150 new cells (the least of three windows) {grew_s}`)
# a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator) # a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator)
var ok = grew_b < 16384 var ok = grew_b < 16384
if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") } if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") }
@ -211,6 +226,8 @@ program Steady {
ok = false ok = false
print("steady: FAILED - Vulkan's counted host allocations grow with the frame") print("steady: FAILED - Vulkan's counted host allocations grow with the frame")
} }
# 4 KB over 150 new cells, in the least of three windows: a chunk's own record or copy kept would be
# 20 to 51 instances' worth (over 150 bytes) a cell, some 25 KB a window
if grew_s >= 4096 { if grew_s >= 4096 {
ok = false ok = false
print("steady: FAILED - streaming new ground leaves memory behind") print("steady: FAILED - streaming new ground leaves memory behind")

BIN
packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json (Stored with Git LFS) Normal file

Binary file not shown.

View file

@ -39,20 +39,106 @@ entry:
ret ptr %h ret ptr %h
} }
; the candidates in order: a Vulkan loader first (the SDK's, so the validation layer can be ; The candidates in order. MoltenVK itself first - it exports every vk* entry point, it is what a bundle
; stacked in development), then MoltenVK itself, which exports every vk* entry point and is what ; ships in Contents/Frameworks and what ludic.render3d carries (lib/macos-arm64), and the program is
; a bundle ships in Contents/Frameworks - no loader, no ICD json - or, found through the ; linked against it, so dlopen hands back the copy already loaded. A Vulkan loader (the SDK's, found
; executable's rpath, what ludic.render3d carries (lib/macos-arm64) for every other build. ; in /usr/local/lib) loads its OWN MoltenVK as an ICD: two in one process, each keeping its pools,
@lvk_paths = internal constant [8 x ptr] [ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3, ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2] ; and the program drew through the SDK's. The loader comes first only when layers are asked for
; (VK_INSTANCE_LAYERS, VK_LOADER_LAYERS_ENABLE, R3D_VK_LOADER), and is then pinned to ours (lvk_pin_icd).
@lvk_direct = internal constant [8 x ptr] [ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2, ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3]
@lvk_layered = internal constant [8 x ptr] [ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3, ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2]
declare i32 @dladdr(ptr, ptr)
declare i32 @setenv(ptr, ptr, i32)
declare ptr @strrchr(ptr, i32)
declare i32 @access(ptr, i32)
@.lvk_e1 = private unnamed_addr constant [19 x i8] c"VK_INSTANCE_LAYERS\00"
@.lvk_e2 = private unnamed_addr constant [24 x i8] c"VK_LOADER_LAYERS_ENABLE\00"
@.lvk_e3 = private unnamed_addr constant [14 x i8] c"R3D_VK_LOADER\00"
@.lvk_df = private unnamed_addr constant [16 x i8] c"VK_DRIVER_FILES\00"
@.lvk_if = private unnamed_addr constant [17 x i8] c"VK_ICD_FILENAMES\00"
@.lvk_gipa = private unnamed_addr constant [22 x i8] c"vkGetInstanceProcAddr\00"
@.lvk_icd = private unnamed_addr constant [23 x i8] c"%.*s/MoltenVK_icd.json\00"
; 1 when a layer is asked for: the loader is what stacks one
define internal i32 @lvk_want_loader() {
entry:
%a = call ptr @getenv(ptr @.lvk_e1)
%b = call ptr @getenv(ptr @.lvk_e2)
%c = call ptr @getenv(ptr @.lvk_e3)
%ha = icmp ne ptr %a, null
%hb = icmp ne ptr %b, null
%hc = icmp ne ptr %c, null
%ab = or i1 %ha, %hb
%abc = or i1 %ab, %hc
%r = zext i1 %abc to i32
ret i32 %r
}
; The loader told to use our MoltenVK and no other: VK_DRIVER_FILES set to the ICD json beside the
; libMoltenVK.dylib the program is linked against (found through dladdr), unless the caller named a
; driver already, or there is no json there (a bundle ships none).
define internal void @lvk_pin_icd() {
entry:
%df = call ptr @getenv(ptr @.lvk_df)
%if = call ptr @getenv(ptr @.lvk_if)
%hdf = icmp ne ptr %df, null
%hif = icmp ne ptr %if, null
%named = or i1 %hdf, %hif
br i1 %named, label %out, label %find
find:
%h = call ptr @dlopen(ptr @.lvk_mr, i32 6)
%noh = icmp eq ptr %h, null
br i1 %noh, label %out, label %sym
sym:
%f = call ptr @dlsym(ptr %h, ptr @.lvk_gipa)
%nof = icmp eq ptr %f, null
br i1 %nof, label %out, label %where
where:
%info = alloca [4 x ptr]
%got = call i32 @dladdr(ptr %f, ptr %info)
%nogot = icmp eq i32 %got, 0
br i1 %nogot, label %out, label %dir
dir:
%fname = load ptr, ptr %info
%slash = call ptr @strrchr(ptr %fname, i32 47)
%noslash = icmp eq ptr %slash, null
br i1 %noslash, label %out, label %json
json:
%pa = ptrtoint ptr %fname to i64
%pb = ptrtoint ptr %slash to i64
%len64 = sub i64 %pb, %pa
%len = trunc i64 %len64 to i32
%buf = alloca [1024 x i8]
%w = call i32 (ptr, i64, ptr, ...) @snprintf(ptr %buf, i64 1024, ptr @.lvk_icd, i32 %len, ptr %fname)
%acc = call i32 @access(ptr %buf, i32 0)
%there = icmp eq i32 %acc, 0
br i1 %there, label %pin, label %out
pin:
%s = call i32 @setenv(ptr @.lvk_df, ptr %buf, i32 0)
br label %out
out:
ret void
}
define i32 @lvk_open() { define i32 @lvk_open() {
entry: entry:
%have = load ptr, ptr @lvk_lib %have = load ptr, ptr @lvk_lib
%open = icmp ne ptr %have, null %open = icmp ne ptr %have, null
br i1 %open, label %ok, label %loop br i1 %open, label %ok, label %pick
pick:
%wl = call i32 @lvk_want_loader()
%layered = icmp ne i32 %wl, 0
br i1 %layered, label %pinned, label %start
pinned:
call void @lvk_pin_icd()
br label %start
start:
%paths = select i1 %layered, ptr @lvk_layered, ptr @lvk_direct
br label %loop
loop: loop:
%i = phi i32 [ 0, %entry ], [ %i1, %next ] %i = phi i32 [ 0, %start ], [ %i1, %next ]
%slot = getelementptr [8 x ptr], ptr @lvk_paths, i32 0, i32 %i %slot = getelementptr [8 x ptr], ptr %paths, i32 0, i32 %i
%path = load ptr, ptr %slot %path = load ptr, ptr %slot
%h = call ptr @lvk_try(ptr %path) %h = call ptr @lvk_try(ptr %path)
%miss = icmp eq ptr %h, null %miss = icmp eq ptr %h, null