Merge commit '1ab5ca14' into lang/foundations
This commit is contained in:
commit
47ba741f6d
4 changed files with 134 additions and 17 deletions
11
changes/one-moltenvk.md
Normal file
11
changes/one-moltenvk.md
Normal file
|
|
@ -0,0 +1,11 @@
|
||||||
|
bump: patch
|
||||||
|
type: fix
|
||||||
|
**One MoltenVK in a process.** On a Mac with the Vulkan SDK installed, the runtime opened the SDK's loader
|
||||||
|
(/usr/local/lib/libvulkan.1.dylib) before MoltenVK, and the loader loaded the SDK's own MoltenVK as its
|
||||||
|
driver - beside ludic.render3d's, which every program is linked against: two copies, each with its pools,
|
||||||
|
and the program drew through the SDK's ("MVKBlockObserver is implemented in both"). MoltenVK is opened
|
||||||
|
first now (dlopen hands back the linked copy), and the loader only when a layer is asked for
|
||||||
|
(VK_INSTANCE_LAYERS, VK_LOADER_LAYERS_ENABLE, R3D_VK_LOADER) - then pinned to ludic.render3d's MoltenVK
|
||||||
|
through VK_DRIVER_FILES and lib/macos-arm64/MoltenVK_icd.json, unless the caller named a driver.
|
||||||
|
examples/rendering/steady.ludic's stream round warms up over 300 cells and takes the least of three
|
||||||
|
windows of 150, as the frame round does: one sample of it read +75 KB on a run whose twin read -5 KB.
|
||||||
|
|
@ -97,21 +97,36 @@ program Steady {
|
||||||
return grew
|
return grew
|
||||||
}
|
}
|
||||||
|
|
||||||
# bytes gained while the camera crosses new ground for 300 cells: a streamed layer's cache holds
|
# the camera over `cells` new cells from cell `from` on, two frames a cell; the heap's gain across them
|
||||||
# at most 256 chunks, so it fills, evicts and compacts on the way. Each chunk once brought its
|
function stream_window(render3d_st: mut Render3dState, from: int, cells: int) -> long {
|
||||||
# own record and copy, made as the ground was found.
|
let before = settled(render3d_st)
|
||||||
|
for step in from .. from + cells {
|
||||||
|
cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0)
|
||||||
|
for f in 0 .. 2 {
|
||||||
|
r3d_frame(render3d_st, float(step * 2 + f) / 60.0)
|
||||||
|
r3d_present(render3d_st)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return settled(render3d_st) - before
|
||||||
|
}
|
||||||
|
|
||||||
|
# bytes gained while the camera crosses new ground: a streamed layer's cache holds at most 256
|
||||||
|
# chunks, so over a warm-up of 300 cells it fills, evicts and compacts; then the least of three
|
||||||
|
# windows of 150 more. A leak grows in every window; a pool finding a new high-water mark, or the
|
||||||
|
# GPU's lag on a busy machine (one sample read +75 KB where its twin read -5 KB), does not repeat
|
||||||
|
# three times. Each chunk once brought its own record and copy, made as the ground was found.
|
||||||
function stream_rounds(render3d_st: mut Render3dState, m: Model) -> long {
|
function stream_rounds(render3d_st: mut Render3dState, m: Model) -> long {
|
||||||
render3d_st.STREAM_MAX_CHUNKS = 256
|
render3d_st.STREAM_MAX_CHUNKS = 256
|
||||||
let l = layer_new(render3d_st, m, 20000, false, 0.0, 0.0, 400.0)
|
let l = layer_new(render3d_st, m, 20000, false, 0.0, 0.0, 400.0)
|
||||||
let st = stream_new(render3d_st, l, 32.0, 96.0, 32.0, 64.0, 80.0, 96.0)
|
let st = stream_new(render3d_st, l, 32.0, 96.0, 32.0, 64.0, 80.0, 96.0)
|
||||||
cam_set(render3d_st, 0.0, 3.0, 0.0, 0.0, -10.0)
|
cam_set(render3d_st, 0.0, 3.0, 0.0, 0.0, -10.0)
|
||||||
frame_rounds(render3d_st, 10)
|
frame_rounds(render3d_st, 10)
|
||||||
let before = settled(render3d_st)
|
stream_window(render3d_st, 1, 300)
|
||||||
for step in 1 .. 301 {
|
var grew = stream_window(render3d_st, 301, 150)
|
||||||
cam_set(render3d_st, float(step) * 32.0, 3.0, float(step % 7) * 32.0, 0.0, -10.0)
|
for w in 1 .. 3 {
|
||||||
frame_rounds(render3d_st, 2)
|
let g = stream_window(render3d_st, 301 + w * 150, 150)
|
||||||
|
if g < grew { grew = g }
|
||||||
}
|
}
|
||||||
let grew = settled(render3d_st) - before
|
|
||||||
# every chunk kept, after all that evicting and compacting, still holds its own cell's instances
|
# every chunk kept, after all that evicting and compacting, still holds its own cell's instances
|
||||||
var wrong = 0
|
var wrong = 0
|
||||||
for i in 0 .. st.n {
|
for i in 0 .. st.n {
|
||||||
|
|
@ -180,7 +195,7 @@ program Steady {
|
||||||
let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf")
|
let text = Fs.read_text("packages/ludic.lab/plate/plate.gltf")
|
||||||
parse_rounds(render3d_st, text, 20)
|
parse_rounds(render3d_st, text, 20)
|
||||||
let grew_p = parse_rounds(render3d_st, text, 200)
|
let grew_p = parse_rounds(render3d_st, text, 200)
|
||||||
print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 300 new cells {grew_s}`)
|
print(`steady: the buffer path gained {grew_b} bytes over 5000 rounds, the frame {grew_f} over 600, a glTF parsed and freed {grew_p} over 200, a model loaded and let go {grew_m} over 200, an actor placed and released {grew_a} over 2000, a frame drawing 20 to 200 actors {grew_r}, a stream over 150 new cells (the least of three windows) {grew_s}`)
|
||||||
# a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator)
|
# a few KB of slack for what the system's own libraries keep (Metal's caches, the allocator)
|
||||||
var ok = grew_b < 16384
|
var ok = grew_b < 16384
|
||||||
if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") }
|
if not ok { print("steady: FAILED - releasing and making a buffer again leaves memory behind") }
|
||||||
|
|
@ -211,6 +226,8 @@ program Steady {
|
||||||
ok = false
|
ok = false
|
||||||
print("steady: FAILED - Vulkan's counted host allocations grow with the frame")
|
print("steady: FAILED - Vulkan's counted host allocations grow with the frame")
|
||||||
}
|
}
|
||||||
|
# 4 KB over 150 new cells, in the least of three windows: a chunk's own record or copy kept would be
|
||||||
|
# 20 to 51 instances' worth (over 150 bytes) a cell, some 25 KB a window
|
||||||
if grew_s >= 4096 {
|
if grew_s >= 4096 {
|
||||||
ok = false
|
ok = false
|
||||||
print("steady: FAILED - streaming new ground leaves memory behind")
|
print("steady: FAILED - streaming new ground leaves memory behind")
|
||||||
|
|
|
||||||
BIN
packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json
(Stored with Git LFS)
Normal file
BIN
packages/ludic.render3d/lib/macos-arm64/MoltenVK_icd.json
(Stored with Git LFS)
Normal file
Binary file not shown.
|
|
@ -39,20 +39,106 @@ entry:
|
||||||
ret ptr %h
|
ret ptr %h
|
||||||
}
|
}
|
||||||
|
|
||||||
; the candidates in order: a Vulkan loader first (the SDK's, so the validation layer can be
|
; The candidates in order. MoltenVK itself first - it exports every vk* entry point, it is what a bundle
|
||||||
; stacked in development), then MoltenVK itself, which exports every vk* entry point and is what
|
; ships in Contents/Frameworks and what ludic.render3d carries (lib/macos-arm64), and the program is
|
||||||
; a bundle ships in Contents/Frameworks - no loader, no ICD json - or, found through the
|
; linked against it, so dlopen hands back the copy already loaded. A Vulkan loader (the SDK's, found
|
||||||
; executable's rpath, what ludic.render3d carries (lib/macos-arm64) for every other build.
|
; in /usr/local/lib) loads its OWN MoltenVK as an ICD: two in one process, each keeping its pools,
|
||||||
@lvk_paths = internal constant [8 x ptr] [ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3, ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2]
|
; and the program drew through the SDK's. The loader comes first only when layers are asked for
|
||||||
|
; (VK_INSTANCE_LAYERS, VK_LOADER_LAYERS_ENABLE, R3D_VK_LOADER), and is then pinned to ours (lvk_pin_icd).
|
||||||
|
@lvk_direct = internal constant [8 x ptr] [ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2, ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3]
|
||||||
|
@lvk_layered = internal constant [8 x ptr] [ptr @.lvk_c0, ptr @.lvk_c1, ptr @.lvk_c2, ptr @.lvk_c3, ptr @.lvk_m0, ptr @.lvk_mr, ptr @.lvk_m1, ptr @.lvk_m2]
|
||||||
|
|
||||||
|
declare i32 @dladdr(ptr, ptr)
|
||||||
|
declare i32 @setenv(ptr, ptr, i32)
|
||||||
|
declare ptr @strrchr(ptr, i32)
|
||||||
|
declare i32 @access(ptr, i32)
|
||||||
|
@.lvk_e1 = private unnamed_addr constant [19 x i8] c"VK_INSTANCE_LAYERS\00"
|
||||||
|
@.lvk_e2 = private unnamed_addr constant [24 x i8] c"VK_LOADER_LAYERS_ENABLE\00"
|
||||||
|
@.lvk_e3 = private unnamed_addr constant [14 x i8] c"R3D_VK_LOADER\00"
|
||||||
|
@.lvk_df = private unnamed_addr constant [16 x i8] c"VK_DRIVER_FILES\00"
|
||||||
|
@.lvk_if = private unnamed_addr constant [17 x i8] c"VK_ICD_FILENAMES\00"
|
||||||
|
@.lvk_gipa = private unnamed_addr constant [22 x i8] c"vkGetInstanceProcAddr\00"
|
||||||
|
@.lvk_icd = private unnamed_addr constant [23 x i8] c"%.*s/MoltenVK_icd.json\00"
|
||||||
|
|
||||||
|
; 1 when a layer is asked for: the loader is what stacks one
|
||||||
|
define internal i32 @lvk_want_loader() {
|
||||||
|
entry:
|
||||||
|
%a = call ptr @getenv(ptr @.lvk_e1)
|
||||||
|
%b = call ptr @getenv(ptr @.lvk_e2)
|
||||||
|
%c = call ptr @getenv(ptr @.lvk_e3)
|
||||||
|
%ha = icmp ne ptr %a, null
|
||||||
|
%hb = icmp ne ptr %b, null
|
||||||
|
%hc = icmp ne ptr %c, null
|
||||||
|
%ab = or i1 %ha, %hb
|
||||||
|
%abc = or i1 %ab, %hc
|
||||||
|
%r = zext i1 %abc to i32
|
||||||
|
ret i32 %r
|
||||||
|
}
|
||||||
|
|
||||||
|
; The loader told to use our MoltenVK and no other: VK_DRIVER_FILES set to the ICD json beside the
|
||||||
|
; libMoltenVK.dylib the program is linked against (found through dladdr), unless the caller named a
|
||||||
|
; driver already, or there is no json there (a bundle ships none).
|
||||||
|
define internal void @lvk_pin_icd() {
|
||||||
|
entry:
|
||||||
|
%df = call ptr @getenv(ptr @.lvk_df)
|
||||||
|
%if = call ptr @getenv(ptr @.lvk_if)
|
||||||
|
%hdf = icmp ne ptr %df, null
|
||||||
|
%hif = icmp ne ptr %if, null
|
||||||
|
%named = or i1 %hdf, %hif
|
||||||
|
br i1 %named, label %out, label %find
|
||||||
|
find:
|
||||||
|
%h = call ptr @dlopen(ptr @.lvk_mr, i32 6)
|
||||||
|
%noh = icmp eq ptr %h, null
|
||||||
|
br i1 %noh, label %out, label %sym
|
||||||
|
sym:
|
||||||
|
%f = call ptr @dlsym(ptr %h, ptr @.lvk_gipa)
|
||||||
|
%nof = icmp eq ptr %f, null
|
||||||
|
br i1 %nof, label %out, label %where
|
||||||
|
where:
|
||||||
|
%info = alloca [4 x ptr]
|
||||||
|
%got = call i32 @dladdr(ptr %f, ptr %info)
|
||||||
|
%nogot = icmp eq i32 %got, 0
|
||||||
|
br i1 %nogot, label %out, label %dir
|
||||||
|
dir:
|
||||||
|
%fname = load ptr, ptr %info
|
||||||
|
%slash = call ptr @strrchr(ptr %fname, i32 47)
|
||||||
|
%noslash = icmp eq ptr %slash, null
|
||||||
|
br i1 %noslash, label %out, label %json
|
||||||
|
json:
|
||||||
|
%pa = ptrtoint ptr %fname to i64
|
||||||
|
%pb = ptrtoint ptr %slash to i64
|
||||||
|
%len64 = sub i64 %pb, %pa
|
||||||
|
%len = trunc i64 %len64 to i32
|
||||||
|
%buf = alloca [1024 x i8]
|
||||||
|
%w = call i32 (ptr, i64, ptr, ...) @snprintf(ptr %buf, i64 1024, ptr @.lvk_icd, i32 %len, ptr %fname)
|
||||||
|
%acc = call i32 @access(ptr %buf, i32 0)
|
||||||
|
%there = icmp eq i32 %acc, 0
|
||||||
|
br i1 %there, label %pin, label %out
|
||||||
|
pin:
|
||||||
|
%s = call i32 @setenv(ptr @.lvk_df, ptr %buf, i32 0)
|
||||||
|
br label %out
|
||||||
|
out:
|
||||||
|
ret void
|
||||||
|
}
|
||||||
|
|
||||||
define i32 @lvk_open() {
|
define i32 @lvk_open() {
|
||||||
entry:
|
entry:
|
||||||
%have = load ptr, ptr @lvk_lib
|
%have = load ptr, ptr @lvk_lib
|
||||||
%open = icmp ne ptr %have, null
|
%open = icmp ne ptr %have, null
|
||||||
br i1 %open, label %ok, label %loop
|
br i1 %open, label %ok, label %pick
|
||||||
|
pick:
|
||||||
|
%wl = call i32 @lvk_want_loader()
|
||||||
|
%layered = icmp ne i32 %wl, 0
|
||||||
|
br i1 %layered, label %pinned, label %start
|
||||||
|
pinned:
|
||||||
|
call void @lvk_pin_icd()
|
||||||
|
br label %start
|
||||||
|
start:
|
||||||
|
%paths = select i1 %layered, ptr @lvk_layered, ptr @lvk_direct
|
||||||
|
br label %loop
|
||||||
loop:
|
loop:
|
||||||
%i = phi i32 [ 0, %entry ], [ %i1, %next ]
|
%i = phi i32 [ 0, %start ], [ %i1, %next ]
|
||||||
%slot = getelementptr [8 x ptr], ptr @lvk_paths, i32 0, i32 %i
|
%slot = getelementptr [8 x ptr], ptr %paths, i32 0, i32 %i
|
||||||
%path = load ptr, ptr %slot
|
%path = load ptr, ptr %slot
|
||||||
%h = call ptr @lvk_try(ptr %path)
|
%h = call ptr @lvk_try(ptr %path)
|
||||||
%miss = icmp eq ptr %h, null
|
%miss = icmp eq ptr %h, null
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue