- a CAMetalLayer on the view (cocoa.ll win_metal_layer), VK_EXT_metal_surface, QuartzCore linked with Vk.*; the drawable measured after the layer sets the backing scale - vk_mac.ll opens MoltenVK directly after any loader: a bundle ships only libMoltenVK.dylib - a covered window is not presented to (win_visible); one frame in flight on macOS (R3D_VK_INFLIGHT), with images, buffers and descriptor pools held until it is done - win_held / win_mouse / win_pad / win_touch / win_text / win_present pass slice elements (arg_buf): every windowed program died on its first input poll - variants.list and SPIR-V for the three NEAR_FADE foliage programs Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
424 lines
12 KiB
LLVM
424 lines
12 KiB
LLVM
; ============================================================================
|
|
; gl.ll — the window-independent half of the OpenGL backend, in LLVM IR.
|
|
;
|
|
; Linked into any program that uses Gl.* (windowed or headless), together with
|
|
; gl_thunks.ll (the generated per-entry-point ABI thunks) and -framework OpenGL.
|
|
; Nothing here touches the window: the NSOpenGLContext lives in cocoa.ll.
|
|
;
|
|
; cgl_offscreen() -> ok a headless 4.1 core context (render into FBOs)
|
|
; fx_to_f32(fx) -> bits Q16.16 -> IEEE float bits (an int)
|
|
; f32_to_fx(bits) -> fx IEEE float bits -> Q16.16
|
|
; mem_* raw little-endian reads/writes on a bytes buffer
|
|
; f_* IEEE-754 float arithmetic on float bits
|
|
; ============================================================================
|
|
|
|
declare i32 @CGLChoosePixelFormat(ptr, ptr, ptr)
|
|
declare i32 @CGLCreateContext(ptr, ptr, ptr)
|
|
declare i32 @CGLSetCurrentContext(ptr)
|
|
declare i32 @CGLDestroyPixelFormat(ptr)
|
|
declare void @llvm.memcpy.p0.p0.i64(ptr, ptr, i64, i1)
|
|
declare void @llvm.memset.p0.i64(ptr, i8, i64, i1)
|
|
declare float @sinf(float)
|
|
declare float @cosf(float)
|
|
declare float @tanf(float)
|
|
declare float @atan2f(float, float)
|
|
declare float @powf(float, float)
|
|
declare float @expf(float)
|
|
declare float @logf(float)
|
|
declare float @floorf(float)
|
|
declare float @fmodf(float, float)
|
|
declare float @ldexpf(float, i32)
|
|
declare float @llvm.sqrt.f32(float)
|
|
declare float @llvm.fabs.f32(float)
|
|
|
|
define i32 @cgl_offscreen() {
|
|
entry:
|
|
; kCGLPFAAccelerated=73, kCGLPFAOpenGLProfile=99 -> kCGLOGLPVersion_GL4_Core (0x4100),
|
|
; kCGLPFAColorSize=8 -> 24, kCGLPFADepthSize=12 -> 24, 0
|
|
%attrs = alloca [8 x i32], align 4
|
|
%a0 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 0
|
|
store i32 73, ptr %a0
|
|
%a1 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 1
|
|
store i32 99, ptr %a1
|
|
%a2 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 2
|
|
store i32 16640, ptr %a2
|
|
%a3 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 3
|
|
store i32 8, ptr %a3
|
|
%a4 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 4
|
|
store i32 24, ptr %a4
|
|
%a5 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 5
|
|
store i32 12, ptr %a5
|
|
%a6 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 6
|
|
store i32 24, ptr %a6
|
|
%a7 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 7
|
|
store i32 0, ptr %a7
|
|
%pix = alloca ptr, align 8
|
|
store ptr null, ptr %pix
|
|
%npix = alloca i32, align 4
|
|
%e1 = call i32 @CGLChoosePixelFormat(ptr %attrs, ptr %pix, ptr %npix)
|
|
%p = load ptr, ptr %pix
|
|
%nop = icmp eq ptr %p, null
|
|
br i1 %nop, label %fail, label %mk
|
|
mk:
|
|
%ctx = alloca ptr, align 8
|
|
store ptr null, ptr %ctx
|
|
%e2 = call i32 @CGLCreateContext(ptr %p, ptr null, ptr %ctx)
|
|
%c = load ptr, ptr %ctx
|
|
%e3 = call i32 @CGLDestroyPixelFormat(ptr %p)
|
|
%noc = icmp eq ptr %c, null
|
|
br i1 %noc, label %fail, label %cur
|
|
cur:
|
|
%e4 = call i32 @CGLSetCurrentContext(ptr %c)
|
|
ret i32 1
|
|
fail:
|
|
ret i32 0
|
|
}
|
|
|
|
; ---- Q16.16 <-> IEEE float ---------------------------------------------------
|
|
define i32 @fx_to_f32(i32 %fx) {
|
|
entry:
|
|
%f = sitofp i32 %fx to float
|
|
%s = fmul float %f, 0x3EF0000000000000
|
|
%b = bitcast float %s to i32
|
|
ret i32 %b
|
|
}
|
|
define i32 @f32_to_fx(i32 %bits) {
|
|
entry:
|
|
%f = bitcast i32 %bits to float
|
|
%s = fmul float %f, 65536.0
|
|
%r = fptosi float %s to i32
|
|
ret i32 %r
|
|
}
|
|
|
|
; ---- raw memory ---------------------------------------------------------------
|
|
define ptr @mem_off(ptr %p, i32 %off) {
|
|
entry:
|
|
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
|
ret ptr %q
|
|
}
|
|
define i32 @mem_get_i32(ptr %p, i32 %off) {
|
|
entry:
|
|
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
|
%v = load i32, ptr %q, align 1
|
|
ret i32 %v
|
|
}
|
|
define void @mem_put_i32(ptr %p, i32 %off, i32 %v) {
|
|
entry:
|
|
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
|
store i32 %v, ptr %q, align 1
|
|
ret void
|
|
}
|
|
define i32 @mem_get_u16(ptr %p, i32 %off) {
|
|
entry:
|
|
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
|
%v = load i16, ptr %q, align 1
|
|
%z = zext i16 %v to i32
|
|
ret i32 %z
|
|
}
|
|
define void @mem_put_u16(ptr %p, i32 %off, i32 %v) {
|
|
entry:
|
|
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
|
%t = trunc i32 %v to i16
|
|
store i16 %t, ptr %q, align 1
|
|
ret void
|
|
}
|
|
define i32 @mem_get_u8(ptr %p, i32 %off) {
|
|
entry:
|
|
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
|
%v = load i8, ptr %q
|
|
%z = zext i8 %v to i32
|
|
ret i32 %z
|
|
}
|
|
define void @mem_put_u8(ptr %p, i32 %off, i32 %v) {
|
|
entry:
|
|
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
|
%t = trunc i32 %v to i8
|
|
store i8 %t, ptr %q
|
|
ret void
|
|
}
|
|
; float element i of a float buffer, as Q16.16 / as bits
|
|
define i32 @mem_get_f32(ptr %p, i32 %i) {
|
|
entry:
|
|
%q = getelementptr inbounds float, ptr %p, i32 %i
|
|
%f = load float, ptr %q, align 1
|
|
%s = fmul float %f, 65536.0
|
|
%r = fptosi float %s to i32
|
|
ret i32 %r
|
|
}
|
|
define void @mem_put_f32(ptr %p, i32 %i, i32 %fx) {
|
|
entry:
|
|
%q = getelementptr inbounds float, ptr %p, i32 %i
|
|
%f = sitofp i32 %fx to float
|
|
%s = fmul float %f, 0x3EF0000000000000
|
|
store float %s, ptr %q, align 1
|
|
ret void
|
|
}
|
|
define i32 @mem_get_f32_bits(ptr %p, i32 %i) {
|
|
entry:
|
|
%q = getelementptr inbounds i32, ptr %p, i32 %i
|
|
%v = load i32, ptr %q, align 1
|
|
ret i32 %v
|
|
}
|
|
define void @mem_put_f32_bits(ptr %p, i32 %i, i32 %bits) {
|
|
entry:
|
|
%q = getelementptr inbounds i32, ptr %p, i32 %i
|
|
store i32 %bits, ptr %q, align 1
|
|
ret void
|
|
}
|
|
define void @mem_copy(ptr %dst, ptr %src, i32 %n) {
|
|
entry:
|
|
%n64 = sext i32 %n to i64
|
|
call void @llvm.memcpy.p0.p0.i64(ptr %dst, ptr %src, i64 %n64, i1 false)
|
|
ret void
|
|
}
|
|
define void @mem_set(ptr %dst, i32 %v, i32 %n) {
|
|
entry:
|
|
%n64 = sext i32 %n to i64
|
|
%b = trunc i32 %v to i8
|
|
call void @llvm.memset.p0.i64(ptr %dst, i8 %b, i64 %n64, i1 false)
|
|
ret void
|
|
}
|
|
|
|
; ---- IEEE float arithmetic on bit patterns ------------------------------------
|
|
define i32 @f_add(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%r = fadd float %x, %y
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_sub(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%r = fsub float %x, %y
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_mul(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%r = fmul float %x, %y
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_div(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%r = fdiv float %x, %y
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_neg(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = fneg float %x
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_sqrt(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @llvm.sqrt.f32(float %x)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_abs(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @llvm.fabs.f32(float %x)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_sin(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @sinf(float %x)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_cos(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @cosf(float %x)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_tan(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @tanf(float %x)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_atan2(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%r = call float @atan2f(float %x, float %y)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_pow(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%r = call float @powf(float %x, float %y)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_exp(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @expf(float %x)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_log(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @logf(float %x)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_floor(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @floorf(float %x)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_mod(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%r = call float @fmodf(float %x, float %y)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_ldexp(i32 %a, i32 %e) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = call float @ldexpf(float %x, i32 %e)
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_min(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%c = fcmp olt float %x, %y
|
|
%r = select i1 %c, float %x, float %y
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_max(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%c = fcmp ogt float %x, %y
|
|
%r = select i1 %c, float %x, float %y
|
|
%o = bitcast float %r to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_lt(i32 %a, i32 %b) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%y = bitcast i32 %b to float
|
|
%c = fcmp olt float %x, %y
|
|
%z = zext i1 %c to i32
|
|
ret i32 %z
|
|
}
|
|
define i32 @f_from_int(i32 %a) {
|
|
entry:
|
|
%f = sitofp i32 %a to float
|
|
%o = bitcast float %f to i32
|
|
ret i32 %o
|
|
}
|
|
define i32 @f_to_int(i32 %a) {
|
|
entry:
|
|
%x = bitcast i32 %a to float
|
|
%r = fptosi float %x to i32
|
|
ret i32 %r
|
|
}
|
|
|
|
; ---- a real monotonic microsecond clock -------------------------------------
|
|
; Time.now() is whole seconds and Time.delta() is a fixed 60 Hz timestep, so
|
|
; neither can measure a frame. gettimeofday gives the wall clock in microseconds,
|
|
; which is what frame pacing and hitch measurement actually need.
|
|
; macOS arm64: struct timeval is { time_t tv_sec (i64), suseconds_t tv_usec (i32) }
|
|
%struct.timeval64 = type { i64, i32 }
|
|
declare i32 @gettimeofday(ptr, ptr)
|
|
declare i32 @nanosleep(ptr, ptr)
|
|
|
|
; Give the CPU back for `us` microseconds. A frame limiter needs this: without a way to wait, a
|
|
; cap can only be a spin, which burns a core and cooks a laptop. The caller sleeps SHORT of its
|
|
; target and spins the last stretch, because no OS sleep is exact.
|
|
define void @gl_sleep_us(i64 %us) {
|
|
entry:
|
|
%pos = icmp sgt i64 %us, 0
|
|
br i1 %pos, label %go, label %out
|
|
go:
|
|
%ts = alloca [16 x i8], align 8
|
|
%sec = sdiv i64 %us, 1000000
|
|
%rem = srem i64 %us, 1000000
|
|
%nsec = mul i64 %rem, 1000
|
|
store i64 %sec, ptr %ts
|
|
%np = getelementptr i8, ptr %ts, i64 8
|
|
store i64 %nsec, ptr %np
|
|
%r = call i32 @nanosleep(ptr %ts, ptr null)
|
|
br label %out
|
|
out:
|
|
ret void
|
|
}
|
|
|
|
define i64 @gl_now_us() {
|
|
entry:
|
|
%tv = alloca %struct.timeval64, align 8
|
|
%r = call i32 @gettimeofday(ptr %tv, ptr null)
|
|
%secp = getelementptr inbounds %struct.timeval64, ptr %tv, i32 0, i32 0
|
|
%usp = getelementptr inbounds %struct.timeval64, ptr %tv, i32 0, i32 1
|
|
%sec = load i64, ptr %secp, align 8
|
|
%us32 = load i32, ptr %usp, align 8
|
|
%us = sext i32 %us32 to i64
|
|
%m = mul i64 %sec, 1000000
|
|
%t = add i64 %m, %us
|
|
ret i64 %t
|
|
}
|
|
|
|
; headless on macOS: no window, so nothing for a graphics API to make a surface on
|
|
define weak ptr @win_native_window() {
|
|
entry:
|
|
ret ptr null
|
|
}
|
|
define weak ptr @win_native_instance() {
|
|
entry:
|
|
ret ptr null
|
|
}
|
|
define weak ptr @win_metal_layer() {
|
|
entry:
|
|
ret ptr null
|
|
}
|
|
; the window is always drawn to here (cocoa.ll asks macOS whether it is covered)
|
|
define weak i32 @win_visible() {
|
|
entry:
|
|
ret i32 1
|
|
}
|
|
; the drawable's size, for a graphics API that follows the window without a GL context (render3d's
|
|
; Vulkan resize check); headless there is no window, so 0 x 0. cocoa.ll's real one wins when linked.
|
|
; win_set_title: cocoa.ll's in a windowed build; nothing to retitle headless
|
|
define weak void @win_set_title(ptr %title) {
|
|
entry:
|
|
ret void
|
|
}
|
|
|
|
define weak void @win_gl_drawable(ptr %out) {
|
|
entry:
|
|
store i32 0, ptr %out
|
|
%o1 = getelementptr i32, ptr %out, i32 1
|
|
store i32 0, ptr %o1
|
|
ret void
|
|
}
|