feat(gl): OpenGL 4.1 and the ludic.render3d renderer
`Gl.*` binds the whole OpenGL 4.1 core API — every entry point of the platform gl3.h with every GL_* constant, generated by `ludic-dev glgen` with per-call ABI thunks. Windowed builds get an NSOpenGLContext on the existing window at Retina resolution; headless builds render into an offscreen CGL context, so a program that uses Gl.* renders and screenshots identically under the test harness. It links gl.ll, the thunks and OpenGL.framework only when used; every other build stays byte-identical. packages/ludic.render3d is a physically based renderer written on that surface: HDRI image-based lighting, GPU-generated terrain with scanned PBR materials, CDLOD, cascaded shadows, glTF with skinning, instanced vegetation with impostors, procedural grass, water, SSAO, and an HDR pipeline with bloom, auto-exposure and ACES. It also carries this session's work on it: the terrain at half its cost (10.3 -> 5.4 ms of frame), the streaming hitch that got worse the longer you played, a resize that emptied the world, and the packaging that lets a game use the renderer from its own repository — `ludic assets`, the material manifest shipping with the package, and shader lookup falling back to the install root. See changes/ for each, with its numbers. The camping game that drove all of it has moved out to its own repository, Maroon Lake; examples/rendering/smooth.ludic stays as the renderer's example here. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
470971bf70
commit
f25289db20
90 changed files with 35316 additions and 19853 deletions
|
|
@ -81,6 +81,7 @@ declare i32 @CGWarpMouseCursorPosition(%NSPoint)
|
|||
@.s_type = private unnamed_addr constant [5 x i8] c"type\00"
|
||||
@.s_keycd = private unnamed_addr constant [8 x i8] c"keyCode\00"
|
||||
@.s_chars = private unnamed_addr constant [28 x i8] c"charactersIgnoringModifiers\00"
|
||||
@.s_modf = private unnamed_addr constant [14 x i8] c"modifierFlags\00"
|
||||
@.s_length = private unnamed_addr constant [7 x i8] c"length\00"
|
||||
@.s_charat = private unnamed_addr constant [18 x i8] c"characterAtIndex:\00"
|
||||
@.s_locwin = private unnamed_addr constant [17 x i8] c"locationInWindow\00"
|
||||
|
|
@ -267,9 +268,12 @@ entry:
|
|||
%sel_alloc = call ptr @sel_registerName(ptr @.s_alloc)
|
||||
%w0 = call ptr @objc_msgSend(ptr %wincls, ptr %sel_alloc)
|
||||
%sel_initw = call ptr @sel_registerName(ptr @.s_initw)
|
||||
; NSWindowStyleMaskTitled|Closable = 3, NSBackingStoreBuffered = 2
|
||||
%win = call ptr (ptr, ptr, %CGRect, i64, i64, i8) @objc_msgSend(ptr %w0, ptr %sel_initw, %CGRect %rect, i64 3, i64 2, i8 0)
|
||||
; NSWindowStyleMaskTitled|Closable|Resizable = 11, NSBackingStoreBuffered = 2
|
||||
%win = call ptr (ptr, ptr, %CGRect, i64, i64, i8) @objc_msgSend(ptr %w0, ptr %sel_initw, %CGRect %rect, i64 11, i64 2, i8 0)
|
||||
store ptr %win, ptr @W_win
|
||||
; NSWindowCollectionBehaviorFullScreenPrimary = 1 << 7: the green button and toggleFullScreen: work
|
||||
%sel_cb = call ptr @sel_registerName(ptr @.s_setcb)
|
||||
%acb = call ptr (ptr, ptr, i64) @objc_msgSend(ptr %win, ptr %sel_cb, i64 128)
|
||||
|
||||
%strcls = call ptr @objc_getClass(ptr @.c_str)
|
||||
%sel_utf8 = call ptr @sel_registerName(ptr @.s_utf8)
|
||||
|
|
@ -343,10 +347,10 @@ entry:
|
|||
i32 36, label %vret
|
||||
i32 53, label %vesc
|
||||
]
|
||||
vw: ret i32 119
|
||||
vs: ret i32 115
|
||||
va: ret i32 97
|
||||
vd: ret i32 100
|
||||
vw: ret i32 128
|
||||
vs: ret i32 129
|
||||
va: ret i32 130
|
||||
vd: ret i32 131
|
||||
vspace: ret i32 32
|
||||
vret: ret i32 10
|
||||
vesc: ret i32 27
|
||||
|
|
@ -361,7 +365,15 @@ chars:
|
|||
take:
|
||||
%c = call i16 (ptr, ptr, i64) @objc_msgSend(ptr %s, ptr %sel_cat, i64 0)
|
||||
%c32 = zext i16 %c to i32
|
||||
ret i32 %c32
|
||||
; charactersIgnoringModifiers still honours Shift, so Shift+W arrives as 'W' and a game
|
||||
; holding 'w' to walk stopped dead the moment the player held Shift to run. Fold A-Z to
|
||||
; a-z: the held set is keyed by the key, Shift itself is reported separately (code 16).
|
||||
%isupA = icmp sge i32 %c32, 65
|
||||
%isupB = icmp sle i32 %c32, 90
|
||||
%isupper = and i1 %isupA, %isupB
|
||||
%lower = add i32 %c32, 32
|
||||
%folded = select i1 %isupper, i32 %lower, i32 %c32
|
||||
ret i32 %folded
|
||||
none:
|
||||
ret i32 0
|
||||
}
|
||||
|
|
@ -401,11 +413,30 @@ handle:
|
|||
br i1 %iskey, label %key, label %notkey
|
||||
notkey:
|
||||
%isup = icmp eq i64 %ty, 11 ; NSEventTypeKeyUp
|
||||
br i1 %isup, label %keyup, label %mouse
|
||||
br i1 %isup, label %keyup, label %notup
|
||||
keyup: ; #50 — release the held key
|
||||
%uv = call i32 @ev_keyval(ptr %ev)
|
||||
call void @win_held_bit(i32 %uv, i32 0)
|
||||
br label %forward
|
||||
notup:
|
||||
%isflags = icmp eq i64 %ty, 12 ; NSEventTypeFlagsChanged: modifier keys
|
||||
br i1 %isflags, label %flags, label %mouse
|
||||
flags: ; Shift is held-key 16 (ctrl 17, alt 18): a run modifier
|
||||
%sel_mf = call ptr @sel_registerName(ptr @.s_modf)
|
||||
%mf = call i64 (ptr, ptr) @objc_msgSend(ptr %ev, ptr %sel_mf)
|
||||
%mf_sh = and i64 %mf, 131072 ; NSEventModifierFlagShift = 1 << 17
|
||||
%sh_on = icmp ne i64 %mf_sh, 0
|
||||
%sh_i = zext i1 %sh_on to i32
|
||||
call void @win_held_bit(i32 16, i32 %sh_i)
|
||||
%mf_ct = and i64 %mf, 262144 ; NSEventModifierFlagControl = 1 << 18
|
||||
%ct_on = icmp ne i64 %mf_ct, 0
|
||||
%ct_i = zext i1 %ct_on to i32
|
||||
call void @win_held_bit(i32 17, i32 %ct_i)
|
||||
%mf_al = and i64 %mf, 524288 ; NSEventModifierFlagOption = 1 << 19
|
||||
%al_on = icmp ne i64 %mf_al, 0
|
||||
%al_i = zext i1 %al_on to i32
|
||||
call void @win_held_bit(i32 18, i32 %al_i)
|
||||
br label %forward
|
||||
mouse: ; #50 — mouse buttons + wheel
|
||||
%ml_d = icmp eq i64 %ty, 1 ; NSEventTypeLeftMouseDown
|
||||
br i1 %ml_d, label %lset, label %ml_u
|
||||
|
|
@ -610,8 +641,17 @@ reldelta:
|
|||
%nmy2 = select i1 %yhi, i32 %fbhm, i32 %nmy1
|
||||
store i32 %nmx2, ptr @W_mx
|
||||
store i32 %nmy2, ptr @W_my
|
||||
; the raw motion too: a clamped cursor stops turning at the edge, the delta must not
|
||||
%p4r = getelementptr i32, ptr %out, i32 4
|
||||
store i32 %ddx, ptr %p4r
|
||||
%p5r = getelementptr i32, ptr %out, i32 5
|
||||
store i32 %ddy, ptr %p5r
|
||||
br label %emit
|
||||
abspos:
|
||||
%p4a = getelementptr i32, ptr %out, i32 4
|
||||
store i32 0, ptr %p4a
|
||||
%p5a = getelementptr i32, ptr %out, i32 5
|
||||
store i32 0, ptr %p5a
|
||||
%win = load ptr, ptr @W_win
|
||||
%nowin = icmp eq ptr %win, null
|
||||
br i1 %nowin, label %emit, label %qpos
|
||||
|
|
@ -1107,3 +1147,264 @@ body:
|
|||
ret:
|
||||
ret void
|
||||
}
|
||||
|
||||
; ============================================================================
|
||||
; OpenGL on the window (Gl.* — runtime/native/gl.ludic). An NSOpenGLContext is
|
||||
; attached to the existing LudicView, at the display's backing resolution, with
|
||||
; a 4.1 core profile. The CPU framebuffer path above is untouched: a Gl program
|
||||
; simply never calls Screen.show, so -drawRect: finds @W_fb null and paints
|
||||
; nothing. The window-independent GL pieces (offscreen CGL contexts, ABI thunks)
|
||||
; live in gl.ll so a headless build never references the window.
|
||||
;
|
||||
; win_gl_attach() -> ok create the context on @W_view (0 = no window yet)
|
||||
; win_gl_resize(w, h) content size in points, scale 1 (mouse mapping follows)
|
||||
; win_gl_swap() flushBuffer (vsync'd)
|
||||
; win_gl_scale() -> int backing pixels per point (2 on Retina)
|
||||
; ============================================================================
|
||||
|
||||
@.c_pixfmt = private unnamed_addr constant [20 x i8] c"NSOpenGLPixelFormat\00"
|
||||
@.c_glctx = private unnamed_addr constant [16 x i8] c"NSOpenGLContext\00"
|
||||
@.s_initattr = private unnamed_addr constant [20 x i8] c"initWithAttributes:\00"
|
||||
@.s_initfmt = private unnamed_addr constant [29 x i8] c"initWithFormat:shareContext:\00"
|
||||
@.s_setview = private unnamed_addr constant [9 x i8] c"setView:\00"
|
||||
@.s_makecur = private unnamed_addr constant [19 x i8] c"makeCurrentContext\00"
|
||||
@.s_flushbuf = private unnamed_addr constant [12 x i8] c"flushBuffer\00"
|
||||
@.s_ctxupd = private unnamed_addr constant [7 x i8] c"update\00"
|
||||
@.s_setvals = private unnamed_addr constant [24 x i8] c"setValues:forParameter:\00"
|
||||
@.s_bestres = private unnamed_addr constant [37 x i8] c"setWantsBestResolutionOpenGLSurface:\00"
|
||||
@.s_bscale = private unnamed_addr constant [19 x i8] c"backingScaleFactor\00"
|
||||
@.s_setcsize = private unnamed_addr constant [16 x i8] c"setContentSize:\00"
|
||||
@.s_togfs = private unnamed_addr constant [18 x i8] c"toggleFullScreen:\00"
|
||||
@.s_bounds = private unnamed_addr constant [7 x i8] c"bounds\00"
|
||||
@.s_setcb = private unnamed_addr constant [23 x i8] c"setCollectionBehavior:\00"
|
||||
|
||||
@W_glctx = internal global ptr null
|
||||
@W_glscale = internal global i32 1
|
||||
|
||||
define i32 @win_gl_attach() {
|
||||
entry:
|
||||
%view = load ptr, ptr @W_view
|
||||
%noview = icmp eq ptr %view, null
|
||||
br i1 %noview, label %fail, label %go
|
||||
go:
|
||||
%sel_alloc = call ptr @sel_registerName(ptr @.s_alloc)
|
||||
; Retina: ask for a backing-resolution surface before the context is attached
|
||||
%sel_br = call ptr @sel_registerName(ptr @.s_bestres)
|
||||
%r0 = call ptr (ptr, ptr, i8) @objc_msgSend(ptr %view, ptr %sel_br, i8 1)
|
||||
; NSOpenGLPixelFormatAttribute[]: OpenGLProfile=99 -> 4.1 core (0x4100),
|
||||
; ColorSize=8 -> 24, AlphaSize=11 -> 8, DepthSize=12 -> 24, StencilSize=13 -> 8,
|
||||
; DoubleBuffer=5, Accelerated=73, 0
|
||||
%attrs = alloca [14 x i32], align 4
|
||||
%a0 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 0
|
||||
store i32 99, ptr %a0
|
||||
%a1 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 1
|
||||
store i32 16640, ptr %a1
|
||||
%a2 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 2
|
||||
store i32 8, ptr %a2
|
||||
%a3 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 3
|
||||
store i32 24, ptr %a3
|
||||
%a4 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 4
|
||||
store i32 11, ptr %a4
|
||||
%a5 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 5
|
||||
store i32 8, ptr %a5
|
||||
%a6 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 6
|
||||
store i32 12, ptr %a6
|
||||
%a7 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 7
|
||||
store i32 24, ptr %a7
|
||||
%a8 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 8
|
||||
store i32 13, ptr %a8
|
||||
%a9 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 9
|
||||
store i32 8, ptr %a9
|
||||
%a10 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 10
|
||||
store i32 5, ptr %a10
|
||||
%a11 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 11
|
||||
store i32 73, ptr %a11
|
||||
%a12 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 12
|
||||
store i32 0, ptr %a12
|
||||
%a13 = getelementptr [14 x i32], ptr %attrs, i32 0, i32 13
|
||||
store i32 0, ptr %a13
|
||||
|
||||
%pfcls = call ptr @objc_getClass(ptr @.c_pixfmt)
|
||||
%pf0 = call ptr @objc_msgSend(ptr %pfcls, ptr %sel_alloc)
|
||||
%sel_ia = call ptr @sel_registerName(ptr @.s_initattr)
|
||||
%pf = call ptr (ptr, ptr, ptr) @objc_msgSend(ptr %pf0, ptr %sel_ia, ptr %attrs)
|
||||
%nopf = icmp eq ptr %pf, null
|
||||
br i1 %nopf, label %fail, label %ctx
|
||||
ctx:
|
||||
%ccls = call ptr @objc_getClass(ptr @.c_glctx)
|
||||
%c0 = call ptr @objc_msgSend(ptr %ccls, ptr %sel_alloc)
|
||||
%sel_if = call ptr @sel_registerName(ptr @.s_initfmt)
|
||||
%c = call ptr (ptr, ptr, ptr, ptr) @objc_msgSend(ptr %c0, ptr %sel_if, ptr %pf, ptr null)
|
||||
%noc = icmp eq ptr %c, null
|
||||
br i1 %noc, label %fail, label %attach
|
||||
attach:
|
||||
%sel_sv = call ptr @sel_registerName(ptr @.s_setview)
|
||||
%r1 = call ptr (ptr, ptr, ptr) @objc_msgSend(ptr %c, ptr %sel_sv, ptr %view)
|
||||
%sel_mc = call ptr @sel_registerName(ptr @.s_makecur)
|
||||
%r2 = call ptr @objc_msgSend(ptr %c, ptr %sel_mc)
|
||||
; swap interval 1 (NSOpenGLCPSwapInterval = 222)
|
||||
%one = alloca i32, align 4
|
||||
store i32 1, ptr %one
|
||||
%sel_sp = call ptr @sel_registerName(ptr @.s_setvals)
|
||||
%r3 = call ptr (ptr, ptr, ptr, i64) @objc_msgSend(ptr %c, ptr %sel_sp, ptr %one, i64 222)
|
||||
store ptr %c, ptr @W_glctx
|
||||
; backing scale factor of the window (2.0 on Retina)
|
||||
%win = load ptr, ptr @W_win
|
||||
%sel_bs = call ptr @sel_registerName(ptr @.s_bscale)
|
||||
%bs = call double (ptr, ptr) @objc_msgSend(ptr %win, ptr %sel_bs)
|
||||
%bsi = fptosi double %bs to i32
|
||||
%bsok = icmp sgt i32 %bsi, 0
|
||||
%bsv = select i1 %bsok, i32 %bsi, i32 1
|
||||
store i32 %bsv, ptr @W_glscale
|
||||
ret i32 1
|
||||
fail:
|
||||
ret i32 0
|
||||
}
|
||||
|
||||
define void @win_gl_resize(i32 %w, i32 %h) {
|
||||
entry:
|
||||
%win = load ptr, ptr @W_win
|
||||
%nowin = icmp eq ptr %win, null
|
||||
br i1 %nowin, label %out, label %go
|
||||
go:
|
||||
store i32 %w, ptr @W_fbw
|
||||
store i32 %h, ptr @W_fbh
|
||||
store i32 1, ptr @W_scale
|
||||
%wd = sitofp i32 %w to double
|
||||
%hd = sitofp i32 %h to double
|
||||
%s0 = insertvalue %NSPoint undef, double %wd, 0
|
||||
%sz = insertvalue %NSPoint %s0, double %hd, 1
|
||||
%sel_cs = call ptr @sel_registerName(ptr @.s_setcsize)
|
||||
%r0 = call ptr (ptr, ptr, %NSPoint) @objc_msgSend(ptr %win, ptr %sel_cs, %NSPoint %sz)
|
||||
%sel_center = call ptr @sel_registerName(ptr @.s_center)
|
||||
%r1 = call ptr @objc_msgSend(ptr %win, ptr %sel_center)
|
||||
%c = load ptr, ptr @W_glctx
|
||||
%noc = icmp eq ptr %c, null
|
||||
br i1 %noc, label %out, label %upd
|
||||
upd:
|
||||
%sel_u = call ptr @sel_registerName(ptr @.s_ctxupd)
|
||||
%r2 = call ptr @objc_msgSend(ptr %c, ptr %sel_u)
|
||||
br label %out
|
||||
out:
|
||||
ret void
|
||||
}
|
||||
|
||||
define void @win_gl_swap() {
|
||||
entry:
|
||||
%c = load ptr, ptr @W_glctx
|
||||
%noc = icmp eq ptr %c, null
|
||||
br i1 %noc, label %out, label %go
|
||||
go:
|
||||
%sel_fb = call ptr @sel_registerName(ptr @.s_flushbuf)
|
||||
%r = call ptr @objc_msgSend(ptr %c, ptr %sel_fb)
|
||||
br label %out
|
||||
out:
|
||||
ret void
|
||||
}
|
||||
|
||||
define i32 @win_gl_scale() {
|
||||
entry:
|
||||
%s = load i32, ptr @W_glscale
|
||||
ret i32 %s
|
||||
}
|
||||
|
||||
; swap interval: 1 = vsync (the default set at attach), 0 = free-running
|
||||
define void @win_gl_swap_interval(i32 %n) {
|
||||
entry:
|
||||
%c = load ptr, ptr @W_glctx
|
||||
%noc = icmp eq ptr %c, null
|
||||
br i1 %noc, label %out, label %go
|
||||
go:
|
||||
%v = alloca i32, align 4
|
||||
store i32 %n, ptr %v
|
||||
%sel_sp = call ptr @sel_registerName(ptr @.s_setvals)
|
||||
%r = call ptr (ptr, ptr, ptr, i64) @objc_msgSend(ptr %c, ptr %sel_sp, ptr %v, i64 222)
|
||||
br label %out
|
||||
out:
|
||||
ret void
|
||||
}
|
||||
|
||||
; ============================================================================
|
||||
; window size, full screen and the drawable (the game's video settings)
|
||||
; ============================================================================
|
||||
; the view's drawable in pixels -> out[0], out[1]; also keeps the points size the
|
||||
; mouse conversion uses in step with a window the user resized or made full screen
|
||||
define void @win_gl_drawable(ptr %out) {
|
||||
entry:
|
||||
%view = load ptr, ptr @W_view
|
||||
%noview = icmp eq ptr %view, null
|
||||
br i1 %noview, label %none, label %go
|
||||
go:
|
||||
%sel_b = call ptr @sel_registerName(ptr @.s_bounds)
|
||||
%r = call %CGRect (ptr, ptr) @objc_msgSend(ptr %view, ptr %sel_b)
|
||||
%w = extractvalue %CGRect %r, 2
|
||||
%h = extractvalue %CGRect %r, 3
|
||||
%sc = load i32, ptr @W_glscale
|
||||
%scd = sitofp i32 %sc to double
|
||||
%pw = fmul double %w, %scd
|
||||
%ph = fmul double %h, %scd
|
||||
%iw = fptosi double %pw to i32
|
||||
%ih = fptosi double %ph to i32
|
||||
store i32 %iw, ptr %out
|
||||
%p1 = getelementptr i32, ptr %out, i32 1
|
||||
store i32 %ih, ptr %p1
|
||||
%ipw = fptosi double %w to i32
|
||||
%iph = fptosi double %h to i32
|
||||
store i32 %ipw, ptr @W_fbw
|
||||
store i32 %iph, ptr @W_fbh
|
||||
ret void
|
||||
none:
|
||||
store i32 0, ptr %out
|
||||
%p1n = getelementptr i32, ptr %out, i32 1
|
||||
store i32 0, ptr %p1n
|
||||
ret void
|
||||
}
|
||||
; tell the context its view changed size
|
||||
define void @win_gl_update() {
|
||||
entry:
|
||||
%c = load ptr, ptr @W_glctx
|
||||
%noc = icmp eq ptr %c, null
|
||||
br i1 %noc, label %out, label %upd
|
||||
upd:
|
||||
%sel_u = call ptr @sel_registerName(ptr @.s_ctxupd)
|
||||
%r = call ptr @objc_msgSend(ptr %c, ptr %sel_u)
|
||||
br label %out
|
||||
out:
|
||||
ret void
|
||||
}
|
||||
define void @win_toggle_fullscreen() {
|
||||
entry:
|
||||
%win = load ptr, ptr @W_win
|
||||
%nowin = icmp eq ptr %win, null
|
||||
br i1 %nowin, label %out, label %go
|
||||
go:
|
||||
%sel_t = call ptr @sel_registerName(ptr @.s_togfs)
|
||||
%r = call ptr (ptr, ptr, ptr) @objc_msgSend(ptr %win, ptr %sel_t, ptr null)
|
||||
br label %out
|
||||
out:
|
||||
ret void
|
||||
}
|
||||
; Retina (backing-resolution) drawable on or off: the pixel scale follows
|
||||
define void @win_gl_retina(i32 %on) {
|
||||
entry:
|
||||
%view = load ptr, ptr @W_view
|
||||
%noview = icmp eq ptr %view, null
|
||||
br i1 %noview, label %out, label %go
|
||||
go:
|
||||
%sel_br = call ptr @sel_registerName(ptr @.s_bestres)
|
||||
%flag = trunc i32 %on to i8
|
||||
%r0 = call ptr (ptr, ptr, i8) @objc_msgSend(ptr %view, ptr %sel_br, i8 %flag)
|
||||
%win = load ptr, ptr @W_win
|
||||
%sel_bs = call ptr @sel_registerName(ptr @.s_bscale)
|
||||
%bs = call double (ptr, ptr) @objc_msgSend(ptr %win, ptr %sel_bs)
|
||||
%bsi = fptosi double %bs to i32
|
||||
%bsok = icmp sgt i32 %bsi, 0
|
||||
%bsv = select i1 %bsok, i32 %bsi, i32 1
|
||||
%ison = icmp ne i32 %on, 0
|
||||
%scale = select i1 %ison, i32 %bsv, i32 1
|
||||
store i32 %scale, ptr @W_glscale
|
||||
call void @win_gl_update()
|
||||
br label %out
|
||||
out:
|
||||
ret void
|
||||
}
|
||||
|
|
|
|||
368
runtime/native/gl.ll
Normal file
368
runtime/native/gl.ll
Normal file
|
|
@ -0,0 +1,368 @@
|
|||
; ============================================================================
|
||||
; gl.ll — the window-independent half of the OpenGL backend, in LLVM IR.
|
||||
;
|
||||
; Linked into any program that uses Gl.* (windowed or headless), together with
|
||||
; gl_thunks.ll (the generated per-entry-point ABI thunks) and -framework OpenGL.
|
||||
; Nothing here touches the window: the NSOpenGLContext lives in cocoa.ll.
|
||||
;
|
||||
; cgl_offscreen() -> ok a headless 4.1 core context (render into FBOs)
|
||||
; fx_to_f32(fx) -> bits Q16.16 -> IEEE float bits (an int)
|
||||
; f32_to_fx(bits) -> fx IEEE float bits -> Q16.16
|
||||
; mem_* raw little-endian reads/writes on a bytes buffer
|
||||
; f_* IEEE-754 float arithmetic on float bits
|
||||
; ============================================================================
|
||||
|
||||
declare i32 @CGLChoosePixelFormat(ptr, ptr, ptr)
|
||||
declare i32 @CGLCreateContext(ptr, ptr, ptr)
|
||||
declare i32 @CGLSetCurrentContext(ptr)
|
||||
declare i32 @CGLDestroyPixelFormat(ptr)
|
||||
declare void @llvm.memcpy.p0.p0.i64(ptr, ptr, i64, i1)
|
||||
declare void @llvm.memset.p0.i64(ptr, i8, i64, i1)
|
||||
declare float @sinf(float)
|
||||
declare float @cosf(float)
|
||||
declare float @tanf(float)
|
||||
declare float @atan2f(float, float)
|
||||
declare float @powf(float, float)
|
||||
declare float @expf(float)
|
||||
declare float @logf(float)
|
||||
declare float @floorf(float)
|
||||
declare float @fmodf(float, float)
|
||||
declare float @ldexpf(float, i32)
|
||||
declare float @llvm.sqrt.f32(float)
|
||||
declare float @llvm.fabs.f32(float)
|
||||
|
||||
define i32 @cgl_offscreen() {
|
||||
entry:
|
||||
; kCGLPFAAccelerated=73, kCGLPFAOpenGLProfile=99 -> kCGLOGLPVersion_GL4_Core (0x4100),
|
||||
; kCGLPFAColorSize=8 -> 24, kCGLPFADepthSize=12 -> 24, 0
|
||||
%attrs = alloca [8 x i32], align 4
|
||||
%a0 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 0
|
||||
store i32 73, ptr %a0
|
||||
%a1 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 1
|
||||
store i32 99, ptr %a1
|
||||
%a2 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 2
|
||||
store i32 16640, ptr %a2
|
||||
%a3 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 3
|
||||
store i32 8, ptr %a3
|
||||
%a4 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 4
|
||||
store i32 24, ptr %a4
|
||||
%a5 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 5
|
||||
store i32 12, ptr %a5
|
||||
%a6 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 6
|
||||
store i32 24, ptr %a6
|
||||
%a7 = getelementptr [8 x i32], ptr %attrs, i32 0, i32 7
|
||||
store i32 0, ptr %a7
|
||||
%pix = alloca ptr, align 8
|
||||
store ptr null, ptr %pix
|
||||
%npix = alloca i32, align 4
|
||||
%e1 = call i32 @CGLChoosePixelFormat(ptr %attrs, ptr %pix, ptr %npix)
|
||||
%p = load ptr, ptr %pix
|
||||
%nop = icmp eq ptr %p, null
|
||||
br i1 %nop, label %fail, label %mk
|
||||
mk:
|
||||
%ctx = alloca ptr, align 8
|
||||
store ptr null, ptr %ctx
|
||||
%e2 = call i32 @CGLCreateContext(ptr %p, ptr null, ptr %ctx)
|
||||
%c = load ptr, ptr %ctx
|
||||
%e3 = call i32 @CGLDestroyPixelFormat(ptr %p)
|
||||
%noc = icmp eq ptr %c, null
|
||||
br i1 %noc, label %fail, label %cur
|
||||
cur:
|
||||
%e4 = call i32 @CGLSetCurrentContext(ptr %c)
|
||||
ret i32 1
|
||||
fail:
|
||||
ret i32 0
|
||||
}
|
||||
|
||||
; ---- Q16.16 <-> IEEE float ---------------------------------------------------
|
||||
define i32 @fx_to_f32(i32 %fx) {
|
||||
entry:
|
||||
%f = sitofp i32 %fx to float
|
||||
%s = fmul float %f, 0x3EF0000000000000
|
||||
%b = bitcast float %s to i32
|
||||
ret i32 %b
|
||||
}
|
||||
define i32 @f32_to_fx(i32 %bits) {
|
||||
entry:
|
||||
%f = bitcast i32 %bits to float
|
||||
%s = fmul float %f, 65536.0
|
||||
%r = fptosi float %s to i32
|
||||
ret i32 %r
|
||||
}
|
||||
|
||||
; ---- raw memory ---------------------------------------------------------------
|
||||
define ptr @mem_off(ptr %p, i32 %off) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
||||
ret ptr %q
|
||||
}
|
||||
define i32 @mem_get_i32(ptr %p, i32 %off) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
||||
%v = load i32, ptr %q, align 1
|
||||
ret i32 %v
|
||||
}
|
||||
define void @mem_put_i32(ptr %p, i32 %off, i32 %v) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
||||
store i32 %v, ptr %q, align 1
|
||||
ret void
|
||||
}
|
||||
define i32 @mem_get_u16(ptr %p, i32 %off) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
||||
%v = load i16, ptr %q, align 1
|
||||
%z = zext i16 %v to i32
|
||||
ret i32 %z
|
||||
}
|
||||
define void @mem_put_u16(ptr %p, i32 %off, i32 %v) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
||||
%t = trunc i32 %v to i16
|
||||
store i16 %t, ptr %q, align 1
|
||||
ret void
|
||||
}
|
||||
define i32 @mem_get_u8(ptr %p, i32 %off) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
||||
%v = load i8, ptr %q
|
||||
%z = zext i8 %v to i32
|
||||
ret i32 %z
|
||||
}
|
||||
define void @mem_put_u8(ptr %p, i32 %off, i32 %v) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i8, ptr %p, i32 %off
|
||||
%t = trunc i32 %v to i8
|
||||
store i8 %t, ptr %q
|
||||
ret void
|
||||
}
|
||||
; float element i of a float buffer, as Q16.16 / as bits
|
||||
define i32 @mem_get_f32(ptr %p, i32 %i) {
|
||||
entry:
|
||||
%q = getelementptr inbounds float, ptr %p, i32 %i
|
||||
%f = load float, ptr %q, align 1
|
||||
%s = fmul float %f, 65536.0
|
||||
%r = fptosi float %s to i32
|
||||
ret i32 %r
|
||||
}
|
||||
define void @mem_put_f32(ptr %p, i32 %i, i32 %fx) {
|
||||
entry:
|
||||
%q = getelementptr inbounds float, ptr %p, i32 %i
|
||||
%f = sitofp i32 %fx to float
|
||||
%s = fmul float %f, 0x3EF0000000000000
|
||||
store float %s, ptr %q, align 1
|
||||
ret void
|
||||
}
|
||||
define i32 @mem_get_f32_bits(ptr %p, i32 %i) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i32, ptr %p, i32 %i
|
||||
%v = load i32, ptr %q, align 1
|
||||
ret i32 %v
|
||||
}
|
||||
define void @mem_put_f32_bits(ptr %p, i32 %i, i32 %bits) {
|
||||
entry:
|
||||
%q = getelementptr inbounds i32, ptr %p, i32 %i
|
||||
store i32 %bits, ptr %q, align 1
|
||||
ret void
|
||||
}
|
||||
define void @mem_copy(ptr %dst, ptr %src, i32 %n) {
|
||||
entry:
|
||||
%n64 = sext i32 %n to i64
|
||||
call void @llvm.memcpy.p0.p0.i64(ptr %dst, ptr %src, i64 %n64, i1 false)
|
||||
ret void
|
||||
}
|
||||
define void @mem_set(ptr %dst, i32 %v, i32 %n) {
|
||||
entry:
|
||||
%n64 = sext i32 %n to i64
|
||||
%b = trunc i32 %v to i8
|
||||
call void @llvm.memset.p0.i64(ptr %dst, i8 %b, i64 %n64, i1 false)
|
||||
ret void
|
||||
}
|
||||
|
||||
; ---- IEEE float arithmetic on bit patterns ------------------------------------
|
||||
define i32 @f_add(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%r = fadd float %x, %y
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_sub(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%r = fsub float %x, %y
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_mul(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%r = fmul float %x, %y
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_div(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%r = fdiv float %x, %y
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_neg(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = fneg float %x
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_sqrt(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @llvm.sqrt.f32(float %x)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_abs(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @llvm.fabs.f32(float %x)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_sin(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @sinf(float %x)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_cos(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @cosf(float %x)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_tan(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @tanf(float %x)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_atan2(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%r = call float @atan2f(float %x, float %y)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_pow(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%r = call float @powf(float %x, float %y)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_exp(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @expf(float %x)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_log(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @logf(float %x)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_floor(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @floorf(float %x)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_mod(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%r = call float @fmodf(float %x, float %y)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_ldexp(i32 %a, i32 %e) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = call float @ldexpf(float %x, i32 %e)
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_min(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%c = fcmp olt float %x, %y
|
||||
%r = select i1 %c, float %x, float %y
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_max(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%c = fcmp ogt float %x, %y
|
||||
%r = select i1 %c, float %x, float %y
|
||||
%o = bitcast float %r to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_lt(i32 %a, i32 %b) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%y = bitcast i32 %b to float
|
||||
%c = fcmp olt float %x, %y
|
||||
%z = zext i1 %c to i32
|
||||
ret i32 %z
|
||||
}
|
||||
define i32 @f_from_int(i32 %a) {
|
||||
entry:
|
||||
%f = sitofp i32 %a to float
|
||||
%o = bitcast float %f to i32
|
||||
ret i32 %o
|
||||
}
|
||||
define i32 @f_to_int(i32 %a) {
|
||||
entry:
|
||||
%x = bitcast i32 %a to float
|
||||
%r = fptosi float %x to i32
|
||||
ret i32 %r
|
||||
}
|
||||
|
||||
; ---- a real monotonic microsecond clock -------------------------------------
|
||||
; Time.now() is whole seconds and Time.delta() is a fixed 60 Hz timestep, so
|
||||
; neither can measure a frame. gettimeofday gives the wall clock in microseconds,
|
||||
; which is what frame pacing and hitch measurement actually need.
|
||||
; macOS arm64: struct timeval is { time_t tv_sec (i64), suseconds_t tv_usec (i32) }
|
||||
%struct.timeval64 = type { i64, i32 }
|
||||
declare i32 @gettimeofday(ptr, ptr)
|
||||
|
||||
define i64 @gl_now_us() {
|
||||
entry:
|
||||
%tv = alloca %struct.timeval64, align 8
|
||||
%r = call i32 @gettimeofday(ptr %tv, ptr null)
|
||||
%secp = getelementptr inbounds %struct.timeval64, ptr %tv, i32 0, i32 0
|
||||
%usp = getelementptr inbounds %struct.timeval64, ptr %tv, i32 0, i32 1
|
||||
%sec = load i64, ptr %secp, align 8
|
||||
%us32 = load i32, ptr %usp, align 8
|
||||
%us = sext i32 %us32 to i64
|
||||
%m = mul i64 %sec, 1000000
|
||||
%t = add i64 %m, %us
|
||||
ret i64 %t
|
||||
}
|
||||
282
runtime/native/gl.ludic
Normal file
282
runtime/native/gl.ludic
Normal file
|
|
@ -0,0 +1,282 @@
|
|||
# ============================================================================
|
||||
# gl.ludic — Gl.*: OpenGL for Ludic.
|
||||
#
|
||||
# The whole OpenGL 4.1 core API is available as Gl.<snake_name>(...) — every
|
||||
# entry point of the platform gl3.h is bound in gl_api.ludic (generated), with
|
||||
# every GL_* constant. float/double parameters take `fixed`; buffers are the raw
|
||||
# `bytes`/`words` pointers Ludic already has, and pixel/vertex data is uploaded
|
||||
# from them as-is. This file adds the small amount of glue the API needs to be
|
||||
# usable from a game: a context on the window (or an offscreen one headless),
|
||||
# the swap, a screenshot, shader/program helpers, and IEEE float helpers so a
|
||||
# program can fill a vertex buffer with real floats from its Q16.16 math.
|
||||
#
|
||||
# Windowed: the NSOpenGLContext is attached to the existing LudicView (cocoa.ll)
|
||||
# at the display's backing resolution. Headless: a CGL context with no drawable
|
||||
# (gl.ll) and a framebuffer object that stands in for the screen, so the same
|
||||
# program renders and screenshots byte-identically under the test harness.
|
||||
# ============================================================================
|
||||
|
||||
import "gl_api.ludic"
|
||||
|
||||
# ---- native glue (cocoa.ll / gl.ll) -------------------------------------------
|
||||
extern function win_gl_attach() -> int = "win_gl_attach"
|
||||
extern function win_gl_resize(w: int, h: int) = "win_gl_resize"
|
||||
extern function win_gl_swap() = "win_gl_swap"
|
||||
extern function win_gl_scale() -> int = "win_gl_scale"
|
||||
extern function win_gl_swap_interval(n: int) = "win_gl_swap_interval"
|
||||
extern function win_gl_drawable(out: pointer) = "win_gl_drawable"
|
||||
extern function win_gl_update() = "win_gl_update"
|
||||
extern function win_gl_retina(on: int) = "win_gl_retina"
|
||||
extern function win_toggle_fullscreen() = "win_toggle_fullscreen"
|
||||
extern function cgl_offscreen() -> int = "cgl_offscreen"
|
||||
# wall clock in microseconds — the only sub-second clock available to a Ludic program
|
||||
extern function gl_now_us() -> long = "gl_now_us"
|
||||
extern function fx_to_f32(fx: fixed) -> int = "fx_to_f32"
|
||||
extern function f32_to_fx(bits: int) -> fixed = "f32_to_fx"
|
||||
extern function mem_off(p: pointer, off: int) -> pointer = "mem_off"
|
||||
extern function mem_get_i32(p: pointer, off: int) -> int = "mem_get_i32"
|
||||
extern function mem_put_i32(p: pointer, off: int, v: int) = "mem_put_i32"
|
||||
extern function mem_get_u16(p: pointer, off: int) -> int = "mem_get_u16"
|
||||
extern function mem_put_u16(p: pointer, off: int, v: int) = "mem_put_u16"
|
||||
extern function mem_get_u8(p: pointer, off: int) -> int = "mem_get_u8"
|
||||
extern function mem_put_u8(p: pointer, off: int, v: int) = "mem_put_u8"
|
||||
extern function mem_get_f32(p: pointer, i: int) -> fixed = "mem_get_f32"
|
||||
extern function mem_put_f32(p: pointer, i: int, v: fixed) = "mem_put_f32"
|
||||
extern function mem_get_f32_bits(p: pointer, i: int) -> int = "mem_get_f32_bits"
|
||||
extern function mem_put_f32_bits(p: pointer, i: int, bits: int) = "mem_put_f32_bits"
|
||||
extern function mem_copy(dst: pointer, src: pointer, n: int) = "mem_copy"
|
||||
extern function mem_set(dst: pointer, v: int, n: int) = "mem_set"
|
||||
# IEEE-754 single precision, carried as its bit pattern in an int
|
||||
extern function f_add(a: int, b: int) -> int = "f_add"
|
||||
extern function f_sub(a: int, b: int) -> int = "f_sub"
|
||||
extern function f_mul(a: int, b: int) -> int = "f_mul"
|
||||
extern function f_div(a: int, b: int) -> int = "f_div"
|
||||
extern function f_neg(a: int) -> int = "f_neg"
|
||||
extern function f_sqrt(a: int) -> int = "f_sqrt"
|
||||
extern function f_abs(a: int) -> int = "f_abs"
|
||||
extern function f_sin(a: int) -> int = "f_sin"
|
||||
extern function f_cos(a: int) -> int = "f_cos"
|
||||
extern function f_tan(a: int) -> int = "f_tan"
|
||||
extern function f_atan2(a: int, b: int) -> int = "f_atan2"
|
||||
extern function f_pow(a: int, b: int) -> int = "f_pow"
|
||||
extern function f_exp(a: int) -> int = "f_exp"
|
||||
extern function f_log(a: int) -> int = "f_log"
|
||||
extern function f_floor(a: int) -> int = "f_floor"
|
||||
extern function f_mod(a: int, b: int) -> int = "f_mod"
|
||||
extern function f_ldexp(a: int, e: int) -> int = "f_ldexp"
|
||||
extern function f_min(a: int, b: int) -> int = "f_min"
|
||||
extern function f_max(a: int, b: int) -> int = "f_max"
|
||||
extern function f_lt(a: int, b: int) -> int = "f_lt"
|
||||
extern function f_from_int(a: int) -> int = "f_from_int"
|
||||
extern function f_to_int(a: int) -> int = "f_to_int"
|
||||
|
||||
# ---- state --------------------------------------------------------------------
|
||||
var gl_is_open: bool = false
|
||||
var gl_w: int = 0 # drawable width, in pixels
|
||||
var gl_h: int = 0
|
||||
var gl_scale: int = 1 # backing pixels per window point
|
||||
var gl_screen: int = 0 # the framebuffer that is "the screen" (an FBO headless)
|
||||
var gl_ids: words = null # one-word scratch for glGen*/glGet*
|
||||
|
||||
function gl_scratch() -> words {
|
||||
if gl_ids == null { gl_ids = words(4) }
|
||||
return gl_ids
|
||||
}
|
||||
|
||||
# Open a GL 4.1 core context on a w x h (points) window titled `title`; headless,
|
||||
# an offscreen context with a w x h framebuffer standing in for the screen.
|
||||
function gl_open(width: int, height: int, title: pointer) -> bool {
|
||||
if gl_is_open { return true }
|
||||
if is_windowed() {
|
||||
if win_gl_attach() == 0 {
|
||||
win_open(width, height, 1, title) # a plain program: no window yet
|
||||
if win_gl_attach() == 0 { return false }
|
||||
}
|
||||
win_gl_resize(width, height)
|
||||
gl_scale = win_gl_scale()
|
||||
gl_w = width * gl_scale
|
||||
gl_h = height * gl_scale
|
||||
gl_screen = 0
|
||||
} else {
|
||||
if cgl_offscreen() == 0 { return false }
|
||||
gl_scale = 1
|
||||
gl_w = width
|
||||
gl_h = height
|
||||
gl_screen = gl_make_screen_fbo(width, height)
|
||||
}
|
||||
gl_bind_framebuffer(GL_FRAMEBUFFER, gl_screen)
|
||||
gl_viewport(0, 0, gl_w, gl_h)
|
||||
gl_is_open = true
|
||||
return true
|
||||
}
|
||||
|
||||
function gl_make_screen_fbo(w: int, h: int) -> int {
|
||||
let ids = gl_scratch()
|
||||
gl_gen_framebuffers(1, ids)
|
||||
let fbo = ids[0]
|
||||
gl_bind_framebuffer(GL_FRAMEBUFFER, fbo)
|
||||
gl_gen_textures(1, ids)
|
||||
let tex = ids[0]
|
||||
gl_bind_texture(GL_TEXTURE_2D, tex)
|
||||
gl_tex_image2d(GL_TEXTURE_2D, 0, GL_RGBA8, w, h, 0, GL_RGBA, GL_UNSIGNED_BYTE, null)
|
||||
gl_tex_parameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
gl_tex_parameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
gl_framebuffer_texture2d(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, tex, 0)
|
||||
gl_gen_renderbuffers(1, ids)
|
||||
let rb = ids[0]
|
||||
gl_bind_renderbuffer(GL_RENDERBUFFER, rb)
|
||||
gl_renderbuffer_storage(GL_RENDERBUFFER, GL_DEPTH24_STENCIL8, w, h)
|
||||
gl_framebuffer_renderbuffer(GL_FRAMEBUFFER, GL_DEPTH_STENCIL_ATTACHMENT, GL_RENDERBUFFER, rb)
|
||||
return fbo
|
||||
}
|
||||
|
||||
# Did the drawable change size (a window drag, full screen, a Retina switch)? Then
|
||||
# gl_w / gl_h follow it and the caller rebuilds its screen-sized targets.
|
||||
var gl_size_buf: words = null
|
||||
function gl_resize_check() -> bool {
|
||||
if not is_windowed() or not gl_is_open { return false }
|
||||
if gl_size_buf == null { gl_size_buf = words(4) }
|
||||
win_gl_drawable(gl_size_buf)
|
||||
let w = gl_size_buf[0]; let h = gl_size_buf[1]
|
||||
if w <= 0 or h <= 0 { return false }
|
||||
if w == gl_w and h == gl_h { return false }
|
||||
win_gl_update()
|
||||
gl_w = w; gl_h = h
|
||||
gl_scale = win_gl_scale()
|
||||
gl_viewport(0, 0, gl_w, gl_h)
|
||||
return true
|
||||
}
|
||||
function gl_set_window(w: int, h: int) -> void { if is_windowed() { win_gl_resize(w, h) } }
|
||||
function gl_toggle_fullscreen() -> void { if is_windowed() { win_toggle_fullscreen() } }
|
||||
function gl_set_retina(on: bool) -> void { if is_windowed() { var v = 0; if on { v = 1 }; win_gl_retina(v) } }
|
||||
# vsync on (1, the default) or off (0); headless has nothing to sync to
|
||||
function gl_vsync(n: int) -> void { if is_windowed() { win_gl_swap_interval(n) } }
|
||||
function gl_width() -> int { return gl_w }
|
||||
function gl_height() -> int { return gl_h }
|
||||
function gl_screen_fbo() -> int { return gl_screen }
|
||||
function gl_pixel_scale() -> int { return gl_scale }
|
||||
|
||||
# Present the frame (vsync'd flushBuffer); headless, just finish the GPU work.
|
||||
function gl_swap() -> void {
|
||||
if is_windowed() { win_gl_swap() }
|
||||
else { gl_finish() }
|
||||
}
|
||||
|
||||
# Write what is on the screen framebuffer to a binary PPM (call before Gl.swap).
|
||||
function gl_screenshot(path: pointer) -> bool {
|
||||
let w = gl_w
|
||||
let h = gl_h
|
||||
let f = file_open(path, "wb")
|
||||
if f == null { return false }
|
||||
let buf = bytes(w * h * 3)
|
||||
gl_bind_framebuffer(GL_READ_FRAMEBUFFER, gl_screen)
|
||||
gl_pixel_storei(GL_PACK_ALIGNMENT, 1)
|
||||
gl_read_pixels(0, 0, w, h, GL_RGB, GL_UNSIGNED_BYTE, buf)
|
||||
let hdr = `P6\n{w} {h}\n255\n`
|
||||
file_write(f, hdr, len(hdr))
|
||||
var y = h - 1
|
||||
while y >= 0 {
|
||||
file_write(f, mem_off(buf, y * w * 3), w * 3)
|
||||
y -= 1
|
||||
}
|
||||
file_close(f)
|
||||
free(buf)
|
||||
return true
|
||||
}
|
||||
|
||||
# Print any pending GL error under a tag; returns the error code (0 = none).
|
||||
function gl_check(tag: pointer) -> int {
|
||||
let e = gl_get_error()
|
||||
if e != 0 { print(`gl error {e} at {tag}`) }
|
||||
return e
|
||||
}
|
||||
|
||||
# ---- shaders --------------------------------------------------------------------
|
||||
var gl_log_buf: string = null
|
||||
|
||||
# Compile one shader stage from source; 0 (and the info log on stdout) on failure.
|
||||
function gl_shader(kind: int, src: pointer) -> int {
|
||||
let id = gl_create_shader(kind)
|
||||
var srcs: pointers = bytes(8)
|
||||
srcs[0] = src
|
||||
gl_shader_source(id, 1, srcs, null)
|
||||
gl_compile_shader(id)
|
||||
let ids = gl_scratch()
|
||||
gl_get_shaderiv(id, GL_COMPILE_STATUS, ids)
|
||||
if ids[0] == 0 {
|
||||
if gl_log_buf == null { gl_log_buf = bytes(8192) }
|
||||
gl_get_shader_info_log(id, 8191, null, gl_log_buf)
|
||||
print("shader compile failed:")
|
||||
print(gl_log_buf)
|
||||
gl_delete_shader(id)
|
||||
return 0
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
# Link a program from a vertex + fragment source pair; 0 on failure.
|
||||
function gl_program(vs: pointer, fs: pointer) -> int {
|
||||
return gl_program5(vs, null, null, null, fs)
|
||||
}
|
||||
|
||||
# Link a program from up to five stages (null = stage absent).
|
||||
function gl_program5(vs: pointer, tcs: pointer, tes: pointer, gs: pointer, fs: pointer) -> int {
|
||||
let prog = gl_create_program()
|
||||
var ok = true
|
||||
if vs != null { let s = gl_shader(GL_VERTEX_SHADER, vs); if s == 0 { ok = false } else { gl_attach_shader(prog, s) } }
|
||||
if tcs != null { let s = gl_shader(GL_TESS_CONTROL_SHADER, tcs); if s == 0 { ok = false } else { gl_attach_shader(prog, s) } }
|
||||
if tes != null { let s = gl_shader(GL_TESS_EVALUATION_SHADER, tes); if s == 0 { ok = false } else { gl_attach_shader(prog, s) } }
|
||||
if gs != null { let s = gl_shader(GL_GEOMETRY_SHADER, gs); if s == 0 { ok = false } else { gl_attach_shader(prog, s) } }
|
||||
if fs != null { let s = gl_shader(GL_FRAGMENT_SHADER, fs); if s == 0 { ok = false } else { gl_attach_shader(prog, s) } }
|
||||
if not ok { gl_delete_program(prog); return 0 }
|
||||
gl_link_program(prog)
|
||||
let ids = gl_scratch()
|
||||
gl_get_programiv(prog, GL_LINK_STATUS, ids)
|
||||
if ids[0] == 0 {
|
||||
if gl_log_buf == null { gl_log_buf = bytes(8192) }
|
||||
gl_get_program_info_log(prog, 8191, null, gl_log_buf)
|
||||
print("program link failed:")
|
||||
print(gl_log_buf)
|
||||
gl_delete_program(prog)
|
||||
return 0
|
||||
}
|
||||
return prog
|
||||
}
|
||||
|
||||
function gl_uniform(prog: int, name: pointer) -> int { return gl_get_uniform_location(prog, name) }
|
||||
|
||||
# ---- buffers of floats ------------------------------------------------------------
|
||||
# A float buffer is plain memory: n IEEE floats, filled from fixed (Gl.put) or
|
||||
# from float bits (Gl.put_bits), uploaded with Gl.buffer_data(…, Gl.bytes_of(n), buf, …).
|
||||
function gl_floats(n: int) -> pointer { return bytes(n * 4) }
|
||||
function gl_bytes_of(n: int) -> int { return n * 4 }
|
||||
function gl_put(buf: pointer, i: int, v: fixed) -> void { mem_put_f32(buf, i, v) }
|
||||
function gl_get(buf: pointer, i: int) -> fixed { return mem_get_f32(buf, i) }
|
||||
function gl_put_bits(buf: pointer, i: int, bits: int) -> void { mem_put_f32_bits(buf, i, bits) }
|
||||
function gl_get_bits(buf: pointer, i: int) -> int { return mem_get_f32_bits(buf, i) }
|
||||
function gl_ptr(buf: pointer, byte_offset: int) -> pointer { return mem_off(buf, byte_offset) }
|
||||
function gl_f32(v: fixed) -> int { return fx_to_f32(v) }
|
||||
function gl_fixed(bits: int) -> fixed { return f32_to_fx(bits) }
|
||||
|
||||
# One VAO, one VBO helper: create a vertex array object and return it, bound.
|
||||
function gl_vao() -> int {
|
||||
let ids = gl_scratch()
|
||||
gl_gen_vertex_arrays(1, ids)
|
||||
gl_bind_vertex_array(ids[0])
|
||||
return ids[0]
|
||||
}
|
||||
function gl_buffer() -> int {
|
||||
let ids = gl_scratch()
|
||||
gl_gen_buffers(1, ids)
|
||||
return ids[0]
|
||||
}
|
||||
function gl_texture() -> int {
|
||||
let ids = gl_scratch()
|
||||
gl_gen_textures(1, ids)
|
||||
return ids[0]
|
||||
}
|
||||
function gl_framebuffer() -> int {
|
||||
let ids = gl_scratch()
|
||||
gl_gen_framebuffers(1, ids)
|
||||
return ids[0]
|
||||
}
|
||||
1387
runtime/native/gl_api.ludic
Normal file
1387
runtime/native/gl_api.ludic
Normal file
File diff suppressed because it is too large
Load diff
3163
runtime/native/gl_thunks.ll
Normal file
3163
runtime/native/gl_thunks.ll
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -7,9 +7,12 @@
|
|||
# a dependency to read a sprite is a poor trade when the algorithm is this
|
||||
# small. So it lives here, in the language.
|
||||
#
|
||||
# The decoder is the canonical-Huffman formulation from Mark Adler's `puff`:
|
||||
# a symbol table plus per-length counts, walked one bit at a time. Slower than
|
||||
# a lookup-table decoder, and entirely fast enough to load sprites at startup.
|
||||
# The decoder is the canonical-Huffman formulation from Mark Adler's `puff`: a
|
||||
# symbol table plus per-length counts. Short codes (<= Z_FAST bits, which is the
|
||||
# overwhelming majority) resolve in a single lookup out of a 512-entry table built
|
||||
# with the symbol table; longer ones fall back to puff's walk, one bit at a time.
|
||||
# The walk alone was fine for sprites, but a PBR scene inflates hundreds of
|
||||
# megabytes of texture at load, and there the table is worth its 2 KB.
|
||||
# ============================================================================
|
||||
|
||||
# ---- bit reader (DEFLATE packs bits least-significant-first) ---------------
|
||||
|
|
@ -21,6 +24,7 @@ var z_bitcnt: int = 0
|
|||
var z_err: int = 0
|
||||
|
||||
function z_start(src: pointer, len: int) -> void {
|
||||
z_tables_once()
|
||||
z_src = src
|
||||
z_len = len
|
||||
z_pos = 0
|
||||
|
|
@ -29,6 +33,17 @@ function z_start(src: pointer, len: int) -> void {
|
|||
z_err = 0
|
||||
}
|
||||
|
||||
# Fill the bit buffer to at least `n` bits without consuming any (n <= 16, so the
|
||||
# buffer never shifts a byte past bit 15 and cannot reach the sign bit).
|
||||
function z_need(n: int) -> void {
|
||||
while z_bitcnt < n {
|
||||
if z_pos >= z_len { return }
|
||||
z_bitbuf = (z_bitbuf | (z_src[z_pos] << z_bitcnt))
|
||||
z_pos += 1
|
||||
z_bitcnt += 8
|
||||
}
|
||||
}
|
||||
|
||||
function z_bits(need: int) -> int {
|
||||
var val = z_bitbuf
|
||||
while z_bitcnt < need {
|
||||
|
|
@ -46,10 +61,17 @@ function z_bits(need: int) -> int {
|
|||
}
|
||||
|
||||
# ---- Huffman tables -------------------------------------------------------
|
||||
# A table is a single buffer: 16 length-counts followed by the symbols in
|
||||
# canonical order. One allocation, no structs.
|
||||
# One buffer per table: 16 length-counts, then a Z_FASTSZ-entry lookup keyed by
|
||||
# the next Z_FAST bits of the stream, then the symbols in canonical order.
|
||||
# A lookup entry is (length << 16) | symbol, or 0 when no code that short matches.
|
||||
# Z_FAST = 10 measured fastest over a 493 MB corpus (9 and 11 are both ~8% slower:
|
||||
# 9 misses the table more often, 11 spends more clearing it per dynamic block).
|
||||
const Z_FAST: int = 10
|
||||
const Z_FASTSZ: int = 1024 # 1 << Z_FAST
|
||||
const Z_SYMS: int = 1040 # 16 + Z_FASTSZ: where the symbols start
|
||||
|
||||
function z_table_new(nsym: int) -> pointer {
|
||||
return words((16 + nsym))
|
||||
return words((Z_SYMS + nsym))
|
||||
}
|
||||
|
||||
# lengths[i] = code length of symbol i (0 = symbol unused)
|
||||
|
|
@ -71,14 +93,65 @@ function z_table_build(table: words, lengths: words, n: int) -> void {
|
|||
for s in 0 .. n {
|
||||
let l = lengths[s]
|
||||
if l != 0 {
|
||||
table[16 + offs[l]] = s
|
||||
table[Z_SYMS + offs[l]] = s
|
||||
offs[l] += 1
|
||||
}
|
||||
}
|
||||
free(offs)
|
||||
|
||||
# ---- the fast lookup ----
|
||||
for i in 0 .. Z_FASTSZ {
|
||||
table[16 + i] = 0
|
||||
}
|
||||
# first canonical code of each length
|
||||
let firstc: words = words(17)
|
||||
var code = 0
|
||||
for l in 1 .. 16 {
|
||||
code = ((code + table[l - 1]) << 1)
|
||||
firstc[l] = code
|
||||
}
|
||||
var idx = 0
|
||||
for l in 1 .. 16 {
|
||||
let cnt = table[l]
|
||||
var k = 0
|
||||
while k < cnt {
|
||||
let sym = table[Z_SYMS + idx]
|
||||
if l <= Z_FAST {
|
||||
# DEFLATE reads a code most-significant-bit first out of a stream packed
|
||||
# least-significant-bit first, so the table is keyed by the reversed code
|
||||
let c = firstc[l] + k
|
||||
var rev = 0
|
||||
var b = 0
|
||||
while b < l {
|
||||
rev = ((rev << 1) | ((c >> b) & 1))
|
||||
b += 1
|
||||
}
|
||||
let entry = ((l << 16) | sym)
|
||||
var j = rev
|
||||
while j < Z_FASTSZ {
|
||||
table[16 + j] = entry
|
||||
j += (1 << l)
|
||||
}
|
||||
}
|
||||
idx += 1
|
||||
k += 1
|
||||
}
|
||||
}
|
||||
free(firstc)
|
||||
}
|
||||
|
||||
function z_decode(table: words) -> int {
|
||||
z_need(Z_FAST)
|
||||
if z_bitcnt >= Z_FAST {
|
||||
let e = table[16 + (z_bitbuf & (Z_FASTSZ - 1))]
|
||||
if e != 0 {
|
||||
let l = (e >> 16)
|
||||
z_bitbuf = (z_bitbuf >> l)
|
||||
z_bitcnt -= l
|
||||
return (e & 65535)
|
||||
}
|
||||
}
|
||||
# a code longer than Z_FAST bits (or a stream too short to peek): walk it
|
||||
var code = 0
|
||||
var first = 0
|
||||
var index = 0
|
||||
|
|
@ -86,7 +159,7 @@ function z_decode(table: words) -> int {
|
|||
code = (code | z_bits(1))
|
||||
let count = table[len]
|
||||
if code - first < count {
|
||||
return table[16 + index + (code - first)]
|
||||
return table[Z_SYMS + index + (code - first)]
|
||||
}
|
||||
index += count
|
||||
first = ((first + count) << 1)
|
||||
|
|
@ -123,6 +196,29 @@ function z_dist_extra(sym: int) -> int {
|
|||
return (sym - 2) / 2
|
||||
}
|
||||
|
||||
# The RFC tables above are pure functions of the symbol; compute them once rather
|
||||
# than dividing per match.
|
||||
var z_lbase: words = null
|
||||
var z_lext: words = null
|
||||
var z_dbase: words = null
|
||||
var z_dext: words = null
|
||||
|
||||
function z_tables_once() -> void {
|
||||
if z_lbase != null { return }
|
||||
z_lbase = words(29)
|
||||
z_lext = words(29)
|
||||
for s in 0 .. 29 {
|
||||
z_lbase[s] = z_len_base(s)
|
||||
z_lext[s] = z_len_extra(s)
|
||||
}
|
||||
z_dbase = words(30)
|
||||
z_dext = words(30)
|
||||
for s in 0 .. 30 {
|
||||
z_dbase[s] = z_dist_base(s)
|
||||
z_dext[s] = z_dist_extra(s)
|
||||
}
|
||||
}
|
||||
|
||||
# ---- block decoders -------------------------------------------------------
|
||||
# `out` is the destination window; returns the new write position, or -1.
|
||||
function z_stored(out: pointer, at: int, cap: int) -> int {
|
||||
|
|
@ -156,15 +252,20 @@ function z_codes(out: pointer, at: int, cap: int, lit: pointer, dist: pointer) -
|
|||
if sym > 256 {
|
||||
let s = sym - 257
|
||||
if s >= 29 { return -1 }
|
||||
let length = z_len_base(s) + z_bits(z_len_extra(s))
|
||||
let length = z_lbase[s] + z_bits(z_lext[s])
|
||||
let d = z_decode(dist)
|
||||
if d < 0 { return -1 }
|
||||
let distance = z_dist_base(d) + z_bits(z_dist_extra(d))
|
||||
if d >= 30 { return -1 }
|
||||
let distance = z_dbase[d] + z_bits(z_dext[d])
|
||||
if distance > w { return -1 }
|
||||
for k in 0 .. length {
|
||||
if w >= cap { return -1 }
|
||||
out[w] = out[w - distance]
|
||||
if w + length > cap { return -1 } # bounds once, not per byte
|
||||
var sp = w - distance
|
||||
var k = 0
|
||||
while k < length {
|
||||
out[w] = out[sp]
|
||||
w += 1
|
||||
sp += 1
|
||||
k += 1
|
||||
}
|
||||
}
|
||||
sym = z_decode(lit)
|
||||
|
|
|
|||
|
|
@ -223,6 +223,9 @@ var in_mx0: int = 0 # x/y at the previous frame (for the delta)
|
|||
var in_my0: int = 0
|
||||
var in_mdx: int = 0 # delta this frame
|
||||
var in_mdy: int = 0
|
||||
var in_rdx: int = 0 # the raw motion the platform reports while captured
|
||||
var in_rdy: int = 0
|
||||
var in_cursor_mode: int = 0
|
||||
var in_mbtn: int = 0 # button bitmask (bit 0 left, 1 right, 2 middle)
|
||||
var in_wheel: int = 0 # wheel delta this frame
|
||||
# gamepads: connected flag, button bitmask, and IN_AXES fixed axes each
|
||||
|
|
@ -298,9 +301,11 @@ function input_device_commit(k: int, replaying: int) -> void {
|
|||
# headless, from the single polled key. Injection (in_sim) is OR-ed on top.
|
||||
if is_windowed() {
|
||||
win_held(in_dev)
|
||||
let mbuf = words(4) # [x, y, button-mask, wheel]
|
||||
let mbuf = words(6) # [x, y, button-mask, wheel, raw dx, raw dy]
|
||||
mbuf[4] = 0; mbuf[5] = 0
|
||||
win_mouse(mbuf)
|
||||
in_mx = mbuf[0]; in_my = mbuf[1]; in_mbtn = mbuf[2]; in_wheel = mbuf[3]
|
||||
in_rdx = mbuf[4]; in_rdy = mbuf[5]
|
||||
# #51 — feed the platform gamepad + touch state into the same buffers the
|
||||
# read APIs use. Each is windowed-only glue (win_pad / win_touch are DCE'd
|
||||
# in a headless build); on hardware they overwrite the injected state.
|
||||
|
|
@ -334,6 +339,8 @@ function input_device_commit(k: int, replaying: int) -> void {
|
|||
# platform above when windowed, by Input.set_mouse before this poll otherwise).
|
||||
in_mdx = in_mx - in_mx0
|
||||
in_mdy = in_my - in_my0
|
||||
# captured (mode 2): the cursor is a clamped reticle, the motion is the raw delta
|
||||
if is_windowed() and in_cursor_mode == 2 { in_mdx = in_rdx; in_mdy = in_rdy }
|
||||
in_mx0 = in_mx
|
||||
in_my0 = in_my
|
||||
}
|
||||
|
|
@ -438,6 +445,7 @@ enum CursorMode { Normal, Hidden, Locked, Confined } # Input.cursor_mode(mode:
|
|||
enum PadButton { A, B, X, Y, LeftShoulder, RightShoulder, Back, Start } # Input.bind_pad(button:) / pad_button
|
||||
enum MouseButton { Left, Right, Middle } # Input.mouse_down(button:)
|
||||
function input_cursor_mode(mode: int) -> void {
|
||||
in_cursor_mode = mode
|
||||
if is_windowed() { win_cursor_mode(mode) }
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -221,12 +221,40 @@ function jp_number(p: JP) -> Val {
|
|||
di -= 1
|
||||
}
|
||||
var raw = ip * 65536 + frac
|
||||
raw = jp_exponent(p, raw)
|
||||
if neg != 0 { raw = -raw }
|
||||
return value_fixed(raw)
|
||||
}
|
||||
if p.i < p.n and (p.s[p.i] == 'e' or p.s[p.i] == 'E') { # 1e-05: an exponent makes it a fixed
|
||||
var raw = jp_exponent(p, ip * 65536)
|
||||
if neg != 0 { raw = -raw }
|
||||
return value_fixed(raw)
|
||||
}
|
||||
if neg != 0 { ip = -ip }
|
||||
return value_int(ip)
|
||||
}
|
||||
# an optional exponent after a number's digits, applied to a raw Q16.16 value. Exporters
|
||||
# write noise like 7.49e-09 for a zero; a fixed rounds that to 0, which is what it was.
|
||||
function jp_exponent(p: JP, raw0: int) -> int {
|
||||
var raw = raw0
|
||||
if p.i >= p.n or (p.s[p.i] != 'e' and p.s[p.i] != 'E') { return raw }
|
||||
p.i += 1
|
||||
var eneg = 0
|
||||
if p.i < p.n and p.s[p.i] == '-' { eneg = 1; p.i += 1 }
|
||||
else if p.i < p.n and p.s[p.i] == '+' { p.i += 1 }
|
||||
var e = 0
|
||||
while p.i < p.n and p.s[p.i] >= '0' and p.s[p.i] <= '9' {
|
||||
e = e * 10 + (p.s[p.i] - 48)
|
||||
p.i += 1
|
||||
}
|
||||
if e > 12 { e = 12 }
|
||||
var k = 0
|
||||
while k < e {
|
||||
if eneg != 0 { raw = raw / 10 } else { raw = raw * 10 }
|
||||
k += 1
|
||||
}
|
||||
return raw
|
||||
}
|
||||
|
||||
function jp_list(p: JP) -> Val {
|
||||
let out = value_list()
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue