feat(gl): OpenGL 4.1 and the ludic.render3d renderer
`Gl.*` binds the whole OpenGL 4.1 core API — every entry point of the platform gl3.h with every GL_* constant, generated by `ludic-dev glgen` with per-call ABI thunks. Windowed builds get an NSOpenGLContext on the existing window at Retina resolution; headless builds render into an offscreen CGL context, so a program that uses Gl.* renders and screenshots identically under the test harness. It links gl.ll, the thunks and OpenGL.framework only when used; every other build stays byte-identical. packages/ludic.render3d is a physically based renderer written on that surface: HDRI image-based lighting, GPU-generated terrain with scanned PBR materials, CDLOD, cascaded shadows, glTF with skinning, instanced vegetation with impostors, procedural grass, water, SSAO, and an HDR pipeline with bloom, auto-exposure and ACES. It also carries this session's work on it: the terrain at half its cost (10.3 -> 5.4 ms of frame), the streaming hitch that got worse the longer you played, a resize that emptied the world, and the packaging that lets a game use the renderer from its own repository — `ludic assets`, the material manifest shipping with the package, and shader lookup falling back to the install root. See changes/ for each, with its numbers. The camping game that drove all of it has moved out to its own repository, Maroon Lake; examples/rendering/smooth.ludic stays as the renderer's example here. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
470971bf70
commit
f25289db20
90 changed files with 35316 additions and 19853 deletions
|
|
@ -7,9 +7,12 @@
|
|||
# a dependency to read a sprite is a poor trade when the algorithm is this
|
||||
# small. So it lives here, in the language.
|
||||
#
|
||||
# The decoder is the canonical-Huffman formulation from Mark Adler's `puff`:
|
||||
# a symbol table plus per-length counts, walked one bit at a time. Slower than
|
||||
# a lookup-table decoder, and entirely fast enough to load sprites at startup.
|
||||
# The decoder is the canonical-Huffman formulation from Mark Adler's `puff`: a
|
||||
# symbol table plus per-length counts. Short codes (<= Z_FAST bits, which is the
|
||||
# overwhelming majority) resolve in a single lookup out of a 512-entry table built
|
||||
# with the symbol table; longer ones fall back to puff's walk, one bit at a time.
|
||||
# The walk alone was fine for sprites, but a PBR scene inflates hundreds of
|
||||
# megabytes of texture at load, and there the table is worth its 2 KB.
|
||||
# ============================================================================
|
||||
|
||||
# ---- bit reader (DEFLATE packs bits least-significant-first) ---------------
|
||||
|
|
@ -21,6 +24,7 @@ var z_bitcnt: int = 0
|
|||
var z_err: int = 0
|
||||
|
||||
function z_start(src: pointer, len: int) -> void {
|
||||
z_tables_once()
|
||||
z_src = src
|
||||
z_len = len
|
||||
z_pos = 0
|
||||
|
|
@ -29,6 +33,17 @@ function z_start(src: pointer, len: int) -> void {
|
|||
z_err = 0
|
||||
}
|
||||
|
||||
# Fill the bit buffer to at least `n` bits without consuming any (n <= 16, so the
|
||||
# buffer never shifts a byte past bit 15 and cannot reach the sign bit).
|
||||
function z_need(n: int) -> void {
|
||||
while z_bitcnt < n {
|
||||
if z_pos >= z_len { return }
|
||||
z_bitbuf = (z_bitbuf | (z_src[z_pos] << z_bitcnt))
|
||||
z_pos += 1
|
||||
z_bitcnt += 8
|
||||
}
|
||||
}
|
||||
|
||||
function z_bits(need: int) -> int {
|
||||
var val = z_bitbuf
|
||||
while z_bitcnt < need {
|
||||
|
|
@ -46,10 +61,17 @@ function z_bits(need: int) -> int {
|
|||
}
|
||||
|
||||
# ---- Huffman tables -------------------------------------------------------
|
||||
# A table is a single buffer: 16 length-counts followed by the symbols in
|
||||
# canonical order. One allocation, no structs.
|
||||
# One buffer per table: 16 length-counts, then a Z_FASTSZ-entry lookup keyed by
|
||||
# the next Z_FAST bits of the stream, then the symbols in canonical order.
|
||||
# A lookup entry is (length << 16) | symbol, or 0 when no code that short matches.
|
||||
# Z_FAST = 10 measured fastest over a 493 MB corpus (9 and 11 are both ~8% slower:
|
||||
# 9 misses the table more often, 11 spends more clearing it per dynamic block).
|
||||
const Z_FAST: int = 10
|
||||
const Z_FASTSZ: int = 1024 # 1 << Z_FAST
|
||||
const Z_SYMS: int = 1040 # 16 + Z_FASTSZ: where the symbols start
|
||||
|
||||
function z_table_new(nsym: int) -> pointer {
|
||||
return words((16 + nsym))
|
||||
return words((Z_SYMS + nsym))
|
||||
}
|
||||
|
||||
# lengths[i] = code length of symbol i (0 = symbol unused)
|
||||
|
|
@ -71,14 +93,65 @@ function z_table_build(table: words, lengths: words, n: int) -> void {
|
|||
for s in 0 .. n {
|
||||
let l = lengths[s]
|
||||
if l != 0 {
|
||||
table[16 + offs[l]] = s
|
||||
table[Z_SYMS + offs[l]] = s
|
||||
offs[l] += 1
|
||||
}
|
||||
}
|
||||
free(offs)
|
||||
|
||||
# ---- the fast lookup ----
|
||||
for i in 0 .. Z_FASTSZ {
|
||||
table[16 + i] = 0
|
||||
}
|
||||
# first canonical code of each length
|
||||
let firstc: words = words(17)
|
||||
var code = 0
|
||||
for l in 1 .. 16 {
|
||||
code = ((code + table[l - 1]) << 1)
|
||||
firstc[l] = code
|
||||
}
|
||||
var idx = 0
|
||||
for l in 1 .. 16 {
|
||||
let cnt = table[l]
|
||||
var k = 0
|
||||
while k < cnt {
|
||||
let sym = table[Z_SYMS + idx]
|
||||
if l <= Z_FAST {
|
||||
# DEFLATE reads a code most-significant-bit first out of a stream packed
|
||||
# least-significant-bit first, so the table is keyed by the reversed code
|
||||
let c = firstc[l] + k
|
||||
var rev = 0
|
||||
var b = 0
|
||||
while b < l {
|
||||
rev = ((rev << 1) | ((c >> b) & 1))
|
||||
b += 1
|
||||
}
|
||||
let entry = ((l << 16) | sym)
|
||||
var j = rev
|
||||
while j < Z_FASTSZ {
|
||||
table[16 + j] = entry
|
||||
j += (1 << l)
|
||||
}
|
||||
}
|
||||
idx += 1
|
||||
k += 1
|
||||
}
|
||||
}
|
||||
free(firstc)
|
||||
}
|
||||
|
||||
function z_decode(table: words) -> int {
|
||||
z_need(Z_FAST)
|
||||
if z_bitcnt >= Z_FAST {
|
||||
let e = table[16 + (z_bitbuf & (Z_FASTSZ - 1))]
|
||||
if e != 0 {
|
||||
let l = (e >> 16)
|
||||
z_bitbuf = (z_bitbuf >> l)
|
||||
z_bitcnt -= l
|
||||
return (e & 65535)
|
||||
}
|
||||
}
|
||||
# a code longer than Z_FAST bits (or a stream too short to peek): walk it
|
||||
var code = 0
|
||||
var first = 0
|
||||
var index = 0
|
||||
|
|
@ -86,7 +159,7 @@ function z_decode(table: words) -> int {
|
|||
code = (code | z_bits(1))
|
||||
let count = table[len]
|
||||
if code - first < count {
|
||||
return table[16 + index + (code - first)]
|
||||
return table[Z_SYMS + index + (code - first)]
|
||||
}
|
||||
index += count
|
||||
first = ((first + count) << 1)
|
||||
|
|
@ -123,6 +196,29 @@ function z_dist_extra(sym: int) -> int {
|
|||
return (sym - 2) / 2
|
||||
}
|
||||
|
||||
# The RFC tables above are pure functions of the symbol; compute them once rather
|
||||
# than dividing per match.
|
||||
var z_lbase: words = null
|
||||
var z_lext: words = null
|
||||
var z_dbase: words = null
|
||||
var z_dext: words = null
|
||||
|
||||
function z_tables_once() -> void {
|
||||
if z_lbase != null { return }
|
||||
z_lbase = words(29)
|
||||
z_lext = words(29)
|
||||
for s in 0 .. 29 {
|
||||
z_lbase[s] = z_len_base(s)
|
||||
z_lext[s] = z_len_extra(s)
|
||||
}
|
||||
z_dbase = words(30)
|
||||
z_dext = words(30)
|
||||
for s in 0 .. 30 {
|
||||
z_dbase[s] = z_dist_base(s)
|
||||
z_dext[s] = z_dist_extra(s)
|
||||
}
|
||||
}
|
||||
|
||||
# ---- block decoders -------------------------------------------------------
|
||||
# `out` is the destination window; returns the new write position, or -1.
|
||||
function z_stored(out: pointer, at: int, cap: int) -> int {
|
||||
|
|
@ -156,15 +252,20 @@ function z_codes(out: pointer, at: int, cap: int, lit: pointer, dist: pointer) -
|
|||
if sym > 256 {
|
||||
let s = sym - 257
|
||||
if s >= 29 { return -1 }
|
||||
let length = z_len_base(s) + z_bits(z_len_extra(s))
|
||||
let length = z_lbase[s] + z_bits(z_lext[s])
|
||||
let d = z_decode(dist)
|
||||
if d < 0 { return -1 }
|
||||
let distance = z_dist_base(d) + z_bits(z_dist_extra(d))
|
||||
if d >= 30 { return -1 }
|
||||
let distance = z_dbase[d] + z_bits(z_dext[d])
|
||||
if distance > w { return -1 }
|
||||
for k in 0 .. length {
|
||||
if w >= cap { return -1 }
|
||||
out[w] = out[w - distance]
|
||||
if w + length > cap { return -1 } # bounds once, not per byte
|
||||
var sp = w - distance
|
||||
var k = 0
|
||||
while k < length {
|
||||
out[w] = out[sp]
|
||||
w += 1
|
||||
sp += 1
|
||||
k += 1
|
||||
}
|
||||
}
|
||||
sym = z_decode(lit)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue