`Gl.*` binds the whole OpenGL 4.1 core API — every entry point of the platform gl3.h with every GL_* constant, generated by `ludic-dev glgen` with per-call ABI thunks. Windowed builds get an NSOpenGLContext on the existing window at Retina resolution; headless builds render into an offscreen CGL context, so a program that uses Gl.* renders and screenshots identically under the test harness. It links gl.ll, the thunks and OpenGL.framework only when used; every other build stays byte-identical. packages/ludic.render3d is a physically based renderer written on that surface: HDRI image-based lighting, GPU-generated terrain with scanned PBR materials, CDLOD, cascaded shadows, glTF with skinning, instanced vegetation with impostors, procedural grass, water, SSAO, and an HDR pipeline with bloom, auto-exposure and ACES. It also carries this session's work on it: the terrain at half its cost (10.3 -> 5.4 ms of frame), the streaming hitch that got worse the longer you played, a resize that emptied the world, and the packaging that lets a game use the renderer from its own repository — `ludic assets`, the material manifest shipping with the package, and shader lookup falling back to the install root. See changes/ for each, with its numbers. The camping game that drove all of it has moved out to its own repository, Maroon Lake; examples/rendering/smooth.ludic stays as the renderer's example here. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
408 lines
12 KiB
Text
408 lines
12 KiB
Text
# ============================================================================
|
|
# inflate.ludic — DEFLATE decompression (RFC 1951), written in Ludic.
|
|
#
|
|
# PNG stores its pixels zlib-compressed, so decoding one means implementing
|
|
# inflate. The C runtime linked zlib for this. We don't: zlib is not present by
|
|
# default on every target Ludic compiles for (Windows especially), and shipping
|
|
# a dependency to read a sprite is a poor trade when the algorithm is this
|
|
# small. So it lives here, in the language.
|
|
#
|
|
# The decoder is the canonical-Huffman formulation from Mark Adler's `puff`: a
|
|
# symbol table plus per-length counts. Short codes (<= Z_FAST bits, which is the
|
|
# overwhelming majority) resolve in a single lookup out of a 512-entry table built
|
|
# with the symbol table; longer ones fall back to puff's walk, one bit at a time.
|
|
# The walk alone was fine for sprites, but a PBR scene inflates hundreds of
|
|
# megabytes of texture at load, and there the table is worth its 2 KB.
|
|
# ============================================================================
|
|
|
|
# ---- bit reader (DEFLATE packs bits least-significant-first) ---------------
|
|
var z_src: pointer = null
|
|
var z_len: int = 0
|
|
var z_pos: int = 0
|
|
var z_bitbuf: int = 0
|
|
var z_bitcnt: int = 0
|
|
var z_err: int = 0
|
|
|
|
function z_start(src: pointer, len: int) -> void {
|
|
z_tables_once()
|
|
z_src = src
|
|
z_len = len
|
|
z_pos = 0
|
|
z_bitbuf = 0
|
|
z_bitcnt = 0
|
|
z_err = 0
|
|
}
|
|
|
|
# Fill the bit buffer to at least `n` bits without consuming any (n <= 16, so the
|
|
# buffer never shifts a byte past bit 15 and cannot reach the sign bit).
|
|
function z_need(n: int) -> void {
|
|
while z_bitcnt < n {
|
|
if z_pos >= z_len { return }
|
|
z_bitbuf = (z_bitbuf | (z_src[z_pos] << z_bitcnt))
|
|
z_pos += 1
|
|
z_bitcnt += 8
|
|
}
|
|
}
|
|
|
|
function z_bits(need: int) -> int {
|
|
var val = z_bitbuf
|
|
while z_bitcnt < need {
|
|
if z_pos >= z_len {
|
|
z_err = 1
|
|
return 0
|
|
}
|
|
val = (val | (z_src[z_pos] << z_bitcnt))
|
|
z_pos += 1
|
|
z_bitcnt += 8
|
|
}
|
|
z_bitbuf = (val >> need)
|
|
z_bitcnt -= need
|
|
return (val & ((1 << need) - 1))
|
|
}
|
|
|
|
# ---- Huffman tables -------------------------------------------------------
|
|
# One buffer per table: 16 length-counts, then a Z_FASTSZ-entry lookup keyed by
|
|
# the next Z_FAST bits of the stream, then the symbols in canonical order.
|
|
# A lookup entry is (length << 16) | symbol, or 0 when no code that short matches.
|
|
# Z_FAST = 10 measured fastest over a 493 MB corpus (9 and 11 are both ~8% slower:
|
|
# 9 misses the table more often, 11 spends more clearing it per dynamic block).
|
|
const Z_FAST: int = 10
|
|
const Z_FASTSZ: int = 1024 # 1 << Z_FAST
|
|
const Z_SYMS: int = 1040 # 16 + Z_FASTSZ: where the symbols start
|
|
|
|
function z_table_new(nsym: int) -> pointer {
|
|
return words((Z_SYMS + nsym))
|
|
}
|
|
|
|
# lengths[i] = code length of symbol i (0 = symbol unused)
|
|
function z_table_build(table: words, lengths: words, n: int) -> void {
|
|
for i in 0 .. 16 {
|
|
table[i] = 0
|
|
}
|
|
for s in 0 .. n {
|
|
let l = lengths[s]
|
|
table[l] += 1
|
|
}
|
|
table[0] = 0 # length 0 means "not present"
|
|
# offset of each length's first symbol
|
|
let offs: words = words(16)
|
|
offs[1] = 0
|
|
for l in 1 .. 15 {
|
|
offs[l + 1] = offs[l] + table[l]
|
|
}
|
|
for s in 0 .. n {
|
|
let l = lengths[s]
|
|
if l != 0 {
|
|
table[Z_SYMS + offs[l]] = s
|
|
offs[l] += 1
|
|
}
|
|
}
|
|
free(offs)
|
|
|
|
# ---- the fast lookup ----
|
|
for i in 0 .. Z_FASTSZ {
|
|
table[16 + i] = 0
|
|
}
|
|
# first canonical code of each length
|
|
let firstc: words = words(17)
|
|
var code = 0
|
|
for l in 1 .. 16 {
|
|
code = ((code + table[l - 1]) << 1)
|
|
firstc[l] = code
|
|
}
|
|
var idx = 0
|
|
for l in 1 .. 16 {
|
|
let cnt = table[l]
|
|
var k = 0
|
|
while k < cnt {
|
|
let sym = table[Z_SYMS + idx]
|
|
if l <= Z_FAST {
|
|
# DEFLATE reads a code most-significant-bit first out of a stream packed
|
|
# least-significant-bit first, so the table is keyed by the reversed code
|
|
let c = firstc[l] + k
|
|
var rev = 0
|
|
var b = 0
|
|
while b < l {
|
|
rev = ((rev << 1) | ((c >> b) & 1))
|
|
b += 1
|
|
}
|
|
let entry = ((l << 16) | sym)
|
|
var j = rev
|
|
while j < Z_FASTSZ {
|
|
table[16 + j] = entry
|
|
j += (1 << l)
|
|
}
|
|
}
|
|
idx += 1
|
|
k += 1
|
|
}
|
|
}
|
|
free(firstc)
|
|
}
|
|
|
|
function z_decode(table: words) -> int {
|
|
z_need(Z_FAST)
|
|
if z_bitcnt >= Z_FAST {
|
|
let e = table[16 + (z_bitbuf & (Z_FASTSZ - 1))]
|
|
if e != 0 {
|
|
let l = (e >> 16)
|
|
z_bitbuf = (z_bitbuf >> l)
|
|
z_bitcnt -= l
|
|
return (e & 65535)
|
|
}
|
|
}
|
|
# a code longer than Z_FAST bits (or a stream too short to peek): walk it
|
|
var code = 0
|
|
var first = 0
|
|
var index = 0
|
|
for len in 1 .. 16 {
|
|
code = (code | z_bits(1))
|
|
let count = table[len]
|
|
if code - first < count {
|
|
return table[Z_SYMS + index + (code - first)]
|
|
}
|
|
index += count
|
|
first = ((first + count) << 1)
|
|
code = (code << 1)
|
|
}
|
|
z_err = 1
|
|
return -1
|
|
}
|
|
|
|
# ---- length / distance code tables (RFC 1951 section 3.2.5) ---------------
|
|
function z_len_base(sym: int) -> int {
|
|
if sym < 8 { return 3 + sym }
|
|
if sym == 28 { return 258 }
|
|
let extra = (sym - 4) / 4
|
|
let group = (1 << extra)
|
|
return 3 + ((group - 1) << 2) + 4 + (sym - 4 - extra * 4) * group
|
|
}
|
|
|
|
function z_len_extra(sym: int) -> int {
|
|
if sym < 8 { return 0 }
|
|
if sym == 28 { return 0 }
|
|
return (sym - 4) / 4
|
|
}
|
|
|
|
function z_dist_base(sym: int) -> int {
|
|
if sym < 4 { return 1 + sym }
|
|
let extra = (sym - 2) / 2
|
|
let group = (1 << extra)
|
|
return 1 + (group << 1) + (sym - 2 - extra * 2) * group
|
|
}
|
|
|
|
function z_dist_extra(sym: int) -> int {
|
|
if sym < 4 { return 0 }
|
|
return (sym - 2) / 2
|
|
}
|
|
|
|
# The RFC tables above are pure functions of the symbol; compute them once rather
|
|
# than dividing per match.
|
|
var z_lbase: words = null
|
|
var z_lext: words = null
|
|
var z_dbase: words = null
|
|
var z_dext: words = null
|
|
|
|
function z_tables_once() -> void {
|
|
if z_lbase != null { return }
|
|
z_lbase = words(29)
|
|
z_lext = words(29)
|
|
for s in 0 .. 29 {
|
|
z_lbase[s] = z_len_base(s)
|
|
z_lext[s] = z_len_extra(s)
|
|
}
|
|
z_dbase = words(30)
|
|
z_dext = words(30)
|
|
for s in 0 .. 30 {
|
|
z_dbase[s] = z_dist_base(s)
|
|
z_dext[s] = z_dist_extra(s)
|
|
}
|
|
}
|
|
|
|
# ---- block decoders -------------------------------------------------------
|
|
# `out` is the destination window; returns the new write position, or -1.
|
|
function z_stored(out: pointer, at: int, cap: int) -> int {
|
|
z_bitbuf = 0
|
|
z_bitcnt = 0 # stored blocks are byte-aligned
|
|
if z_pos + 4 > z_len { return -1 }
|
|
let n = z_src[z_pos] + (z_src[z_pos + 1] << 8)
|
|
z_pos += 4 # LEN then its one's complement
|
|
var w = at
|
|
for i in 0 .. n {
|
|
if z_pos >= z_len { return -1 }
|
|
if w >= cap { return -1 }
|
|
out[w] = z_src[z_pos]
|
|
w += 1
|
|
z_pos += 1
|
|
}
|
|
return w
|
|
}
|
|
|
|
function z_codes(out: pointer, at: int, cap: int, lit: pointer, dist: pointer) -> int {
|
|
var w = at
|
|
var sym = z_decode(lit)
|
|
while sym != 256 {
|
|
if z_err != 0 { return -1 }
|
|
if sym < 0 { return -1 }
|
|
if sym < 256 {
|
|
if w >= cap { return -1 }
|
|
out[w] = sym
|
|
w += 1
|
|
}
|
|
if sym > 256 {
|
|
let s = sym - 257
|
|
if s >= 29 { return -1 }
|
|
let length = z_lbase[s] + z_bits(z_lext[s])
|
|
let d = z_decode(dist)
|
|
if d < 0 { return -1 }
|
|
if d >= 30 { return -1 }
|
|
let distance = z_dbase[d] + z_bits(z_dext[d])
|
|
if distance > w { return -1 }
|
|
if w + length > cap { return -1 } # bounds once, not per byte
|
|
var sp = w - distance
|
|
var k = 0
|
|
while k < length {
|
|
out[w] = out[sp]
|
|
w += 1
|
|
sp += 1
|
|
k += 1
|
|
}
|
|
}
|
|
sym = z_decode(lit)
|
|
}
|
|
return w
|
|
}
|
|
|
|
function z_fixed_tables(lit: pointer, dist: pointer) -> void {
|
|
let lengths: words = words(288)
|
|
for i in 0 .. 144 { lengths[i] = 8 }
|
|
for i in 144 .. 256 { lengths[i] = 9 }
|
|
for i in 256 .. 280 { lengths[i] = 7 }
|
|
for i in 280 .. 288 { lengths[i] = 8 }
|
|
z_table_build(lit, lengths, 288)
|
|
for i in 0 .. 30 { lengths[i] = 5 }
|
|
z_table_build(dist, lengths, 30)
|
|
free(lengths)
|
|
}
|
|
|
|
function z_dynamic_tables(lit: pointer, dist: pointer) -> int {
|
|
let nlen = z_bits(5) + 257
|
|
let ndist = z_bits(5) + 1
|
|
let ncode = z_bits(4) + 4
|
|
if nlen > 286 { return 0 }
|
|
if ndist > 30 { return 0 }
|
|
|
|
let lengths: words = words(320)
|
|
for i in 0 .. 19 { lengths[i] = 0 }
|
|
# the code-length alphabet is transmitted in this fixed permutation
|
|
# 16,17,18,0,8,7,9,6,10,5,11,4,12,3,13,2,14,1,15 — biased by '0' so it is one literal
|
|
let order = "@AB08796:5;4<3=2>1?"
|
|
for i in 0 .. ncode {
|
|
lengths[order[i] - 48] = z_bits(3)
|
|
}
|
|
let clen = z_table_new(19)
|
|
z_table_build(clen, lengths, 19)
|
|
|
|
var n = 0
|
|
while n < nlen + ndist {
|
|
let sym = z_decode(clen)
|
|
if sym < 0 { return 0 }
|
|
if sym < 16 {
|
|
lengths[n] = sym
|
|
n += 1
|
|
}
|
|
if sym >= 16 {
|
|
var prev = 0
|
|
var rep = 0
|
|
if sym == 16 {
|
|
if n == 0 { return 0 }
|
|
prev = lengths[n - 1]
|
|
rep = 3 + z_bits(2)
|
|
}
|
|
if sym == 17 { rep = 3 + z_bits(3) }
|
|
if sym == 18 { rep = 11 + z_bits(7) }
|
|
for k in 0 .. rep {
|
|
if n < 320 {
|
|
lengths[n] = prev
|
|
n += 1
|
|
}
|
|
}
|
|
}
|
|
}
|
|
z_table_build(lit, lengths, nlen)
|
|
# the distance lengths follow the literal ones in the same buffer
|
|
let dl: words = words(32)
|
|
for i in 0 .. ndist { dl[i] = lengths[nlen + i] }
|
|
z_table_build(dist, dl, ndist)
|
|
free(dl)
|
|
free(lengths)
|
|
free(clen)
|
|
return 1
|
|
}
|
|
|
|
# Inflate a raw DEFLATE stream. Returns bytes written, or -1.
|
|
function z_inflate(src: pointer, len: int, out: pointer, cap: int) -> int {
|
|
z_start(src, len)
|
|
let lit = z_table_new(288)
|
|
let dist = z_table_new(30)
|
|
var w = 0
|
|
var final = 0
|
|
while final == 0 {
|
|
final = z_bits(1)
|
|
let btype = z_bits(2)
|
|
if z_err != 0 { return -1 }
|
|
if btype == 0 { w = z_stored(out, w, cap) }
|
|
if btype == 1 {
|
|
z_fixed_tables(lit, dist)
|
|
w = z_codes(out, w, cap, lit, dist)
|
|
}
|
|
if btype == 2 {
|
|
if z_dynamic_tables(lit, dist) == 0 { return -1 }
|
|
w = z_codes(out, w, cap, lit, dist)
|
|
}
|
|
if btype == 3 { return -1 }
|
|
if w < 0 { return -1 }
|
|
}
|
|
free(lit)
|
|
free(dist)
|
|
return w
|
|
}
|
|
|
|
# zlib wrapper (RFC 1950): two header bytes, then DEFLATE, then Adler-32.
|
|
function z_uncompress(src: pointer, len: int, out: pointer, cap: int) -> int {
|
|
if len < 2 { return -1 }
|
|
let cmf = src[0]
|
|
if (cmf & 15) != 8 { return -1 }
|
|
return z_inflate(offset(src, 2), len - 2, out, cap)
|
|
}
|
|
|
|
# gzip framing (RFC 1952): a 10-byte header (magic 1f 8b, CM=8, FLG, 4-byte MTIME,
|
|
# XFL, OS), optional FEXTRA/FNAME/FCOMMENT/FHCRC fields selected by FLG, then the
|
|
# same DEFLATE body zlib carries, then an 8-byte CRC32 + ISIZE trailer. We skip
|
|
# the header + optional fields, inflate the body, and ignore the trailer — the
|
|
# CRC is a redundancy check, not needed to decode (PNG likewise ignores ancillary
|
|
# CRCs). Returns bytes written, or -1.
|
|
function z_gunzip(src: pointer, len: int, out: pointer, cap: int) -> int {
|
|
if len < 18 { return -1 } # 10 header + 8 trailer minimum
|
|
if src[0] != 31 { return -1 } # 0x1f
|
|
if src[1] != 139 { return -1 } # 0x8b
|
|
if src[2] != 8 { return -1 } # CM must be DEFLATE
|
|
let flg = src[3]
|
|
var pos = 10 # past the fixed header
|
|
if (flg & 4) != 0 { # FEXTRA: 2-byte length then that many bytes
|
|
if pos + 2 > len { return -1 }
|
|
let xlen = src[pos] + (src[pos + 1] << 8)
|
|
pos = pos + 2 + xlen
|
|
}
|
|
if (flg & 8) != 0 { # FNAME: NUL-terminated
|
|
while pos < len and src[pos] != 0 { pos += 1 }
|
|
pos += 1
|
|
}
|
|
if (flg & 16) != 0 { # FCOMMENT: NUL-terminated
|
|
while pos < len and src[pos] != 0 { pos += 1 }
|
|
pos += 1
|
|
}
|
|
if (flg & 2) != 0 { pos += 2 } # FHCRC: 2-byte header CRC
|
|
if pos + 8 > len { return -1 }
|
|
return z_inflate(offset(src, pos), len - pos - 8, out, cap)
|
|
}
|