ZInflate (z_inflate_new) holds the bit reader, the RFC tables and every table and scratch list a block builds, rebuilt in place: an inflate allocates nothing, and a worker thread inflates in a context of its own (made on the program's thread). z_inflate_in / z_uncompress_in / z_gunzip_in take it; z_inflate / z_uncompress / z_gunzip and every caller (image, atlas, tiled, render3d's png_decode) are unchanged and work in RtInflateState's one context. The old per-call lit/dist tables were also leaked on an error return; there are none now. Compiles: Maroon Lake headless, examples/library/tiled_p2 and tiled_p6. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
430 lines
13 KiB
Text
430 lines
13 KiB
Text
# ============================================================================
|
|
# inflate.ludic — DEFLATE decompression (RFC 1951), written in Ludic.
|
|
#
|
|
# PNG stores its pixels zlib-compressed, so decoding one means implementing
|
|
# inflate. The C runtime linked zlib for this. We don't: zlib is not present by
|
|
# default on every target Ludic compiles for (Windows especially), and shipping
|
|
# a dependency to read a sprite is a poor trade when the algorithm is this
|
|
# small. So it lives here, in the language.
|
|
#
|
|
# The decoder is the canonical-Huffman formulation from Mark Adler's `puff`: a
|
|
# symbol table plus per-length counts. Short codes (<= Z_FAST bits, which is the
|
|
# overwhelming majority) resolve in a single lookup out of a 512-entry table built
|
|
# with the symbol table; longer ones fall back to puff's walk, one bit at a time.
|
|
# The walk alone was fine for sprites, but a PBR scene inflates hundreds of
|
|
# megabytes of texture at load, and there the table is worth its 2 KB.
|
|
# ============================================================================
|
|
|
|
# ---- the context ------------------------------------------------------------
|
|
# Everything one inflate works in: the bit reader, the RFC's length and distance tables, and every
|
|
# table and scratch list a block builds, made once with the context (z_inflate_new) and rebuilt in
|
|
# place - so an inflate allocates nothing, and each thread that inflates holds a context of its own.
|
|
# RtInflateState keeps one for the callers that inflate on the program's own thread.
|
|
export property ZInflate {
|
|
src: pointer = null
|
|
len: int = 0
|
|
pos: int = 0
|
|
bitbuf: int = 0
|
|
bitcnt: int = 0
|
|
err: int = 0
|
|
lbase: words = null
|
|
lext: words = null
|
|
dbase: words = null
|
|
dext: words = null
|
|
lit: words = null
|
|
dist: words = null
|
|
clen: words = null
|
|
lengths: words = null
|
|
dl: words = null
|
|
offs: words = null
|
|
firstc: words = null
|
|
}
|
|
|
|
export state RtInflateState {
|
|
z_ctx: ZInflate = z_inflate_new()
|
|
}
|
|
|
|
@alloc_ok("once per context: every table an inflate works in, rebuilt in place ever after")
|
|
export function z_inflate_new() -> ZInflate {
|
|
let z = new ZInflate
|
|
z.lbase = words(29)
|
|
z.lext = words(29)
|
|
for s in 0 .. 29 {
|
|
z.lbase[s] = z_len_base(s)
|
|
z.lext[s] = z_len_extra(s)
|
|
}
|
|
z.dbase = words(30)
|
|
z.dext = words(30)
|
|
for s in 0 .. 30 {
|
|
z.dbase[s] = z_dist_base(s)
|
|
z.dext[s] = z_dist_extra(s)
|
|
}
|
|
z.lit = z_table_new(288)
|
|
z.dist = z_table_new(30)
|
|
z.clen = z_table_new(19)
|
|
z.lengths = words(320)
|
|
z.dl = words(32)
|
|
z.offs = words(16)
|
|
z.firstc = words(17)
|
|
return z
|
|
}
|
|
|
|
function z_start(z: ZInflate, src: pointer, len: int) -> void {
|
|
z.src = src
|
|
z.len = len
|
|
z.pos = 0
|
|
z.bitbuf = 0
|
|
z.bitcnt = 0
|
|
z.err = 0
|
|
}
|
|
|
|
# Fill the bit buffer to at least `n` bits without consuming any (n <= 16, so the
|
|
# buffer never shifts a byte past bit 15 and cannot reach the sign bit).
|
|
function z_need(z: ZInflate, n: int) -> void {
|
|
while z.bitcnt < n {
|
|
if z.pos >= z.len { return }
|
|
z.bitbuf = (z.bitbuf | (z.src[z.pos] << z.bitcnt))
|
|
z.pos += 1
|
|
z.bitcnt += 8
|
|
}
|
|
}
|
|
|
|
function z_bits(z: ZInflate, need: int) -> int {
|
|
var val = z.bitbuf
|
|
while z.bitcnt < need {
|
|
if z.pos >= z.len {
|
|
z.err = 1
|
|
return 0
|
|
}
|
|
val = (val | (z.src[z.pos] << z.bitcnt))
|
|
z.pos += 1
|
|
z.bitcnt += 8
|
|
}
|
|
z.bitbuf = (val >> need)
|
|
z.bitcnt -= need
|
|
return (val & ((1 << need) - 1))
|
|
}
|
|
|
|
# ---- Huffman tables -------------------------------------------------------
|
|
# One buffer per table: 16 length-counts, then a Z_FASTSZ-entry lookup keyed by
|
|
# the next Z_FAST bits of the stream, then the symbols in canonical order.
|
|
# A lookup entry is (length << 16) | symbol, or 0 when no code that short matches.
|
|
# Z_FAST = 10 measured fastest over a 493 MB corpus (9 and 11 are both ~8% slower:
|
|
# 9 misses the table more often, 11 spends more clearing it per dynamic block).
|
|
const Z_FAST: int = 10
|
|
const Z_FASTSZ: int = 1024 # 1 << Z_FAST
|
|
const Z_SYMS: int = 1040 # 16 + Z_FASTSZ: where the symbols start
|
|
|
|
function z_table_new(nsym: int) -> words {
|
|
return words((Z_SYMS + nsym))
|
|
}
|
|
|
|
# lengths[i] = code length of symbol i (0 = symbol unused)
|
|
function z_table_build(z: ZInflate, table: words, lengths: words, n: int) -> void {
|
|
for i in 0 .. 16 {
|
|
table[i] = 0
|
|
}
|
|
for s in 0 .. n {
|
|
let l = lengths[s]
|
|
table[l] += 1
|
|
}
|
|
table[0] = 0 # length 0 means "not present"
|
|
# offset of each length's first symbol
|
|
let offs = z.offs
|
|
offs[1] = 0
|
|
for l in 1 .. 15 {
|
|
offs[l + 1] = offs[l] + table[l]
|
|
}
|
|
for s in 0 .. n {
|
|
let l = lengths[s]
|
|
if l != 0 {
|
|
table[Z_SYMS + offs[l]] = s
|
|
offs[l] += 1
|
|
}
|
|
}
|
|
|
|
# ---- the fast lookup ----
|
|
for i in 0 .. Z_FASTSZ {
|
|
table[16 + i] = 0
|
|
}
|
|
# first canonical code of each length
|
|
let firstc = z.firstc
|
|
var code = 0
|
|
for l in 1 .. 16 {
|
|
code = ((code + table[l - 1]) << 1)
|
|
firstc[l] = code
|
|
}
|
|
var idx = 0
|
|
for l in 1 .. 16 {
|
|
let cnt = table[l]
|
|
var k = 0
|
|
while k < cnt {
|
|
let sym = table[Z_SYMS + idx]
|
|
if l <= Z_FAST {
|
|
# DEFLATE reads a code most-significant-bit first out of a stream packed
|
|
# least-significant-bit first, so the table is keyed by the reversed code
|
|
let c = firstc[l] + k
|
|
var rev = 0
|
|
var b = 0
|
|
while b < l {
|
|
rev = ((rev << 1) | ((c >> b) & 1))
|
|
b += 1
|
|
}
|
|
let entry = ((l << 16) | sym)
|
|
var j = rev
|
|
while j < Z_FASTSZ {
|
|
table[16 + j] = entry
|
|
j += (1 << l)
|
|
}
|
|
}
|
|
idx += 1
|
|
k += 1
|
|
}
|
|
}
|
|
}
|
|
|
|
function z_decode(z: ZInflate, table: words) -> int {
|
|
z_need(z, Z_FAST)
|
|
if z.bitcnt >= Z_FAST {
|
|
let e = table[16 + (z.bitbuf & (Z_FASTSZ - 1))]
|
|
if e != 0 {
|
|
let l = (e >> 16)
|
|
z.bitbuf = (z.bitbuf >> l)
|
|
z.bitcnt -= l
|
|
return (e & 65535)
|
|
}
|
|
}
|
|
# a code longer than Z_FAST bits (or a stream too short to peek): walk it
|
|
var code = 0
|
|
var first = 0
|
|
var index = 0
|
|
for len in 1 .. 16 {
|
|
code = (code | z_bits(z, 1))
|
|
let count = table[len]
|
|
if code - first < count {
|
|
return table[Z_SYMS + index + (code - first)]
|
|
}
|
|
index += count
|
|
first = ((first + count) << 1)
|
|
code = (code << 1)
|
|
}
|
|
z.err = 1
|
|
return -1
|
|
}
|
|
|
|
# ---- length / distance code tables (RFC 1951 section 3.2.5) ---------------
|
|
function z_len_base(sym: int) -> int {
|
|
if sym < 8 { return 3 + sym }
|
|
if sym == 28 { return 258 }
|
|
let extra = (sym - 4) / 4
|
|
let group = (1 << extra)
|
|
return 3 + ((group - 1) << 2) + 4 + (sym - 4 - extra * 4) * group
|
|
}
|
|
|
|
function z_len_extra(sym: int) -> int {
|
|
if sym < 8 { return 0 }
|
|
if sym == 28 { return 0 }
|
|
return (sym - 4) / 4
|
|
}
|
|
|
|
function z_dist_base(sym: int) -> int {
|
|
if sym < 4 { return 1 + sym }
|
|
let extra = (sym - 2) / 2
|
|
let group = (1 << extra)
|
|
return 1 + (group << 1) + (sym - 2 - extra * 2) * group
|
|
}
|
|
|
|
function z_dist_extra(sym: int) -> int {
|
|
if sym < 4 { return 0 }
|
|
return (sym - 2) / 2
|
|
}
|
|
|
|
# The RFC tables above are pure functions of the symbol, computed once per context (z_inflate_new)
|
|
# rather than divided per match.
|
|
|
|
# ---- block decoders -------------------------------------------------------
|
|
# `out` is the destination window; returns the new write position, or -1.
|
|
function z_stored(z: ZInflate, out: pointer, at: int, cap: int) -> int {
|
|
z.bitbuf = 0
|
|
z.bitcnt = 0 # stored blocks are byte-aligned
|
|
if z.pos + 4 > z.len { return -1 }
|
|
let n = z.src[z.pos] + (z.src[z.pos + 1] << 8)
|
|
z.pos += 4 # LEN then its one's complement
|
|
var w = at
|
|
for i in 0 .. n {
|
|
if z.pos >= z.len { return -1 }
|
|
if w >= cap { return -1 }
|
|
out[w] = z.src[z.pos]
|
|
w += 1
|
|
z.pos += 1
|
|
}
|
|
return w
|
|
}
|
|
|
|
function z_codes(z: ZInflate, out: pointer, at: int, cap: int, lit: words, dist: words) -> int {
|
|
var w = at
|
|
var sym = z_decode(z, lit)
|
|
while sym != 256 {
|
|
if z.err != 0 { return -1 }
|
|
if sym < 0 { return -1 }
|
|
if sym < 256 {
|
|
if w >= cap { return -1 }
|
|
out[w] = sym
|
|
w += 1
|
|
}
|
|
if sym > 256 {
|
|
let s = sym - 257
|
|
if s >= 29 { return -1 }
|
|
let length = z.lbase[s] + z_bits(z, z.lext[s])
|
|
let d = z_decode(z, dist)
|
|
if d < 0 { return -1 }
|
|
if d >= 30 { return -1 }
|
|
let distance = z.dbase[d] + z_bits(z, z.dext[d])
|
|
if distance > w { return -1 }
|
|
if w + length > cap { return -1 } # bounds once, not per byte
|
|
var sp = w - distance
|
|
var k = 0
|
|
while k < length {
|
|
out[w] = out[sp]
|
|
w += 1
|
|
sp += 1
|
|
k += 1
|
|
}
|
|
}
|
|
sym = z_decode(z, lit)
|
|
}
|
|
return w
|
|
}
|
|
|
|
function z_fixed_tables(z: ZInflate, lit: words, dist: words) -> void {
|
|
let lengths = z.lengths
|
|
for i in 0 .. 144 { lengths[i] = 8 }
|
|
for i in 144 .. 256 { lengths[i] = 9 }
|
|
for i in 256 .. 280 { lengths[i] = 7 }
|
|
for i in 280 .. 288 { lengths[i] = 8 }
|
|
z_table_build(z, lit, lengths, 288)
|
|
for i in 0 .. 30 { lengths[i] = 5 }
|
|
z_table_build(z, dist, lengths, 30)
|
|
}
|
|
|
|
function z_dynamic_tables(z: ZInflate, lit: words, dist: words) -> int {
|
|
let nlen = z_bits(z, 5) + 257
|
|
let ndist = z_bits(z, 5) + 1
|
|
let ncode = z_bits(z, 4) + 4
|
|
if nlen > 286 { return 0 }
|
|
if ndist > 30 { return 0 }
|
|
|
|
let lengths = z.lengths
|
|
for i in 0 .. 19 { lengths[i] = 0 }
|
|
# the code-length alphabet is transmitted in this fixed permutation
|
|
# 16,17,18,0,8,7,9,6,10,5,11,4,12,3,13,2,14,1,15 — biased by '0' so it is one literal
|
|
let order = "@AB08796:5;4<3=2>1?"
|
|
for i in 0 .. ncode {
|
|
lengths[order[i] - 48] = z_bits(z, 3)
|
|
}
|
|
let clen = z.clen
|
|
z_table_build(z, clen, lengths, 19)
|
|
|
|
var n = 0
|
|
while n < nlen + ndist {
|
|
let sym = z_decode(z, clen)
|
|
if sym < 0 { return 0 }
|
|
if sym < 16 {
|
|
lengths[n] = sym
|
|
n += 1
|
|
}
|
|
if sym >= 16 {
|
|
var prev = 0
|
|
var rep = 0
|
|
if sym == 16 {
|
|
if n == 0 { return 0 }
|
|
prev = lengths[n - 1]
|
|
rep = 3 + z_bits(z, 2)
|
|
}
|
|
if sym == 17 { rep = 3 + z_bits(z, 3) }
|
|
if sym == 18 { rep = 11 + z_bits(z, 7) }
|
|
for k in 0 .. rep {
|
|
if n < 320 {
|
|
lengths[n] = prev
|
|
n += 1
|
|
}
|
|
}
|
|
}
|
|
}
|
|
z_table_build(z, lit, lengths, nlen)
|
|
# the distance lengths follow the literal ones in the same buffer
|
|
let dl = z.dl
|
|
for i in 0 .. ndist { dl[i] = lengths[nlen + i] }
|
|
z_table_build(z, dist, dl, ndist)
|
|
return 1
|
|
}
|
|
|
|
# Inflate a raw DEFLATE stream in context z. Returns bytes written, or -1.
|
|
export function z_inflate_in(z: ZInflate, src: pointer, len: int, out: pointer, cap: int) -> int {
|
|
z_start(z, src, len)
|
|
let lit = z.lit
|
|
let dist = z.dist
|
|
var w = 0
|
|
var final = 0
|
|
while final == 0 {
|
|
final = z_bits(z, 1)
|
|
let btype = z_bits(z, 2)
|
|
if z.err != 0 { return -1 }
|
|
if btype == 0 { w = z_stored(z, out, w, cap) }
|
|
if btype == 1 {
|
|
z_fixed_tables(z, lit, dist)
|
|
w = z_codes(z, out, w, cap, lit, dist)
|
|
}
|
|
if btype == 2 {
|
|
if z_dynamic_tables(z, lit, dist) == 0 { return -1 }
|
|
w = z_codes(z, out, w, cap, lit, dist)
|
|
}
|
|
if btype == 3 { return -1 }
|
|
if w < 0 { return -1 }
|
|
}
|
|
return w
|
|
}
|
|
|
|
# the program thread's inflate, in the context its state holds
|
|
function z_inflate(rt_inflate_st: RtInflateState, src: pointer, len: int, out: pointer, cap: int) -> int { return z_inflate_in(rt_inflate_st.z_ctx, src, len, out, cap) }
|
|
function z_uncompress(rt_inflate_st: RtInflateState, src: pointer, len: int, out: pointer, cap: int) -> int { return z_uncompress_in(rt_inflate_st.z_ctx, src, len, out, cap) }
|
|
function z_gunzip(rt_inflate_st: RtInflateState, src: pointer, len: int, out: pointer, cap: int) -> int { return z_gunzip_in(rt_inflate_st.z_ctx, src, len, out, cap) }
|
|
|
|
# zlib wrapper (RFC 1950): two header bytes, then DEFLATE, then Adler-32.
|
|
export function z_uncompress_in(z: ZInflate, src: pointer, len: int, out: pointer, cap: int) -> int {
|
|
if len < 2 { return -1 }
|
|
let cmf = src[0]
|
|
if (cmf & 15) != 8 { return -1 }
|
|
return z_inflate_in(z, offset(src, 2), len - 2, out, cap)
|
|
}
|
|
|
|
# gzip framing (RFC 1952): a 10-byte header (magic 1f 8b, CM=8, FLG, 4-byte MTIME,
|
|
# XFL, OS), optional FEXTRA/FNAME/FCOMMENT/FHCRC fields selected by FLG, then the
|
|
# same DEFLATE body zlib carries, then an 8-byte CRC32 + ISIZE trailer. We skip
|
|
# the header + optional fields, inflate the body, and ignore the trailer — the
|
|
# CRC is a redundancy check, not needed to decode (PNG likewise ignores ancillary
|
|
# CRCs). Returns bytes written, or -1.
|
|
export function z_gunzip_in(z: ZInflate, src: pointer, len: int, out: pointer, cap: int) -> int {
|
|
if len < 18 { return -1 } # 10 header + 8 trailer minimum
|
|
if src[0] != 31 { return -1 } # 0x1f
|
|
if src[1] != 139 { return -1 } # 0x8b
|
|
if src[2] != 8 { return -1 } # CM must be DEFLATE
|
|
let flg = src[3]
|
|
var pos = 10 # past the fixed header
|
|
if (flg & 4) != 0 { # FEXTRA: 2-byte length then that many bytes
|
|
if pos + 2 > len { return -1 }
|
|
let xlen = src[pos] + (src[pos + 1] << 8)
|
|
pos = pos + 2 + xlen
|
|
}
|
|
if (flg & 8) != 0 { # FNAME: NUL-terminated
|
|
while pos < len and src[pos] != 0 { pos += 1 }
|
|
pos += 1
|
|
}
|
|
if (flg & 16) != 0 { # FCOMMENT: NUL-terminated
|
|
while pos < len and src[pos] != 0 { pos += 1 }
|
|
pos += 1
|
|
}
|
|
if (flg & 2) != 0 { pos += 2 } # FHCRC: 2-byte header CRC
|
|
if pos + 8 > len { return -1 }
|
|
return z_inflate_in(z, offset(src, pos), len - pos - 8, out, cap)
|
|
}
|