# ============================================================================ # inflate.ludic — DEFLATE decompression (RFC 1951), written in Ludic. # # PNG stores its pixels zlib-compressed, so decoding one means implementing # inflate. The C runtime linked zlib for this. We don't: zlib is not present by # default on every target Ludic compiles for (Windows especially), and shipping # a dependency to read a sprite is a poor trade when the algorithm is this # small. So it lives here, in the language. # # The decoder is the canonical-Huffman formulation from Mark Adler's `puff`: a # symbol table plus per-length counts. Short codes (<= Z_FAST bits, which is the # overwhelming majority) resolve in a single lookup out of a 512-entry table built # with the symbol table; longer ones fall back to puff's walk, one bit at a time. # The walk alone was fine for sprites, but a PBR scene inflates hundreds of # megabytes of texture at load, and there the table is worth its 2 KB. # ============================================================================ # ---- the context ------------------------------------------------------------ # Everything one inflate works in: the bit reader, the RFC's length and distance tables, and every # table and scratch list a block builds, made once with the context (z_inflate_new) and rebuilt in # place - so an inflate allocates nothing, and each thread that inflates holds a context of its own. # RtInflateState keeps one for the callers that inflate on the program's own thread. export property ZInflate { src: pointer = null len: int = 0 pos: int = 0 bitbuf: int = 0 bitcnt: int = 0 err: int = 0 lbase: words = null lext: words = null dbase: words = null dext: words = null lit: words = null dist: words = null clen: words = null lengths: words = null dl: words = null offs: words = null firstc: words = null } export state RtInflateState { z_ctx: ZInflate = z_inflate_new() } @alloc_ok("once per context: every table an inflate works in, rebuilt in place ever after") export function z_inflate_new() -> ZInflate { let z = new ZInflate z.lbase = words(29) z.lext = words(29) for s in 0 .. 29 { z.lbase[s] = z_len_base(s) z.lext[s] = z_len_extra(s) } z.dbase = words(30) z.dext = words(30) for s in 0 .. 30 { z.dbase[s] = z_dist_base(s) z.dext[s] = z_dist_extra(s) } z.lit = z_table_new(288) z.dist = z_table_new(30) z.clen = z_table_new(19) z.lengths = words(320) z.dl = words(32) z.offs = words(16) z.firstc = words(17) return z } function z_start(z: ZInflate, src: pointer, len: int) -> void { z.src = src z.len = len z.pos = 0 z.bitbuf = 0 z.bitcnt = 0 z.err = 0 } # Fill the bit buffer to at least `n` bits without consuming any (n <= 16, so the # buffer never shifts a byte past bit 15 and cannot reach the sign bit). function z_need(z: ZInflate, n: int) -> void { while z.bitcnt < n { if z.pos >= z.len { return } z.bitbuf = (z.bitbuf | (z.src[z.pos] << z.bitcnt)) z.pos += 1 z.bitcnt += 8 } } function z_bits(z: ZInflate, need: int) -> int { var val = z.bitbuf while z.bitcnt < need { if z.pos >= z.len { z.err = 1 return 0 } val = (val | (z.src[z.pos] << z.bitcnt)) z.pos += 1 z.bitcnt += 8 } z.bitbuf = (val >> need) z.bitcnt -= need return (val & ((1 << need) - 1)) } # ---- Huffman tables ------------------------------------------------------- # One buffer per table: 16 length-counts, then a Z_FASTSZ-entry lookup keyed by # the next Z_FAST bits of the stream, then the symbols in canonical order. # A lookup entry is (length << 16) | symbol, or 0 when no code that short matches. # Z_FAST = 10 measured fastest over a 493 MB corpus (9 and 11 are both ~8% slower: # 9 misses the table more often, 11 spends more clearing it per dynamic block). const Z_FAST: int = 10 const Z_FASTSZ: int = 1024 # 1 << Z_FAST const Z_SYMS: int = 1040 # 16 + Z_FASTSZ: where the symbols start function z_table_new(nsym: int) -> words { return words((Z_SYMS + nsym)) } # lengths[i] = code length of symbol i (0 = symbol unused) function z_table_build(z: ZInflate, table: words, lengths: words, n: int) -> void { for i in 0 .. 16 { table[i] = 0 } for s in 0 .. n { let l = lengths[s] table[l] += 1 } table[0] = 0 # length 0 means "not present" # offset of each length's first symbol let offs = z.offs offs[1] = 0 for l in 1 .. 15 { offs[l + 1] = offs[l] + table[l] } for s in 0 .. n { let l = lengths[s] if l != 0 { table[Z_SYMS + offs[l]] = s offs[l] += 1 } } # ---- the fast lookup ---- for i in 0 .. Z_FASTSZ { table[16 + i] = 0 } # first canonical code of each length let firstc = z.firstc var code = 0 for l in 1 .. 16 { code = ((code + table[l - 1]) << 1) firstc[l] = code } var idx = 0 for l in 1 .. 16 { let cnt = table[l] var k = 0 while k < cnt { let sym = table[Z_SYMS + idx] if l <= Z_FAST { # DEFLATE reads a code most-significant-bit first out of a stream packed # least-significant-bit first, so the table is keyed by the reversed code let c = firstc[l] + k var rev = 0 var b = 0 while b < l { rev = ((rev << 1) | ((c >> b) & 1)) b += 1 } let entry = ((l << 16) | sym) var j = rev while j < Z_FASTSZ { table[16 + j] = entry j += (1 << l) } } idx += 1 k += 1 } } } function z_decode(z: ZInflate, table: words) -> int { z_need(z, Z_FAST) if z.bitcnt >= Z_FAST { let e = table[16 + (z.bitbuf & (Z_FASTSZ - 1))] if e != 0 { let l = (e >> 16) z.bitbuf = (z.bitbuf >> l) z.bitcnt -= l return (e & 65535) } } # a code longer than Z_FAST bits (or a stream too short to peek): walk it var code = 0 var first = 0 var index = 0 for len in 1 .. 16 { code = (code | z_bits(z, 1)) let count = table[len] if code - first < count { return table[Z_SYMS + index + (code - first)] } index += count first = ((first + count) << 1) code = (code << 1) } z.err = 1 return -1 } # ---- length / distance code tables (RFC 1951 section 3.2.5) --------------- function z_len_base(sym: int) -> int { if sym < 8 { return 3 + sym } if sym == 28 { return 258 } let extra = (sym - 4) / 4 let group = (1 << extra) return 3 + ((group - 1) << 2) + 4 + (sym - 4 - extra * 4) * group } function z_len_extra(sym: int) -> int { if sym < 8 { return 0 } if sym == 28 { return 0 } return (sym - 4) / 4 } function z_dist_base(sym: int) -> int { if sym < 4 { return 1 + sym } let extra = (sym - 2) / 2 let group = (1 << extra) return 1 + (group << 1) + (sym - 2 - extra * 2) * group } function z_dist_extra(sym: int) -> int { if sym < 4 { return 0 } return (sym - 2) / 2 } # The RFC tables above are pure functions of the symbol, computed once per context (z_inflate_new) # rather than divided per match. # ---- block decoders ------------------------------------------------------- # `out` is the destination window; returns the new write position, or -1. function z_stored(z: ZInflate, out: pointer, at: int, cap: int) -> int { z.bitbuf = 0 z.bitcnt = 0 # stored blocks are byte-aligned if z.pos + 4 > z.len { return -1 } let n = z.src[z.pos] + (z.src[z.pos + 1] << 8) z.pos += 4 # LEN then its one's complement var w = at for i in 0 .. n { if z.pos >= z.len { return -1 } if w >= cap { return -1 } out[w] = z.src[z.pos] w += 1 z.pos += 1 } return w } function z_codes(z: ZInflate, out: pointer, at: int, cap: int, lit: words, dist: words) -> int { var w = at var sym = z_decode(z, lit) while sym != 256 { if z.err != 0 { return -1 } if sym < 0 { return -1 } if sym < 256 { if w >= cap { return -1 } out[w] = sym w += 1 } if sym > 256 { let s = sym - 257 if s >= 29 { return -1 } let length = z.lbase[s] + z_bits(z, z.lext[s]) let d = z_decode(z, dist) if d < 0 { return -1 } if d >= 30 { return -1 } let distance = z.dbase[d] + z_bits(z, z.dext[d]) if distance > w { return -1 } if w + length > cap { return -1 } # bounds once, not per byte var sp = w - distance var k = 0 while k < length { out[w] = out[sp] w += 1 sp += 1 k += 1 } } sym = z_decode(z, lit) } return w } function z_fixed_tables(z: ZInflate, lit: words, dist: words) -> void { let lengths = z.lengths for i in 0 .. 144 { lengths[i] = 8 } for i in 144 .. 256 { lengths[i] = 9 } for i in 256 .. 280 { lengths[i] = 7 } for i in 280 .. 288 { lengths[i] = 8 } z_table_build(z, lit, lengths, 288) for i in 0 .. 30 { lengths[i] = 5 } z_table_build(z, dist, lengths, 30) } function z_dynamic_tables(z: ZInflate, lit: words, dist: words) -> int { let nlen = z_bits(z, 5) + 257 let ndist = z_bits(z, 5) + 1 let ncode = z_bits(z, 4) + 4 if nlen > 286 { return 0 } if ndist > 30 { return 0 } let lengths = z.lengths for i in 0 .. 19 { lengths[i] = 0 } # the code-length alphabet is transmitted in this fixed permutation # 16,17,18,0,8,7,9,6,10,5,11,4,12,3,13,2,14,1,15 — biased by '0' so it is one literal let order = "@AB08796:5;4<3=2>1?" for i in 0 .. ncode { lengths[order[i] - 48] = z_bits(z, 3) } let clen = z.clen z_table_build(z, clen, lengths, 19) var n = 0 while n < nlen + ndist { let sym = z_decode(z, clen) if sym < 0 { return 0 } if sym < 16 { lengths[n] = sym n += 1 } if sym >= 16 { var prev = 0 var rep = 0 if sym == 16 { if n == 0 { return 0 } prev = lengths[n - 1] rep = 3 + z_bits(z, 2) } if sym == 17 { rep = 3 + z_bits(z, 3) } if sym == 18 { rep = 11 + z_bits(z, 7) } for k in 0 .. rep { if n < 320 { lengths[n] = prev n += 1 } } } } z_table_build(z, lit, lengths, nlen) # the distance lengths follow the literal ones in the same buffer let dl = z.dl for i in 0 .. ndist { dl[i] = lengths[nlen + i] } z_table_build(z, dist, dl, ndist) return 1 } # Inflate a raw DEFLATE stream in context z. Returns bytes written, or -1. export function z_inflate_in(z: ZInflate, src: pointer, len: int, out: pointer, cap: int) -> int { z_start(z, src, len) let lit = z.lit let dist = z.dist var w = 0 var final = 0 while final == 0 { final = z_bits(z, 1) let btype = z_bits(z, 2) if z.err != 0 { return -1 } if btype == 0 { w = z_stored(z, out, w, cap) } if btype == 1 { z_fixed_tables(z, lit, dist) w = z_codes(z, out, w, cap, lit, dist) } if btype == 2 { if z_dynamic_tables(z, lit, dist) == 0 { return -1 } w = z_codes(z, out, w, cap, lit, dist) } if btype == 3 { return -1 } if w < 0 { return -1 } } return w } # the program thread's inflate, in the context its state holds function z_inflate(rt_inflate_st: RtInflateState, src: pointer, len: int, out: pointer, cap: int) -> int { return z_inflate_in(rt_inflate_st.z_ctx, src, len, out, cap) } function z_uncompress(rt_inflate_st: RtInflateState, src: pointer, len: int, out: pointer, cap: int) -> int { return z_uncompress_in(rt_inflate_st.z_ctx, src, len, out, cap) } function z_gunzip(rt_inflate_st: RtInflateState, src: pointer, len: int, out: pointer, cap: int) -> int { return z_gunzip_in(rt_inflate_st.z_ctx, src, len, out, cap) } # zlib wrapper (RFC 1950): two header bytes, then DEFLATE, then Adler-32. export function z_uncompress_in(z: ZInflate, src: pointer, len: int, out: pointer, cap: int) -> int { if len < 2 { return -1 } let cmf = src[0] if (cmf & 15) != 8 { return -1 } return z_inflate_in(z, offset(src, 2), len - 2, out, cap) } # gzip framing (RFC 1952): a 10-byte header (magic 1f 8b, CM=8, FLG, 4-byte MTIME, # XFL, OS), optional FEXTRA/FNAME/FCOMMENT/FHCRC fields selected by FLG, then the # same DEFLATE body zlib carries, then an 8-byte CRC32 + ISIZE trailer. We skip # the header + optional fields, inflate the body, and ignore the trailer — the # CRC is a redundancy check, not needed to decode (PNG likewise ignores ancillary # CRCs). Returns bytes written, or -1. export function z_gunzip_in(z: ZInflate, src: pointer, len: int, out: pointer, cap: int) -> int { if len < 18 { return -1 } # 10 header + 8 trailer minimum if src[0] != 31 { return -1 } # 0x1f if src[1] != 139 { return -1 } # 0x8b if src[2] != 8 { return -1 } # CM must be DEFLATE let flg = src[3] var pos = 10 # past the fixed header if (flg & 4) != 0 { # FEXTRA: 2-byte length then that many bytes if pos + 2 > len { return -1 } let xlen = src[pos] + (src[pos + 1] << 8) pos = pos + 2 + xlen } if (flg & 8) != 0 { # FNAME: NUL-terminated while pos < len and src[pos] != 0 { pos += 1 } pos += 1 } if (flg & 16) != 0 { # FCOMMENT: NUL-terminated while pos < len and src[pos] != 0 { pos += 1 } pos += 1 } if (flg & 2) != 0 { pos += 2 } # FHCRC: 2-byte header CRC if pos + 8 > len { return -1 } return z_inflate_in(z, offset(src, pos), len - pos - 8, out, cap) }