# ============================================================================ # inflate.ludic — DEFLATE decompression (RFC 1951), written in Ludic. # # PNG stores its pixels zlib-compressed, so decoding one means implementing # inflate. The C runtime linked zlib for this. We don't: zlib is not present by # default on every target Ludic compiles for (Windows especially), and shipping # a dependency to read a sprite is a poor trade when the algorithm is this # small. So it lives here, in the language. # # The decoder is the canonical-Huffman formulation from Mark Adler's `puff`: a # symbol table plus per-length counts. Short codes (<= Z_FAST bits, which is the # overwhelming majority) resolve in a single lookup out of a 512-entry table built # with the symbol table; longer ones fall back to puff's walk, one bit at a time. # The walk alone was fine for sprites, but a PBR scene inflates hundreds of # megabytes of texture at load, and there the table is worth its 2 KB. # ============================================================================ # ---- bit reader (DEFLATE packs bits least-significant-first) --------------- var z_src: pointer = null var z_len: int = 0 var z_pos: int = 0 var z_bitbuf: int = 0 var z_bitcnt: int = 0 var z_err: int = 0 function z_start(src: pointer, len: int) -> void { z_tables_once() z_src = src z_len = len z_pos = 0 z_bitbuf = 0 z_bitcnt = 0 z_err = 0 } # Fill the bit buffer to at least `n` bits without consuming any (n <= 16, so the # buffer never shifts a byte past bit 15 and cannot reach the sign bit). function z_need(n: int) -> void { while z_bitcnt < n { if z_pos >= z_len { return } z_bitbuf = (z_bitbuf | (z_src[z_pos] << z_bitcnt)) z_pos += 1 z_bitcnt += 8 } } function z_bits(need: int) -> int { var val = z_bitbuf while z_bitcnt < need { if z_pos >= z_len { z_err = 1 return 0 } val = (val | (z_src[z_pos] << z_bitcnt)) z_pos += 1 z_bitcnt += 8 } z_bitbuf = (val >> need) z_bitcnt -= need return (val & ((1 << need) - 1)) } # ---- Huffman tables ------------------------------------------------------- # One buffer per table: 16 length-counts, then a Z_FASTSZ-entry lookup keyed by # the next Z_FAST bits of the stream, then the symbols in canonical order. # A lookup entry is (length << 16) | symbol, or 0 when no code that short matches. # Z_FAST = 10 measured fastest over a 493 MB corpus (9 and 11 are both ~8% slower: # 9 misses the table more often, 11 spends more clearing it per dynamic block). const Z_FAST: int = 10 const Z_FASTSZ: int = 1024 # 1 << Z_FAST const Z_SYMS: int = 1040 # 16 + Z_FASTSZ: where the symbols start function z_table_new(nsym: int) -> words { return words((Z_SYMS + nsym)) } # lengths[i] = code length of symbol i (0 = symbol unused) function z_table_build(table: words, lengths: words, n: int) -> void { for i in 0 .. 16 { table[i] = 0 } for s in 0 .. n { let l = lengths[s] table[l] += 1 } table[0] = 0 # length 0 means "not present" # offset of each length's first symbol let offs: words = words(16) offs[1] = 0 for l in 1 .. 15 { offs[l + 1] = offs[l] + table[l] } for s in 0 .. n { let l = lengths[s] if l != 0 { table[Z_SYMS + offs[l]] = s offs[l] += 1 } } free(offs) # ---- the fast lookup ---- for i in 0 .. Z_FASTSZ { table[16 + i] = 0 } # first canonical code of each length let firstc: words = words(17) var code = 0 for l in 1 .. 16 { code = ((code + table[l - 1]) << 1) firstc[l] = code } var idx = 0 for l in 1 .. 16 { let cnt = table[l] var k = 0 while k < cnt { let sym = table[Z_SYMS + idx] if l <= Z_FAST { # DEFLATE reads a code most-significant-bit first out of a stream packed # least-significant-bit first, so the table is keyed by the reversed code let c = firstc[l] + k var rev = 0 var b = 0 while b < l { rev = ((rev << 1) | ((c >> b) & 1)) b += 1 } let entry = ((l << 16) | sym) var j = rev while j < Z_FASTSZ { table[16 + j] = entry j += (1 << l) } } idx += 1 k += 1 } } free(firstc) } function z_decode(table: words) -> int { z_need(Z_FAST) if z_bitcnt >= Z_FAST { let e = table[16 + (z_bitbuf & (Z_FASTSZ - 1))] if e != 0 { let l = (e >> 16) z_bitbuf = (z_bitbuf >> l) z_bitcnt -= l return (e & 65535) } } # a code longer than Z_FAST bits (or a stream too short to peek): walk it var code = 0 var first = 0 var index = 0 for len in 1 .. 16 { code = (code | z_bits(1)) let count = table[len] if code - first < count { return table[Z_SYMS + index + (code - first)] } index += count first = ((first + count) << 1) code = (code << 1) } z_err = 1 return -1 } # ---- length / distance code tables (RFC 1951 section 3.2.5) --------------- function z_len_base(sym: int) -> int { if sym < 8 { return 3 + sym } if sym == 28 { return 258 } let extra = (sym - 4) / 4 let group = (1 << extra) return 3 + ((group - 1) << 2) + 4 + (sym - 4 - extra * 4) * group } function z_len_extra(sym: int) -> int { if sym < 8 { return 0 } if sym == 28 { return 0 } return (sym - 4) / 4 } function z_dist_base(sym: int) -> int { if sym < 4 { return 1 + sym } let extra = (sym - 2) / 2 let group = (1 << extra) return 1 + (group << 1) + (sym - 2 - extra * 2) * group } function z_dist_extra(sym: int) -> int { if sym < 4 { return 0 } return (sym - 2) / 2 } # The RFC tables above are pure functions of the symbol; compute them once rather # than dividing per match. var z_lbase: words = null var z_lext: words = null var z_dbase: words = null var z_dext: words = null function z_tables_once() -> void { if z_lbase != null { return } z_lbase = words(29) z_lext = words(29) for s in 0 .. 29 { z_lbase[s] = z_len_base(s) z_lext[s] = z_len_extra(s) } z_dbase = words(30) z_dext = words(30) for s in 0 .. 30 { z_dbase[s] = z_dist_base(s) z_dext[s] = z_dist_extra(s) } } # ---- block decoders ------------------------------------------------------- # `out` is the destination window; returns the new write position, or -1. function z_stored(out: pointer, at: int, cap: int) -> int { z_bitbuf = 0 z_bitcnt = 0 # stored blocks are byte-aligned if z_pos + 4 > z_len { return -1 } let n = z_src[z_pos] + (z_src[z_pos + 1] << 8) z_pos += 4 # LEN then its one's complement var w = at for i in 0 .. n { if z_pos >= z_len { return -1 } if w >= cap { return -1 } out[w] = z_src[z_pos] w += 1 z_pos += 1 } return w } function z_codes(out: pointer, at: int, cap: int, lit: words, dist: words) -> int { var w = at var sym = z_decode(lit) while sym != 256 { if z_err != 0 { return -1 } if sym < 0 { return -1 } if sym < 256 { if w >= cap { return -1 } out[w] = sym w += 1 } if sym > 256 { let s = sym - 257 if s >= 29 { return -1 } let length = z_lbase[s] + z_bits(z_lext[s]) let d = z_decode(dist) if d < 0 { return -1 } if d >= 30 { return -1 } let distance = z_dbase[d] + z_bits(z_dext[d]) if distance > w { return -1 } if w + length > cap { return -1 } # bounds once, not per byte var sp = w - distance var k = 0 while k < length { out[w] = out[sp] w += 1 sp += 1 k += 1 } } sym = z_decode(lit) } return w } function z_fixed_tables(lit: words, dist: words) -> void { let lengths: words = words(288) for i in 0 .. 144 { lengths[i] = 8 } for i in 144 .. 256 { lengths[i] = 9 } for i in 256 .. 280 { lengths[i] = 7 } for i in 280 .. 288 { lengths[i] = 8 } z_table_build(lit, lengths, 288) for i in 0 .. 30 { lengths[i] = 5 } z_table_build(dist, lengths, 30) free(lengths) } function z_dynamic_tables(lit: words, dist: words) -> int { let nlen = z_bits(5) + 257 let ndist = z_bits(5) + 1 let ncode = z_bits(4) + 4 if nlen > 286 { return 0 } if ndist > 30 { return 0 } let lengths: words = words(320) for i in 0 .. 19 { lengths[i] = 0 } # the code-length alphabet is transmitted in this fixed permutation # 16,17,18,0,8,7,9,6,10,5,11,4,12,3,13,2,14,1,15 — biased by '0' so it is one literal let order = "@AB08796:5;4<3=2>1?" for i in 0 .. ncode { lengths[order[i] - 48] = z_bits(3) } let clen = z_table_new(19) z_table_build(clen, lengths, 19) var n = 0 while n < nlen + ndist { let sym = z_decode(clen) if sym < 0 { return 0 } if sym < 16 { lengths[n] = sym n += 1 } if sym >= 16 { var prev = 0 var rep = 0 if sym == 16 { if n == 0 { return 0 } prev = lengths[n - 1] rep = 3 + z_bits(2) } if sym == 17 { rep = 3 + z_bits(3) } if sym == 18 { rep = 11 + z_bits(7) } for k in 0 .. rep { if n < 320 { lengths[n] = prev n += 1 } } } } z_table_build(lit, lengths, nlen) # the distance lengths follow the literal ones in the same buffer let dl: words = words(32) for i in 0 .. ndist { dl[i] = lengths[nlen + i] } z_table_build(dist, dl, ndist) free(dl) free(lengths) free(clen) return 1 } # Inflate a raw DEFLATE stream. Returns bytes written, or -1. function z_inflate(src: pointer, len: int, out: pointer, cap: int) -> int { z_start(src, len) let lit = z_table_new(288) let dist = z_table_new(30) var w = 0 var final = 0 while final == 0 { final = z_bits(1) let btype = z_bits(2) if z_err != 0 { return -1 } if btype == 0 { w = z_stored(out, w, cap) } if btype == 1 { z_fixed_tables(lit, dist) w = z_codes(out, w, cap, lit, dist) } if btype == 2 { if z_dynamic_tables(lit, dist) == 0 { return -1 } w = z_codes(out, w, cap, lit, dist) } if btype == 3 { return -1 } if w < 0 { return -1 } } free(lit) free(dist) return w } # zlib wrapper (RFC 1950): two header bytes, then DEFLATE, then Adler-32. function z_uncompress(src: pointer, len: int, out: pointer, cap: int) -> int { if len < 2 { return -1 } let cmf = src[0] if (cmf & 15) != 8 { return -1 } return z_inflate(offset(src, 2), len - 2, out, cap) } # gzip framing (RFC 1952): a 10-byte header (magic 1f 8b, CM=8, FLG, 4-byte MTIME, # XFL, OS), optional FEXTRA/FNAME/FCOMMENT/FHCRC fields selected by FLG, then the # same DEFLATE body zlib carries, then an 8-byte CRC32 + ISIZE trailer. We skip # the header + optional fields, inflate the body, and ignore the trailer — the # CRC is a redundancy check, not needed to decode (PNG likewise ignores ancillary # CRCs). Returns bytes written, or -1. function z_gunzip(src: pointer, len: int, out: pointer, cap: int) -> int { if len < 18 { return -1 } # 10 header + 8 trailer minimum if src[0] != 31 { return -1 } # 0x1f if src[1] != 139 { return -1 } # 0x8b if src[2] != 8 { return -1 } # CM must be DEFLATE let flg = src[3] var pos = 10 # past the fixed header if (flg & 4) != 0 { # FEXTRA: 2-byte length then that many bytes if pos + 2 > len { return -1 } let xlen = src[pos] + (src[pos + 1] << 8) pos = pos + 2 + xlen } if (flg & 8) != 0 { # FNAME: NUL-terminated while pos < len and src[pos] != 0 { pos += 1 } pos += 1 } if (flg & 16) != 0 { # FCOMMENT: NUL-terminated while pos < len and src[pos] != 0 { pos += 1 } pos += 1 } if (flg & 2) != 0 { pos += 2 } # FHCRC: 2-byte header CRC if pos + 8 > len { return -1 } return z_inflate(offset(src, pos), len - pos - 8, out, cap) }