feat(stdlib): Tiled P0 — XML reader + base64 decode + gzip framing (#67)
The three parsing primitives the TMX path needs that were not already in the tree (zlib inflate + the JSON reader already shipped): - Xml.* — a minimal, deterministic pure-Ludic XML reader for the element/attribute/CDATA subset TMX/TSX/TX use, spliced on demand. Handles nested elements, single/double-quoted attributes, text + <![CDATA[…]]>, comments, the <?xml?> prolog and <!DOCTYPE>, the five predefined entities and numeric character references. - Base64.* — standard base64 (RFC 4648) decode + a matching pure-Ludic encoder; the decoder ignores the whitespace Tiled wraps into <data>. Splicing Base64.* also pulls in inflate.ludic so a plain tool can run the full base64 -> zlib/gzip decode chain. - z_gunzip — gzip framing (RFC 1952) over the existing DEFLATE inflater: skip the 10-byte header + optional fields, inflate the body, ignore the CRC32/ISIZE trailer. Golden corpus vendored under assets/tiled-fixtures/ (mapeditor/tiled examples, attributed, + hand-authored multi-encoding fixtures). New example library/tiled_p0.ludic proves all three (21 assertions); docs pages + inventory added so check-impl stays green. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
parent
0b149a6876
commit
79776d1f6a
35 changed files with 20549 additions and 18415 deletions
87
runtime/native/base64.ludic
Normal file
87
runtime/native/base64.ludic
Normal file
|
|
@ -0,0 +1,87 @@
|
|||
# ============================================================================
|
||||
# base64.ludic — standard base64 (RFC 4648) encode + decode, in Ludic (`Base64.*`).
|
||||
#
|
||||
# `Crypto.base64` already encodes (hand-written IR, tied to the crypto prelude);
|
||||
# the Tiled data pipeline (issue #67) needs a *decoder* — base64 layer data feeds
|
||||
# the DEFLATE inflater. This is that decoder, plus a matching pure-Ludic encoder
|
||||
# so the namespace is self-contained and spliceable on its own (no crypto prelude
|
||||
# dependency). Deterministic, allocation the only cost.
|
||||
#
|
||||
# ludicc splices this file when a program mentions `Base64.*` (parse.ludic). The
|
||||
# decoder ignores ASCII whitespace (TMX embeds newlines/indentation in its base64
|
||||
# `<data>`), and '=' padding ends the stream.
|
||||
# ============================================================================
|
||||
|
||||
# the standard alphabet value of a base64 character, or -1 for anything else
|
||||
# (whitespace, '=', stray bytes — the decoder skips them / stops on '=').
|
||||
function b64_val(c: int) -> int {
|
||||
if c >= 65 and c <= 90 { return c - 65 } # 'A'..'Z' -> 0..25
|
||||
if c >= 97 and c <= 122 { return c - 71 } # 'a'..'z' -> 26..51
|
||||
if c >= 48 and c <= 57 { return c + 4 } # '0'..'9' -> 52..61
|
||||
if c == 43 { return 62 } # '+'
|
||||
if c == 47 { return 63 } # '/'
|
||||
return 0 - 1
|
||||
}
|
||||
|
||||
# decode NUL-terminated base64 `src` into the caller's `out` buffer; returns the
|
||||
# number of bytes written. Whitespace is ignored; '=' padding terminates. `out`
|
||||
# must hold at least (len(src)*3)/4 bytes.
|
||||
function b64_decode(src: pointer, out: pointer) -> int {
|
||||
var acc = 0
|
||||
var nbits = 0
|
||||
var w = 0
|
||||
var i = 0
|
||||
var c = src[i]
|
||||
while c != 0 {
|
||||
if c == 61 { return w } # '=' -> done
|
||||
let v = b64_val(c)
|
||||
if v >= 0 {
|
||||
acc = (acc << 6) | v
|
||||
nbits = nbits + 6
|
||||
if nbits >= 8 {
|
||||
nbits = nbits - 8
|
||||
out[w] = (acc >> nbits) & 255
|
||||
w = w + 1
|
||||
}
|
||||
}
|
||||
i = i + 1
|
||||
c = src[i]
|
||||
}
|
||||
return w
|
||||
}
|
||||
|
||||
# decode base64 text -> a fresh NUL-terminated string of the decoded bytes.
|
||||
# (For binary payloads that may contain NUL, decode into a sized buffer with
|
||||
# `b64_decode` and track the returned length instead.)
|
||||
function base64_decode(src: pointer) -> pointer {
|
||||
let cap = len(src) + 4
|
||||
let out = bytes(cap)
|
||||
let w = b64_decode(src, out)
|
||||
out[w] = 0
|
||||
return out
|
||||
}
|
||||
|
||||
# encode the bytes of NUL-terminated `src` as standard base64 (with '=' padding).
|
||||
function base64_encode(src: pointer) -> pointer {
|
||||
let tab = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"
|
||||
let n = len(src)
|
||||
var out = ""
|
||||
var i = 0
|
||||
while i < n {
|
||||
let b0 = src[i]
|
||||
var b1 = 0
|
||||
var b2 = 0
|
||||
var have = 1
|
||||
if i + 1 < n { b1 = src[i + 1]; have = 2 }
|
||||
if i + 2 < n { b2 = src[i + 2]; have = 3 }
|
||||
let e0 = b0 >> 2
|
||||
let e1 = ((b0 & 3) << 4) | (b1 >> 4)
|
||||
let e2 = ((b1 & 15) << 2) | (b2 >> 6)
|
||||
let e3 = b2 & 63
|
||||
out = out + tab[e0..e0 + 1] + tab[e1..e1 + 1]
|
||||
if have >= 2 { out = out + tab[e2..e2 + 1] } else { out = out + "=" }
|
||||
if have >= 3 { out = out + tab[e3..e3 + 1] } else { out = out + "=" }
|
||||
i = i + 3
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
|
@ -274,3 +274,34 @@ function z_uncompress(src: pointer, len: int, out: pointer, cap: int) -> int {
|
|||
if (cmf & 15) != 8 { return -1 }
|
||||
return z_inflate(offset(src, 2), len - 2, out, cap)
|
||||
}
|
||||
|
||||
# gzip framing (RFC 1952): a 10-byte header (magic 1f 8b, CM=8, FLG, 4-byte MTIME,
|
||||
# XFL, OS), optional FEXTRA/FNAME/FCOMMENT/FHCRC fields selected by FLG, then the
|
||||
# same DEFLATE body zlib carries, then an 8-byte CRC32 + ISIZE trailer. We skip
|
||||
# the header + optional fields, inflate the body, and ignore the trailer — the
|
||||
# CRC is a redundancy check, not needed to decode (PNG likewise ignores ancillary
|
||||
# CRCs). Returns bytes written, or -1.
|
||||
function z_gunzip(src: pointer, len: int, out: pointer, cap: int) -> int {
|
||||
if len < 18 { return -1 } # 10 header + 8 trailer minimum
|
||||
if src[0] != 31 { return -1 } # 0x1f
|
||||
if src[1] != 139 { return -1 } # 0x8b
|
||||
if src[2] != 8 { return -1 } # CM must be DEFLATE
|
||||
let flg = src[3]
|
||||
var pos = 10 # past the fixed header
|
||||
if (flg & 4) != 0 { # FEXTRA: 2-byte length then that many bytes
|
||||
if pos + 2 > len { return -1 }
|
||||
let xlen = src[pos] + (src[pos + 1] << 8)
|
||||
pos = pos + 2 + xlen
|
||||
}
|
||||
if (flg & 8) != 0 { # FNAME: NUL-terminated
|
||||
while pos < len and src[pos] != 0 { pos = pos + 1 }
|
||||
pos = pos + 1
|
||||
}
|
||||
if (flg & 16) != 0 { # FCOMMENT: NUL-terminated
|
||||
while pos < len and src[pos] != 0 { pos = pos + 1 }
|
||||
pos = pos + 1
|
||||
}
|
||||
if (flg & 2) != 0 { pos = pos + 2 } # FHCRC: 2-byte header CRC
|
||||
if pos + 8 > len { return -1 }
|
||||
return z_inflate(offset(src, pos), len - pos - 8, out, cap)
|
||||
}
|
||||
|
|
|
|||
301
runtime/native/xml.ludic
Normal file
301
runtime/native/xml.ludic
Normal file
|
|
@ -0,0 +1,301 @@
|
|||
# ============================================================================
|
||||
# xml.ludic — a minimal pure-Ludic XML reader (`Xml.*`), for the TMX/TSX/TX
|
||||
# subset Tiled emits (Tiled design §0.1, issue #67). It handles exactly what the
|
||||
# native Tiled formats use: elements, single/double-quoted attributes, nested
|
||||
# children, text/CDATA content, comments, the `<?xml?>` prolog and `<!DOCTYPE>`,
|
||||
# and the five predefined entities plus numeric character references. It is NOT
|
||||
# a validating or namespace-aware parser — best-effort, deterministic, allocation
|
||||
# the only cost, the same contract as the JSON reader (value.ludic).
|
||||
#
|
||||
# ludicc splices this file when a program mentions `Xml.*` (parse.ludic), exactly
|
||||
# like `Regex.*`/`Value.*`. The tree is plain Ludic over heap records.
|
||||
#
|
||||
# A node is one element: its `tag`, parallel attribute `akeys`/`avals`, ordered
|
||||
# child elements `kids`, and the concatenated character data `text` (the CSV in a
|
||||
# `<data encoding="csv">` element, or a `<property>` string, lands here).
|
||||
# ============================================================================
|
||||
|
||||
property Xml {
|
||||
tag: pointer = null # element name ("" for the synthetic empty node)
|
||||
text: pointer = null # concatenated character data of this element
|
||||
akeys: []pointer # attribute names
|
||||
avals: []pointer # attribute values (entity-decoded)
|
||||
kids: []Xml # child elements, in document order
|
||||
}
|
||||
|
||||
function xml_new(tag: pointer) -> Xml {
|
||||
let n = new Xml
|
||||
n.tag = tag
|
||||
n.text = ""
|
||||
n.akeys = new []pointer
|
||||
n.avals = new []pointer
|
||||
n.kids = new []Xml
|
||||
return n
|
||||
}
|
||||
|
||||
# --- accessors --------------------------------------------------------------
|
||||
function xml_tag(n: Xml) -> pointer { if n.tag == null { return "" }; return n.tag }
|
||||
function xml_text(n: Xml) -> pointer { if n.text == null { return "" }; return n.text }
|
||||
function xml_child_count(n: Xml) -> int { return len(n.kids) }
|
||||
function xml_child(n: Xml, i: int) -> Xml {
|
||||
if i < 0 or i >= len(n.kids) { return xml_new("") }
|
||||
return n.kids[i]
|
||||
}
|
||||
function xml_attr_count(n: Xml) -> int { return len(n.akeys) }
|
||||
function xml_has(n: Xml, key: pointer) -> int {
|
||||
var i = 0
|
||||
while i < len(n.akeys) { if n.akeys[i] == key { return 1 }; i = i + 1 }
|
||||
return 0
|
||||
}
|
||||
function xml_attr(n: Xml, key: pointer) -> pointer {
|
||||
var i = 0
|
||||
while i < len(n.akeys) { if n.akeys[i] == key { return n.avals[i] }; i = i + 1 }
|
||||
return ""
|
||||
}
|
||||
# attribute as an integer (decimal, optional leading '-'); `dflt` when absent.
|
||||
function xml_attr_int(n: Xml, key: pointer, dflt: int) -> int {
|
||||
if xml_has(n, key) == 0 { return dflt }
|
||||
return xml_atoi(xml_attr(n, key))
|
||||
}
|
||||
|
||||
# the first direct child named `tag`, or the synthetic empty node if none.
|
||||
function xml_find(n: Xml, tag: pointer) -> Xml {
|
||||
var i = 0
|
||||
while i < len(n.kids) { if n.kids[i].tag == tag { return n.kids[i] }; i = i + 1 }
|
||||
return xml_new("")
|
||||
}
|
||||
# count direct children named `tag`.
|
||||
function xml_count(n: Xml, tag: pointer) -> int {
|
||||
var c = 0
|
||||
var i = 0
|
||||
while i < len(n.kids) { if n.kids[i].tag == tag { c = c + 1 }; i = i + 1 }
|
||||
return c
|
||||
}
|
||||
|
||||
# parse a signed decimal integer prefix of `s` (stops at the first non-digit).
|
||||
function xml_atoi(s: pointer) -> int {
|
||||
var i = 0
|
||||
let n = len(s)
|
||||
var neg = 0
|
||||
if i < n and s[i] == 45 { neg = 1; i = i + 1 } # '-'
|
||||
var v = 0
|
||||
while i < n and s[i] >= 48 and s[i] <= 57 {
|
||||
v = v * 10 + (s[i] - 48)
|
||||
i = i + 1
|
||||
}
|
||||
if neg != 0 { return 0 - v }
|
||||
return v
|
||||
}
|
||||
|
||||
# --- entity decoding --------------------------------------------------------
|
||||
# expand the five predefined entities and &#NN; / &#xHH; numeric references in a
|
||||
# raw run. Only bytes 0..255 of a character reference are emitted (Ludic strings
|
||||
# are byte strings); a code point above that is written as its low byte, which is
|
||||
# ample for the ASCII/Latin-1 text Tiled attributes carry.
|
||||
function xml_unescape(s: pointer) -> pointer {
|
||||
# fast path: no '&' means nothing to expand
|
||||
var k = 0
|
||||
let m = len(s)
|
||||
var amp = 0
|
||||
while k < m { if s[k] == 38 { amp = 1; k = m } else { k = k + 1 } }
|
||||
if amp == 0 { return s }
|
||||
var out = ""
|
||||
var i = 0
|
||||
while i < m {
|
||||
let c = s[i]
|
||||
if c != 38 { # not '&'
|
||||
# copy the run up to the next '&' in one slice
|
||||
var j = i
|
||||
while j < m and s[j] != 38 { j = j + 1 }
|
||||
out = out + s[i..j]
|
||||
i = j
|
||||
} else {
|
||||
# find the ';'
|
||||
var j = i + 1
|
||||
while j < m and s[j] != 59 { j = j + 1 } # ';'
|
||||
if j >= m { out = out + s[i..m]; i = m }
|
||||
else {
|
||||
let ent = s[i + 1..j]
|
||||
if ent == "amp" { out = out + "&" }
|
||||
else { if ent == "lt" { out = out + "<" }
|
||||
else { if ent == "gt" { out = out + ">" }
|
||||
else { if ent == "quot" { out = out + "\"" }
|
||||
else { if ent == "apos" { out = out + "'" }
|
||||
else {
|
||||
if len(ent) >= 2 and ent[0] == 35 { # '#' numeric reference
|
||||
var code = 0
|
||||
if ent[1] == 120 or ent[1] == 88 { # '#x' hex
|
||||
var h = 2
|
||||
while h < len(ent) { code = code * 16 + xml_hexval(ent[h]); h = h + 1 }
|
||||
} else {
|
||||
var d = 1
|
||||
while d < len(ent) { code = code * 10 + (ent[d] - 48); d = d + 1 }
|
||||
}
|
||||
out = out + xml_byte(code & 255)
|
||||
} else {
|
||||
out = out + "&" + ent + ";" # unknown entity, keep literal
|
||||
}
|
||||
} } } } }
|
||||
i = j + 1
|
||||
}
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
function xml_hexval(c: int) -> int {
|
||||
if c >= 48 and c <= 57 { return c - 48 }
|
||||
if c >= 97 and c <= 102 { return c - 87 }
|
||||
if c >= 65 and c <= 70 { return c - 55 }
|
||||
return 0
|
||||
}
|
||||
|
||||
# a one-byte string holding byte value `b` (1..255); "" for 0 (a NUL can't sit in
|
||||
# a Ludic string). Built by slicing a 256-byte table of every byte value.
|
||||
var xml_bytetab: pointer = null
|
||||
function xml_byte(b: int) -> pointer {
|
||||
if b <= 0 { return "" }
|
||||
if xml_bytetab == null {
|
||||
let t = bytes(257)
|
||||
var i = 0
|
||||
while i < 256 { t[i] = i + 1; i = i + 1 } # table[i] = byte (i+1), so 0 never appears
|
||||
t[256] = 0
|
||||
xml_bytetab = t
|
||||
}
|
||||
return xml_bytetab[b - 1..b]
|
||||
}
|
||||
|
||||
# --- parser -----------------------------------------------------------------
|
||||
property XP { s: pointer = null, i: int = 0, n: int = 0 }
|
||||
|
||||
function xp_ws(c: int) -> bool { return c == 32 or c == 9 or c == 10 or c == 13 }
|
||||
|
||||
function xp_skip_ws(p: XP) -> void {
|
||||
while p.i < p.n and xp_ws(p.s[p.i]) { p.i = p.i + 1 }
|
||||
}
|
||||
|
||||
# skip a `<?...?>`, `<!-- ... -->` or `<!DOCTYPE ...>` at the cursor. Returns true
|
||||
# if it consumed one (cursor on '<').
|
||||
function xp_skip_misc(p: XP) -> bool {
|
||||
if p.i + 1 >= p.n or p.s[p.i] != 60 { return false } # '<'
|
||||
let c = p.s[p.i + 1]
|
||||
if c == 63 { # '<?' ... '?>'
|
||||
p.i = p.i + 2
|
||||
while p.i + 1 < p.n and not (p.s[p.i] == 63 and p.s[p.i + 1] == 62) { p.i = p.i + 1 }
|
||||
p.i = p.i + 2
|
||||
return true
|
||||
}
|
||||
if c == 33 { # '<!'
|
||||
if p.i + 3 < p.n and p.s[p.i + 2] == 45 and p.s[p.i + 3] == 45 { # '<!--' comment
|
||||
p.i = p.i + 4
|
||||
while p.i + 2 < p.n and not (p.s[p.i] == 45 and p.s[p.i + 1] == 45 and p.s[p.i + 2] == 62) { p.i = p.i + 1 }
|
||||
p.i = p.i + 3
|
||||
return true
|
||||
}
|
||||
# '<!DOCTYPE ...>' or other declaration — skip to the matching '>'
|
||||
p.i = p.i + 2
|
||||
while p.i < p.n and p.s[p.i] != 62 { p.i = p.i + 1 }
|
||||
p.i = p.i + 1
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
# read a name (element or attribute): letters, digits, '_', '-', ':', '.'
|
||||
function xp_name(p: XP) -> pointer {
|
||||
let start = p.i
|
||||
while p.i < p.n {
|
||||
let c = p.s[p.i]
|
||||
let ok = (c >= 65 and c <= 90) or (c >= 97 and c <= 122) or (c >= 48 and c <= 57)
|
||||
if ok or c == 95 or c == 45 or c == 58 or c == 46 { p.i = p.i + 1 }
|
||||
else { break }
|
||||
}
|
||||
return p.s[start..p.i]
|
||||
}
|
||||
|
||||
# parse `key="value"` / `key='value'` attributes into the element node.
|
||||
function xp_attrs(p: XP, node: Xml) -> void {
|
||||
while true {
|
||||
xp_skip_ws(p)
|
||||
if p.i >= p.n { return }
|
||||
let c = p.s[p.i]
|
||||
if c == 62 or c == 47 or c == 63 { return } # '>' '/' '?'
|
||||
let key = xp_name(p)
|
||||
if len(key) == 0 { p.i = p.i + 1; continue } # stray char, don't stall
|
||||
xp_skip_ws(p)
|
||||
var val = ""
|
||||
if p.i < p.n and p.s[p.i] == 61 { # '='
|
||||
p.i = p.i + 1
|
||||
xp_skip_ws(p)
|
||||
if p.i < p.n and (p.s[p.i] == 34 or p.s[p.i] == 39) {
|
||||
let q = p.s[p.i]
|
||||
p.i = p.i + 1
|
||||
let start = p.i
|
||||
while p.i < p.n and p.s[p.i] != q { p.i = p.i + 1 }
|
||||
val = xml_unescape(p.s[start..p.i])
|
||||
p.i = p.i + 1 # skip closing quote
|
||||
}
|
||||
}
|
||||
push(node.akeys, key)
|
||||
push(node.avals, val)
|
||||
}
|
||||
}
|
||||
|
||||
# parse one element (cursor on its opening '<'). Recurses for children.
|
||||
function xp_element(p: XP) -> Xml {
|
||||
p.i = p.i + 1 # skip '<'
|
||||
let name = xp_name(p)
|
||||
let node = xml_new(name)
|
||||
xp_attrs(p, node)
|
||||
# self-closing '/>'
|
||||
if p.i < p.n and p.s[p.i] == 47 { # '/'
|
||||
p.i = p.i + 1
|
||||
if p.i < p.n and p.s[p.i] == 62 { p.i = p.i + 1 } # '>'
|
||||
return node
|
||||
}
|
||||
if p.i < p.n and p.s[p.i] == 62 { p.i = p.i + 1 } # '>'
|
||||
# content until the matching close tag
|
||||
while p.i < p.n {
|
||||
if p.s[p.i] == 60 { # '<'
|
||||
if p.i + 1 < p.n and p.s[p.i + 1] == 47 { # '</' close
|
||||
p.i = p.i + 2
|
||||
let cn = xp_name(p)
|
||||
while p.i < p.n and p.s[p.i] != 62 { p.i = p.i + 1 }
|
||||
p.i = p.i + 1 # skip '>'
|
||||
return node
|
||||
}
|
||||
if p.i + 3 < p.n and p.s[p.i + 1] == 33 and p.s[p.i + 2] == 91 { # '<![' CDATA
|
||||
# <![CDATA[ ... ]]>
|
||||
p.i = p.i + 9 # past "<![CDATA["
|
||||
let start = p.i
|
||||
while p.i + 2 < p.n and not (p.s[p.i] == 93 and p.s[p.i + 1] == 93 and p.s[p.i + 2] == 62) { p.i = p.i + 1 }
|
||||
node.text = node.text + p.s[start..p.i]
|
||||
p.i = p.i + 3 # past "]]>"
|
||||
} else {
|
||||
if xp_skip_misc(p) { } # comment / PI inside content
|
||||
else { push(node.kids, xp_element(p)) }
|
||||
}
|
||||
} else {
|
||||
# character data run up to the next '<'
|
||||
let start = p.i
|
||||
while p.i < p.n and p.s[p.i] != 60 { p.i = p.i + 1 }
|
||||
node.text = node.text + xml_unescape(p.s[start..p.i])
|
||||
}
|
||||
}
|
||||
return node
|
||||
}
|
||||
|
||||
# parse a whole document -> its root element (or the synthetic empty node).
|
||||
function xml_parse(s: pointer) -> Xml {
|
||||
let p = new XP
|
||||
p.s = s; p.i = 0; p.n = len(s)
|
||||
while p.i < p.n {
|
||||
xp_skip_ws(p)
|
||||
if p.i >= p.n { break }
|
||||
if p.s[p.i] == 60 { # '<'
|
||||
if xp_skip_misc(p) { } # prolog / comment / doctype
|
||||
else { return xp_element(p) }
|
||||
} else { p.i = p.i + 1 }
|
||||
}
|
||||
return xml_new("")
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue