ludic/runtime/native/xml.ludic

365 lines
13 KiB
Text

# ============================================================================
# xml.ludic — a minimal pure-Ludic XML reader (`Xml.*`), for the TMX/TSX/TX
# subset Tiled emits (Tiled design §0.1, issue #67). It handles exactly what the
# native Tiled formats use: elements, single/double-quoted attributes, nested
# children, text/CDATA content, comments, the `<?xml?>` prolog and `<!DOCTYPE>`,
# and the five predefined entities plus numeric character references. It is NOT
# a validating or namespace-aware parser — best-effort, deterministic, allocation
# the only cost, the same contract as the JSON reader (value.ludic).
#
# ludicc splices this file when a program mentions `Xml.*` (parse.ludic), exactly
# like `Regex.*`/`Value.*`. The tree is plain Ludic over heap records.
#
# A node is one element: its `tag`, parallel attribute `akeys`/`avals`, ordered
# child elements `kids`, and the concatenated character data `text` (the CSV in a
# `<data encoding="csv">` element, or a `<property>` string, lands here).
# ============================================================================
property Xml {
tag: pointer = null # element name ("" for the synthetic empty node)
text: pointer = null # concatenated character data of this element
akeys: []pointer # attribute names
avals: []pointer # attribute values (entity-decoded)
kids: []Xml # child elements, in document order
at: int = 0 # where its '<' is in the text, for an error that names a line
mixed: []Xml # its elements and its runs of text ("#text"), in document order
joined: bool = false # `text` is a join made here (not a run a #text node shares): free to replace
}
function xml_new(tag: pointer) -> Xml {
let n = new Xml
n.tag = tag
n.text = ""
n.akeys = new []pointer
n.avals = new []pointer
n.kids = new []Xml
n.mixed = new []Xml
return n
}
# --- accessors --------------------------------------------------------------
function xml_tag(n: Xml) -> string { if n.tag == null { return "" }; return n.tag }
function xml_text(n: Xml) -> string { if n.text == null { return "" }; return n.text }
function xml_child_count(n: Xml) -> int { return len(n.kids) }
function xml_child(n: Xml, i: int) -> Xml {
if i < 0 or i >= len(n.kids) { return xml_new("") }
return n.kids[i]
}
function xml_attr_count(n: Xml) -> int { return len(n.akeys) }
function xml_has(n: Xml, key: pointer) -> int {
var i = 0
while i < len(n.akeys) { if n.akeys[i] == key { return 1 }; i += 1 }
return 0
}
function xml_attr(n: Xml, key: pointer) -> string {
var i = 0
while i < len(n.akeys) { if n.akeys[i] == key { return n.avals[i] }; i += 1 }
return ""
}
# attribute as an integer (decimal, optional leading '-'); `dflt` when absent.
function xml_attr_int(n: Xml, key: pointer, dflt: int) -> int {
if xml_has(n, key) == 0 { return dflt }
return xml_atoi(xml_attr(n, key))
}
# the first direct child named `tag`, or the synthetic empty node if none.
function xml_find(n: Xml, tag: pointer) -> Xml {
var i = 0
while i < len(n.kids) { if n.kids[i].tag == tag { return n.kids[i] }; i += 1 }
return xml_new("")
}
# count direct children named `tag`.
function xml_count(n: Xml, tag: pointer) -> int {
var c = 0
var i = 0
while i < len(n.kids) { if n.kids[i].tag == tag { c += 1 }; i += 1 }
return c
}
# parse a signed decimal integer prefix of `s` (stops at the first non-digit).
function xml_atoi(s: pointer) -> int {
var i = 0
let n = len(s)
var neg = 0
if i < n and s[i] == '-' { neg = 1; i += 1 } # '-'
var v = 0
while i < n and s[i] >= '0' and s[i] <= '9' {
v = v * 10 + (s[i] - 48)
i += 1
}
if neg != 0 { return -v }
return v
}
# --- entity decoding --------------------------------------------------------
# expand the five predefined entities and &#NN; / &#xHH; numeric references in a
# raw run. Only bytes 0..255 of a character reference are emitted (Ludic strings
# are byte strings); a code point above that is written as its low byte, which is
# ample for the ASCII/Latin-1 text Tiled attributes carry.
function xml_unescape(rt_xml_st: mut RtXmlState, s: pointer) -> string {
# fast path: no '&' means nothing to expand
var k = 0
let m = len(s)
var amp = 0
while k < m { if s[k] == '&' { amp = 1; k = m } else { k += 1 } }
if amp == 0 { return s }
var out = ""
var i = 0
while i < m {
let c = s[i]
if c != '&' { # not '&'
# copy the run up to the next '&' in one slice
var j = i
while j < m and s[j] != '&' { j += 1 }
out += s[i..j]
i = j
} else {
# find the ';'
var j = i + 1
while j < m and s[j] != ';' { j += 1 } # ';'
if j >= m { out += s[i..m]; i = m }
else {
let ent = s[i + 1..j]
if ent == "amp" { out += "&" }
else { if ent == "lt" { out += "<" }
else { if ent == "gt" { out += ">" }
else { if ent == "quot" { out += "\"" }
else { if ent == "apos" { out += "'" }
else {
if len(ent) >= 2 and ent[0] == '#' { # '#' numeric reference
var code = 0
if ent[1] == 'x' or ent[1] == 'X' { # '#x' hex
var h = 2
while h < len(ent) { code = code * 16 + xml_hexval(ent[h]); h += 1 }
} else {
var d = 1
while d < len(ent) { code = code * 10 + (ent[d] - 48); d += 1 }
}
out += xml_byte(rt_xml_st, code & 255)
} else {
out = out + "&" + ent + ";" # unknown entity, keep literal
}
} } } } }
i = j + 1
}
}
}
return out
}
function xml_hexval(c: int) -> int {
if c >= '0' and c <= '9' { return c - 48 }
if c >= 'a' and c <= 'f' { return c - 87 }
if c >= 'A' and c <= 'F' { return c - 55 }
return 0
}
# a one-byte string holding byte value `b` (1..255); "" for 0 (a NUL can't sit in
# a Ludic string). Built by slicing a 256-byte table of every byte value.
export state RtXmlState {
xml_bytetab: pointer = null
}
function xml_byte(rt_xml_st: mut RtXmlState, b: int) -> string {
if b <= 0 { return "" }
if rt_xml_st.xml_bytetab == null {
let t = bytes(257)
var i = 0
while i < 256 { t[i] = i + 1; i += 1 } # table[i] = byte (i+1), so 0 never appears
t[256] = 0
rt_xml_st.xml_bytetab = t
}
return rt_xml_st.xml_bytetab[b - 1..b]
}
# --- parser -----------------------------------------------------------------
property XP { s: pointer = null, i: int = 0, n: int = 0 }
function xp_ws(c: int) -> bool { return c == ' ' or c == '\t' or c == '\n' or c == '\r' }
function xp_skip_ws(p: XP) -> void {
while p.i < p.n and xp_ws(p.s[p.i]) { p.i += 1 }
}
# skip a `<?...?>`, `<!-- ... -->` or `<!DOCTYPE ...>` at the cursor. Returns true
# if it consumed one (cursor on '<').
function xp_skip_misc(p: XP) -> bool {
if p.i + 1 >= p.n or p.s[p.i] != '<' { return false } # '<'
let c = p.s[p.i + 1]
if c == '?' { # '<?' ... '?>'
p.i += 2
while p.i + 1 < p.n and not (p.s[p.i] == '?' and p.s[p.i + 1] == '>') { p.i += 1 }
p.i += 2
return true
}
if c == '!' { # '<!'
if p.i + 3 < p.n and p.s[p.i + 2] == '-' and p.s[p.i + 3] == '-' { # '<!--' comment
p.i += 4
while p.i + 2 < p.n and not (p.s[p.i] == '-' and p.s[p.i + 1] == '-' and p.s[p.i + 2] == '>') { p.i += 1 }
p.i += 3
return true
}
# '<!DOCTYPE ...>' or other declaration — skip to the matching '>'
p.i += 2
while p.i < p.n and p.s[p.i] != '>' { p.i += 1 }
p.i += 1
return true
}
return false
}
# read a name (element or attribute): letters, digits, '_', '-', ':', '.'
function xp_name(p: XP) -> string {
let start = p.i
while p.i < p.n {
let c = p.s[p.i]
let ok = (c >= 'A' and c <= 'Z') or (c >= 'a' and c <= 'z') or (c >= '0' and c <= '9')
if ok or c == '_' or c == '-' or c == ':' or c == '.' { p.i += 1 }
else { break }
}
return p.s[start..p.i]
}
# parse `key="value"` / `key='value'` attributes into the element node.
function xp_attrs(rt_xml_st: mut RtXmlState, p: XP, node: Xml) -> void {
while true {
xp_skip_ws(p)
if p.i >= p.n { return }
let c = p.s[p.i]
if c == '>' or c == '/' or c == '?' { return } # '>' '/' '?'
let key = xp_name(p)
if len(key) == 0 { p.i += 1; continue } # stray char, don't stall
xp_skip_ws(p)
var val = ""
if p.i < p.n and p.s[p.i] == '=' { # '='
p.i += 1
xp_skip_ws(p)
if p.i < p.n and (p.s[p.i] == '"' or p.s[p.i] == '\'') {
let q = p.s[p.i]
p.i += 1
let start = p.i
while p.i < p.n and p.s[p.i] != q { p.i += 1 }
val = xml_unescape(rt_xml_st, p.s[start..p.i])
p.i += 1 # skip closing quote
}
}
push(node.akeys, key)
push(node.avals, val)
}
}
# parse one element (cursor on its opening '<'). Recurses for children.
function xp_element(rt_xml_st: mut RtXmlState, p: XP) -> Xml {
let at = p.i
p.i += 1 # skip '<'
let name = xp_name(p)
let node = xml_new(name)
node.at = at
xp_attrs(rt_xml_st, p, node)
# self-closing '/>'
if p.i < p.n and p.s[p.i] == '/' { # '/'
p.i += 1
if p.i < p.n and p.s[p.i] == '>' { p.i += 1 } # '>'
return node
}
if p.i < p.n and p.s[p.i] == '>' { p.i += 1 } # '>'
# content until the matching close tag
while p.i < p.n {
if p.s[p.i] == '<' { # '<'
if p.i + 1 < p.n and p.s[p.i + 1] == '/' { # '</' close
p.i += 2
let cn = xp_name(p)
free(cn) # the closing name is only read past
while p.i < p.n and p.s[p.i] != '>' { p.i += 1 }
p.i += 1 # skip '>'
return node
}
if p.i + 3 < p.n and p.s[p.i + 1] == '!' and p.s[p.i + 2] == '[' { # '<![' CDATA
# <![CDATA[ ... ]]>
p.i += 9 # past "<![CDATA["
let start = p.i
while p.i + 2 < p.n and not (p.s[p.i] == ']' and p.s[p.i + 1] == ']' and p.s[p.i + 2] == '>') { p.i += 1 }
xp_text_add(node, p.s[start..p.i], true)
p.i += 3 # past "]]>"
} else {
if xp_skip_misc(p) { } # comment / PI inside content
else {
let kid = xp_element(rt_xml_st, p)
push(node.kids, kid)
push(node.mixed, kid)
}
}
} else {
# character data run up to the next '<'
let start = p.i
while p.i < p.n and p.s[p.i] != '<' { p.i += 1 }
let run = xml_unescape(rt_xml_st, p.s[start..p.i])
let kept = xp_text_run(node, run)
xp_text_add(node, run, not kept)
}
}
return node
}
# a run of text among an element's children, kept in order unless it is only white space
function xp_text_run(node: Xml, run: string) -> bool {
var blank = true
var i = 0
while i < len(run) {
let c = run[i]
if not (c == ' ' or c == '\t' or c == '\n' or c == '\r') { blank = false }
i += 1
}
if blank { return false }
let t = xml_new("#text")
t.text = run
push(node.mixed, t)
return true
}
# a run joined onto the element's text. What the join leaves behind is given back: the text it
# replaces when that was a join of its own, and the run when nothing else keeps it (`mine`)
function xp_text_add(node: Xml, run: string, mine: bool) -> void {
if len(node.text) == 0 {
node.text = run # its first text is the run itself
node.joined = mine
return
}
let old = node.text
node.text = old + run
if node.joined { free(old) }
if mine { free(run) }
node.joined = true
}
# parse a whole document -> its root element (or the synthetic empty node).
function xml_parse(rt_xml_st: mut RtXmlState, s: pointer) -> Xml {
let p = new XP
p.s = s; p.i = 0; p.n = len(s)
while p.i < p.n {
xp_skip_ws(p)
if p.i >= p.n { break }
if p.s[p.i] == '<' { # '<'
if xp_skip_misc(p) { } # prolog / comment / doctype
else {
let root = xp_element(rt_xml_st, p)
free(p)
return root
}
} else { p.i += 1 }
}
free(p)
return xml_new("")
}
# the tree's records and lists given back once a reader has built what it keeps from it; its
# strings (tags, text, names, values) are not, since what was built may hold them. Every element
# is in its parent's `mixed`, with the runs of text
function xml_free(n: Xml) -> void {
if n == null { return }
for i in 0 .. len(n.mixed) { xml_free(n.mixed[i]) }
free(n.akeys)
free(n.avals)
free(n.kids)
free(n.mixed)
free(n)
}