ludic/examples/lang/json_unicode.ludic

34 lines
1.5 KiB
Text

# json_unicode.ludic - Json.parse decodes \uXXXX to UTF-8 (a surrogate pair joined into one code point,
# a lone surrogate or bad hex as U+FFFD) and \t \r \b \f, as Python's json.dump writes non-ASCII.
# Checked byte by byte, so this file stays ASCII. Prints `json unicode 8 of 8`.
program JsonUnicode {
# the string value of {"s": <json>}
function str_of(json: string) -> string {
let v = Json.parse("{\"s\": " + json + "}")
return Value.as_str(Value.get(v, "s"))
}
function bytes_are(s: string, want: []int) -> bool {
if len(s) != len(want) { return false }
for i in 0 .. len(want) {
if (s[i] & 255) != want[i] { return false }
}
return true
}
function check(name: string, got: string, want: []int, ok: int) -> int {
if bytes_are(got, want) { return ok + 1 }
print(`json unicode: {name} wrong ({len(got)} bytes)`)
return ok
}
entry {
var ok = 0
ok = check("two bytes", str_of("\"\\u015f\""), [197, 159], ok)
ok = check("ascii around", str_of("\"i\\u0131x\""), [105, 196, 177, 120], ok)
ok = check("three bytes", str_of("\"\\u20ac\""), [226, 130, 172], ok)
ok = check("surrogate pair", str_of("\"\\ud83d\\ude00\""), [240, 159, 152, 128], ok)
ok = check("lone high", str_of("\"\\ud83dx\""), [239, 191, 189, 120], ok)
ok = check("lone low", str_of("\"\\ude00\""), [239, 191, 189], ok)
ok = check("bad hex", str_of("\"\\uZZ12\""), [239, 191, 189, 90, 90, 49, 50], ok)
ok = check("controls", str_of("\"\\t\\r\\b\\f\\n\""), [9, 13, 8, 12, 10], ok)
print(`json unicode {ok} of 8`)
}
}