34 lines
1.5 KiB
Text
34 lines
1.5 KiB
Text
# json_unicode.ludic - Json.parse decodes \uXXXX to UTF-8 (a surrogate pair joined into one code point,
|
|
# a lone surrogate or bad hex as U+FFFD) and \t \r \b \f, as Python's json.dump writes non-ASCII.
|
|
# Checked byte by byte, so this file stays ASCII. Prints `json unicode 8 of 8`.
|
|
program JsonUnicode {
|
|
# the string value of {"s": <json>}
|
|
function str_of(json: string) -> string {
|
|
let v = Json.parse("{\"s\": " + json + "}")
|
|
return Value.as_str(Value.get(v, "s"))
|
|
}
|
|
function bytes_are(s: string, want: []int) -> bool {
|
|
if len(s) != len(want) { return false }
|
|
for i in 0 .. len(want) {
|
|
if (s[i] & 255) != want[i] { return false }
|
|
}
|
|
return true
|
|
}
|
|
function check(name: string, got: string, want: []int, ok: int) -> int {
|
|
if bytes_are(got, want) { return ok + 1 }
|
|
print(`json unicode: {name} wrong ({len(got)} bytes)`)
|
|
return ok
|
|
}
|
|
entry {
|
|
var ok = 0
|
|
ok = check("two bytes", str_of("\"\\u015f\""), [197, 159], ok)
|
|
ok = check("ascii around", str_of("\"i\\u0131x\""), [105, 196, 177, 120], ok)
|
|
ok = check("three bytes", str_of("\"\\u20ac\""), [226, 130, 172], ok)
|
|
ok = check("surrogate pair", str_of("\"\\ud83d\\ude00\""), [240, 159, 152, 128], ok)
|
|
ok = check("lone high", str_of("\"\\ud83dx\""), [239, 191, 189, 120], ok)
|
|
ok = check("lone low", str_of("\"\\ude00\""), [239, 191, 189], ok)
|
|
ok = check("bad hex", str_of("\"\\uZZ12\""), [239, 191, 189, 90, 90, 49, 50], ok)
|
|
ok = check("controls", str_of("\"\\t\\r\\b\\f\\n\""), [9, 13, 8, 12, 10], ok)
|
|
print(`json unicode {ok} of 8`)
|
|
}
|
|
}
|