# json_unicode.ludic - Json.parse decodes \uXXXX to UTF-8 (a surrogate pair joined into one code point, # a lone surrogate or bad hex as U+FFFD) and \t \r \b \f, as Python's json.dump writes non-ASCII. # Checked byte by byte, so this file stays ASCII. Prints `json unicode 8 of 8`. program JsonUnicode { # the string value of {"s": } function str_of(json: string) -> string { let v = Json.parse("{\"s\": " + json + "}") return Value.as_str(Value.get(v, "s")) } function bytes_are(s: string, want: []int) -> bool { if len(s) != len(want) { return false } for i in 0 .. len(want) { if (s[i] & 255) != want[i] { return false } } return true } function check(name: string, got: string, want: []int, ok: int) -> int { if bytes_are(got, want) { return ok + 1 } print(`json unicode: {name} wrong ({len(got)} bytes)`) return ok } entry { var ok = 0 ok = check("two bytes", str_of("\"\\u015f\""), [197, 159], ok) ok = check("ascii around", str_of("\"i\\u0131x\""), [105, 196, 177, 120], ok) ok = check("three bytes", str_of("\"\\u20ac\""), [226, 130, 172], ok) ok = check("surrogate pair", str_of("\"\\ud83d\\ude00\""), [240, 159, 152, 128], ok) ok = check("lone high", str_of("\"\\ud83dx\""), [239, 191, 189, 120], ok) ok = check("lone low", str_of("\"\\ude00\""), [239, 191, 189], ok) ok = check("bad hex", str_of("\"\\uZZ12\""), [239, 191, 189, 90, 90, 49, 50], ok) ok = check("controls", str_of("\"\\t\\r\\b\\f\\n\""), [9, 13, 8, 12, 10], ok) print(`json unicode {ok} of 8`) } }