diff --git a/changes/json-unicode.md b/changes/json-unicode.md new file mode 100644 index 00000000..833b2a3d --- /dev/null +++ b/changes/json-unicode.md @@ -0,0 +1,5 @@ +bump: patch +type: fix +**`Json.parse` decodes `\uXXXX`.** An escaped code point is UTF-8 now - a surrogate pair joined into one, a +lone surrogate or bad hex as U+FFFD - where the backslash was dropped and the hex kept as text ("iu015f"), +and `\t`, `\r`, `\b` and `\f` are the characters they name. Python's `json.dump` writes non-ASCII this way. diff --git a/examples/lang/json_unicode.ludic b/examples/lang/json_unicode.ludic new file mode 100644 index 00000000..ee2553b4 --- /dev/null +++ b/examples/lang/json_unicode.ludic @@ -0,0 +1,34 @@ +# json_unicode.ludic - Json.parse decodes \uXXXX to UTF-8 (a surrogate pair joined into one code point, +# a lone surrogate or bad hex as U+FFFD) and \t \r \b \f, as Python's json.dump writes non-ASCII. +# Checked byte by byte, so this file stays ASCII. Prints `json unicode 8 of 8`. +program JsonUnicode { + # the string value of {"s": } + function str_of(json: string) -> string { + let v = Json.parse("{\"s\": " + json + "}") + return Value.as_str(Value.get(v, "s")) + } + function bytes_are(s: string, want: []int) -> bool { + if len(s) != len(want) { return false } + for i in 0 .. len(want) { + if (s[i] & 255) != want[i] { return false } + } + return true + } + function check(name: string, got: string, want: []int, ok: int) -> int { + if bytes_are(got, want) { return ok + 1 } + print(`json unicode: {name} wrong ({len(got)} bytes)`) + return ok + } + entry { + var ok = 0 + ok = check("two bytes", str_of("\"\\u015f\""), [197, 159], ok) + ok = check("ascii around", str_of("\"i\\u0131x\""), [105, 196, 177, 120], ok) + ok = check("three bytes", str_of("\"\\u20ac\""), [226, 130, 172], ok) + ok = check("surrogate pair", str_of("\"\\ud83d\\ude00\""), [240, 159, 152, 128], ok) + ok = check("lone high", str_of("\"\\ud83dx\""), [239, 191, 189, 120], ok) + ok = check("lone low", str_of("\"\\ude00\""), [239, 191, 189], ok) + ok = check("bad hex", str_of("\"\\uZZ12\""), [239, 191, 189, 90, 90, 49, 50], ok) + ok = check("controls", str_of("\"\\t\\r\\b\\f\\n\""), [9, 13, 8, 12, 10], ok) + print(`json unicode {ok} of 8`) + } +} diff --git a/runtime/native/value.ludic b/runtime/native/value.ludic index 3e15ffa2..4ed5243a 100644 --- a/runtime/native/value.ludic +++ b/runtime/native/value.ludic @@ -616,6 +616,16 @@ function jp_string_esc(p: JP) -> string { if p.i < p.n { let e = p.s[p.i] if e == 'n' { out += "\n" } + else if e == 't' { out += "\t" } + else if e == 'r' { out += "\r" } + else if e == 'b' or e == 'f' or e == 'u' { + var code = 8 + if e == 'f' { code = 12 } + if e == 'u' { code = jp_code(p) } # leaves p.i on the escape's last hex digit + let u = jp_utf8(code) + out += u + free(u) + } else { out += p.s[p.i..p.i + 1] } # \" \\ \/ -> the literal char p.i += 1 } @@ -627,6 +637,53 @@ function jp_string_esc(p: JP) -> string { return out } +# \uXXXX, p.i on the 'u': the code point, a surrogate pair joined (\ud83d\ude00 is one emoji); a +# lone or broken surrogate, or bad hex, is U+FFFD. Python's json.dump writes non-ASCII this way. +function jp_code(p: JP) -> int { + let hi = jp_hex4(p, p.i + 1) + if hi < 0 { return 65533 } # p.i stays on the 'u': what follows is read as text + p.i += 4 + if hi >= 55296 and hi <= 56319 { # a high surrogate: a low one must follow + if p.i + 6 < p.n and p.s[p.i + 1] == '\\' and p.s[p.i + 2] == 'u' { + let lo = jp_hex4(p, p.i + 3) + if lo >= 56320 and lo <= 57343 { + p.i += 6 + return 65536 + ((hi - 55296) << 10) + (lo - 56320) + } + } + return 65533 + } + if hi >= 56320 and hi <= 57343 { return 65533 } # a low surrogate on its own + return hi +} +# four hex digits at `at`, or -1 +function jp_hex4(p: JP, at: int) -> int { + if at + 4 > p.n { return -1 } + var v = 0 + for k in 0 .. 4 { + let c = p.s[at + k] + var d = -1 + if c >= '0' and c <= '9' { d = c - 48 } + else if c >= 'a' and c <= 'f' { d = c - 87 } + else if c >= 'A' and c <= 'F' { d = c - 55 } + if d < 0 { return -1 } + v = v * 16 + d + } + return v +} +# one code point as a UTF-8 string +function jp_utf8(c: int) -> string { + let out = bytes(5) + var n = 0 + if c < 128 { out[0] = c; n = 1 } + else if c < 2048 { out[0] = 192 | (c >> 6); out[1] = 128 | (c & 63); n = 2 } + else if c < 65536 { out[0] = 224 | (c >> 12); out[1] = 128 | ((c >> 6) & 63); out[2] = 128 | (c & 63); n = 3 } + else { out[0] = 240 | (c >> 18); out[1] = 128 | ((c >> 12) & 63); out[2] = 128 | ((c >> 6) & 63); out[3] = 128 | (c & 63); n = 4 } + out[n] = 0 + let t: string = out + return t +} + # read a number; a '.' makes it a fixed node, otherwise an int node. function jp_number(p: JP) -> Val { var neg = 0 diff --git a/tools/ludic-cli/test.ludic b/tools/ludic-cli/test.ludic index 772b7c35..acf64efc 100644 --- a/tools/ludic-cli/test.ludic +++ b/tools/ludic-cli/test.ludic @@ -1213,6 +1213,7 @@ function cmd_dev_test() -> int { feat_case("lang/value_list_regrow", "", "bad 0 xs 2", "value_list_regrow.ludic (a model list filled in place that shrinks and grows back keeps its cut-off items as spares: nothing kept, the fence lets it through)") feat_case("lang/json_free_edited", "", "names Sam|Ada grew 32", "json_free_edited.ludic (Json.free_all after a migration put literals in and a loader set a string: only the parser's strings are freed, a kept copy reads on, a thousand loads keep nothing)") feat_case("lang/json_saves", "", "1 2 3 saves grew 0", "json_saves.ludic (Json.write_file saves through a kept buffer: the text Json.encode gives, 1000 saves keep nothing)") + feat_case("lang/json_unicode", "", "json unicode 8 of 8", "json_unicode.ludic (Json.parse decodes \\uXXXX to UTF-8, a surrogate pair joined, a lone one or bad hex as U+FFFD, and \\t \\r \\b \\f)") feat_case("lang/alloc_fence", "", "frames 300 kept 0 bad 0", "alloc_fence.ludic (25.1: a frame that makes only what it frees or reuses keeps nothing, and the fence passes it)") feat_case("lang/alloc_fence_declared", "", "bad 0 kept 0", "alloc_fence_declared.ludic (25.1: @alloc_ok on a function, a statement and a generic's statement is declared at run time; the failing fence lets it through)") alloc_fence_leak_case()