Json.parse decodes \uXXXX to UTF-8 - a surrogate pair joined, a lone surrogate or bad hex as U+FFFD - and \t \r \b \f, where it dropped the backslash and kept the hex as text; examples/lang/json_unicode checks it byte by byte

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-29 21:07:10 +03:00
parent 313f95cd1f
commit 4376ddf0ec
4 changed files with 97 additions and 0 deletions

5
changes/json-unicode.md Normal file
View file

@ -0,0 +1,5 @@
bump: patch
type: fix
**`Json.parse` decodes `\uXXXX`.** An escaped code point is UTF-8 now - a surrogate pair joined into one, a
lone surrogate or bad hex as U+FFFD - where the backslash was dropped and the hex kept as text ("iu015f"),
and `\t`, `\r`, `\b` and `\f` are the characters they name. Python's `json.dump` writes non-ASCII this way.

View file

@ -0,0 +1,34 @@
# json_unicode.ludic - Json.parse decodes \uXXXX to UTF-8 (a surrogate pair joined into one code point,
# a lone surrogate or bad hex as U+FFFD) and \t \r \b \f, as Python's json.dump writes non-ASCII.
# Checked byte by byte, so this file stays ASCII. Prints `json unicode 8 of 8`.
program JsonUnicode {
# the string value of {"s": <json>}
function str_of(json: string) -> string {
let v = Json.parse("{\"s\": " + json + "}")
return Value.as_str(Value.get(v, "s"))
}
function bytes_are(s: string, want: []int) -> bool {
if len(s) != len(want) { return false }
for i in 0 .. len(want) {
if (s[i] & 255) != want[i] { return false }
}
return true
}
function check(name: string, got: string, want: []int, ok: int) -> int {
if bytes_are(got, want) { return ok + 1 }
print(`json unicode: {name} wrong ({len(got)} bytes)`)
return ok
}
entry {
var ok = 0
ok = check("two bytes", str_of("\"\\u015f\""), [197, 159], ok)
ok = check("ascii around", str_of("\"i\\u0131x\""), [105, 196, 177, 120], ok)
ok = check("three bytes", str_of("\"\\u20ac\""), [226, 130, 172], ok)
ok = check("surrogate pair", str_of("\"\\ud83d\\ude00\""), [240, 159, 152, 128], ok)
ok = check("lone high", str_of("\"\\ud83dx\""), [239, 191, 189, 120], ok)
ok = check("lone low", str_of("\"\\ude00\""), [239, 191, 189], ok)
ok = check("bad hex", str_of("\"\\uZZ12\""), [239, 191, 189, 90, 90, 49, 50], ok)
ok = check("controls", str_of("\"\\t\\r\\b\\f\\n\""), [9, 13, 8, 12, 10], ok)
print(`json unicode {ok} of 8`)
}
}

View file

@ -616,6 +616,16 @@ function jp_string_esc(p: JP) -> string {
if p.i < p.n {
let e = p.s[p.i]
if e == 'n' { out += "\n" }
else if e == 't' { out += "\t" }
else if e == 'r' { out += "\r" }
else if e == 'b' or e == 'f' or e == 'u' {
var code = 8
if e == 'f' { code = 12 }
if e == 'u' { code = jp_code(p) } # leaves p.i on the escape's last hex digit
let u = jp_utf8(code)
out += u
free(u)
}
else { out += p.s[p.i..p.i + 1] } # \" \\ \/ -> the literal char
p.i += 1
}
@ -627,6 +637,53 @@ function jp_string_esc(p: JP) -> string {
return out
}
# \uXXXX, p.i on the 'u': the code point, a surrogate pair joined (\ud83d\ude00 is one emoji); a
# lone or broken surrogate, or bad hex, is U+FFFD. Python's json.dump writes non-ASCII this way.
function jp_code(p: JP) -> int {
let hi = jp_hex4(p, p.i + 1)
if hi < 0 { return 65533 } # p.i stays on the 'u': what follows is read as text
p.i += 4
if hi >= 55296 and hi <= 56319 { # a high surrogate: a low one must follow
if p.i + 6 < p.n and p.s[p.i + 1] == '\\' and p.s[p.i + 2] == 'u' {
let lo = jp_hex4(p, p.i + 3)
if lo >= 56320 and lo <= 57343 {
p.i += 6
return 65536 + ((hi - 55296) << 10) + (lo - 56320)
}
}
return 65533
}
if hi >= 56320 and hi <= 57343 { return 65533 } # a low surrogate on its own
return hi
}
# four hex digits at `at`, or -1
function jp_hex4(p: JP, at: int) -> int {
if at + 4 > p.n { return -1 }
var v = 0
for k in 0 .. 4 {
let c = p.s[at + k]
var d = -1
if c >= '0' and c <= '9' { d = c - 48 }
else if c >= 'a' and c <= 'f' { d = c - 87 }
else if c >= 'A' and c <= 'F' { d = c - 55 }
if d < 0 { return -1 }
v = v * 16 + d
}
return v
}
# one code point as a UTF-8 string
function jp_utf8(c: int) -> string {
let out = bytes(5)
var n = 0
if c < 128 { out[0] = c; n = 1 }
else if c < 2048 { out[0] = 192 | (c >> 6); out[1] = 128 | (c & 63); n = 2 }
else if c < 65536 { out[0] = 224 | (c >> 12); out[1] = 128 | ((c >> 6) & 63); out[2] = 128 | (c & 63); n = 3 }
else { out[0] = 240 | (c >> 18); out[1] = 128 | ((c >> 12) & 63); out[2] = 128 | ((c >> 6) & 63); out[3] = 128 | (c & 63); n = 4 }
out[n] = 0
let t: string = out
return t
}
# read a number; a '.' makes it a fixed node, otherwise an int node.
function jp_number(p: JP) -> Val {
var neg = 0

View file

@ -1213,6 +1213,7 @@ function cmd_dev_test() -> int {
feat_case("lang/value_list_regrow", "", "bad 0 xs 2", "value_list_regrow.ludic (a model list filled in place that shrinks and grows back keeps its cut-off items as spares: nothing kept, the fence lets it through)")
feat_case("lang/json_free_edited", "", "names Sam|Ada grew 32", "json_free_edited.ludic (Json.free_all after a migration put literals in and a loader set a string: only the parser's strings are freed, a kept copy reads on, a thousand loads keep nothing)")
feat_case("lang/json_saves", "", "1 2 3 saves grew 0", "json_saves.ludic (Json.write_file saves through a kept buffer: the text Json.encode gives, 1000 saves keep nothing)")
feat_case("lang/json_unicode", "", "json unicode 8 of 8", "json_unicode.ludic (Json.parse decodes \\uXXXX to UTF-8, a surrogate pair joined, a lone one or bad hex as U+FFFD, and \\t \\r \\b \\f)")
feat_case("lang/alloc_fence", "", "frames 300 kept 0 bad 0", "alloc_fence.ludic (25.1: a frame that makes only what it frees or reuses keeps nothing, and the fence passes it)")
feat_case("lang/alloc_fence_declared", "", "bad 0 kept 0", "alloc_fence_declared.ludic (25.1: @alloc_ok on a function, a statement and a generic's statement is declared at run time; the failing fence lets it through)")
alloc_fence_leak_case()