Json.parse decodes \uXXXX to UTF-8 - a surrogate pair joined, a lone surrogate or bad hex as U+FFFD - and \t \r \b \f, where it dropped the backslash and kept the hex as text; examples/lang/json_unicode checks it byte by byte
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
313f95cd1f
commit
4376ddf0ec
4 changed files with 97 additions and 0 deletions
5
changes/json-unicode.md
Normal file
5
changes/json-unicode.md
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
bump: patch
|
||||
type: fix
|
||||
**`Json.parse` decodes `\uXXXX`.** An escaped code point is UTF-8 now - a surrogate pair joined into one, a
|
||||
lone surrogate or bad hex as U+FFFD - where the backslash was dropped and the hex kept as text ("iu015f"),
|
||||
and `\t`, `\r`, `\b` and `\f` are the characters they name. Python's `json.dump` writes non-ASCII this way.
|
||||
34
examples/lang/json_unicode.ludic
Normal file
34
examples/lang/json_unicode.ludic
Normal file
|
|
@ -0,0 +1,34 @@
|
|||
# json_unicode.ludic - Json.parse decodes \uXXXX to UTF-8 (a surrogate pair joined into one code point,
|
||||
# a lone surrogate or bad hex as U+FFFD) and \t \r \b \f, as Python's json.dump writes non-ASCII.
|
||||
# Checked byte by byte, so this file stays ASCII. Prints `json unicode 8 of 8`.
|
||||
program JsonUnicode {
|
||||
# the string value of {"s": <json>}
|
||||
function str_of(json: string) -> string {
|
||||
let v = Json.parse("{\"s\": " + json + "}")
|
||||
return Value.as_str(Value.get(v, "s"))
|
||||
}
|
||||
function bytes_are(s: string, want: []int) -> bool {
|
||||
if len(s) != len(want) { return false }
|
||||
for i in 0 .. len(want) {
|
||||
if (s[i] & 255) != want[i] { return false }
|
||||
}
|
||||
return true
|
||||
}
|
||||
function check(name: string, got: string, want: []int, ok: int) -> int {
|
||||
if bytes_are(got, want) { return ok + 1 }
|
||||
print(`json unicode: {name} wrong ({len(got)} bytes)`)
|
||||
return ok
|
||||
}
|
||||
entry {
|
||||
var ok = 0
|
||||
ok = check("two bytes", str_of("\"\\u015f\""), [197, 159], ok)
|
||||
ok = check("ascii around", str_of("\"i\\u0131x\""), [105, 196, 177, 120], ok)
|
||||
ok = check("three bytes", str_of("\"\\u20ac\""), [226, 130, 172], ok)
|
||||
ok = check("surrogate pair", str_of("\"\\ud83d\\ude00\""), [240, 159, 152, 128], ok)
|
||||
ok = check("lone high", str_of("\"\\ud83dx\""), [239, 191, 189, 120], ok)
|
||||
ok = check("lone low", str_of("\"\\ude00\""), [239, 191, 189], ok)
|
||||
ok = check("bad hex", str_of("\"\\uZZ12\""), [239, 191, 189, 90, 90, 49, 50], ok)
|
||||
ok = check("controls", str_of("\"\\t\\r\\b\\f\\n\""), [9, 13, 8, 12, 10], ok)
|
||||
print(`json unicode {ok} of 8`)
|
||||
}
|
||||
}
|
||||
|
|
@ -616,6 +616,16 @@ function jp_string_esc(p: JP) -> string {
|
|||
if p.i < p.n {
|
||||
let e = p.s[p.i]
|
||||
if e == 'n' { out += "\n" }
|
||||
else if e == 't' { out += "\t" }
|
||||
else if e == 'r' { out += "\r" }
|
||||
else if e == 'b' or e == 'f' or e == 'u' {
|
||||
var code = 8
|
||||
if e == 'f' { code = 12 }
|
||||
if e == 'u' { code = jp_code(p) } # leaves p.i on the escape's last hex digit
|
||||
let u = jp_utf8(code)
|
||||
out += u
|
||||
free(u)
|
||||
}
|
||||
else { out += p.s[p.i..p.i + 1] } # \" \\ \/ -> the literal char
|
||||
p.i += 1
|
||||
}
|
||||
|
|
@ -627,6 +637,53 @@ function jp_string_esc(p: JP) -> string {
|
|||
return out
|
||||
}
|
||||
|
||||
# \uXXXX, p.i on the 'u': the code point, a surrogate pair joined (\ud83d\ude00 is one emoji); a
|
||||
# lone or broken surrogate, or bad hex, is U+FFFD. Python's json.dump writes non-ASCII this way.
|
||||
function jp_code(p: JP) -> int {
|
||||
let hi = jp_hex4(p, p.i + 1)
|
||||
if hi < 0 { return 65533 } # p.i stays on the 'u': what follows is read as text
|
||||
p.i += 4
|
||||
if hi >= 55296 and hi <= 56319 { # a high surrogate: a low one must follow
|
||||
if p.i + 6 < p.n and p.s[p.i + 1] == '\\' and p.s[p.i + 2] == 'u' {
|
||||
let lo = jp_hex4(p, p.i + 3)
|
||||
if lo >= 56320 and lo <= 57343 {
|
||||
p.i += 6
|
||||
return 65536 + ((hi - 55296) << 10) + (lo - 56320)
|
||||
}
|
||||
}
|
||||
return 65533
|
||||
}
|
||||
if hi >= 56320 and hi <= 57343 { return 65533 } # a low surrogate on its own
|
||||
return hi
|
||||
}
|
||||
# four hex digits at `at`, or -1
|
||||
function jp_hex4(p: JP, at: int) -> int {
|
||||
if at + 4 > p.n { return -1 }
|
||||
var v = 0
|
||||
for k in 0 .. 4 {
|
||||
let c = p.s[at + k]
|
||||
var d = -1
|
||||
if c >= '0' and c <= '9' { d = c - 48 }
|
||||
else if c >= 'a' and c <= 'f' { d = c - 87 }
|
||||
else if c >= 'A' and c <= 'F' { d = c - 55 }
|
||||
if d < 0 { return -1 }
|
||||
v = v * 16 + d
|
||||
}
|
||||
return v
|
||||
}
|
||||
# one code point as a UTF-8 string
|
||||
function jp_utf8(c: int) -> string {
|
||||
let out = bytes(5)
|
||||
var n = 0
|
||||
if c < 128 { out[0] = c; n = 1 }
|
||||
else if c < 2048 { out[0] = 192 | (c >> 6); out[1] = 128 | (c & 63); n = 2 }
|
||||
else if c < 65536 { out[0] = 224 | (c >> 12); out[1] = 128 | ((c >> 6) & 63); out[2] = 128 | (c & 63); n = 3 }
|
||||
else { out[0] = 240 | (c >> 18); out[1] = 128 | ((c >> 12) & 63); out[2] = 128 | ((c >> 6) & 63); out[3] = 128 | (c & 63); n = 4 }
|
||||
out[n] = 0
|
||||
let t: string = out
|
||||
return t
|
||||
}
|
||||
|
||||
# read a number; a '.' makes it a fixed node, otherwise an int node.
|
||||
function jp_number(p: JP) -> Val {
|
||||
var neg = 0
|
||||
|
|
|
|||
|
|
@ -1213,6 +1213,7 @@ function cmd_dev_test() -> int {
|
|||
feat_case("lang/value_list_regrow", "", "bad 0 xs 2", "value_list_regrow.ludic (a model list filled in place that shrinks and grows back keeps its cut-off items as spares: nothing kept, the fence lets it through)")
|
||||
feat_case("lang/json_free_edited", "", "names Sam|Ada grew 32", "json_free_edited.ludic (Json.free_all after a migration put literals in and a loader set a string: only the parser's strings are freed, a kept copy reads on, a thousand loads keep nothing)")
|
||||
feat_case("lang/json_saves", "", "1 2 3 saves grew 0", "json_saves.ludic (Json.write_file saves through a kept buffer: the text Json.encode gives, 1000 saves keep nothing)")
|
||||
feat_case("lang/json_unicode", "", "json unicode 8 of 8", "json_unicode.ludic (Json.parse decodes \\uXXXX to UTF-8, a surrogate pair joined, a lone one or bad hex as U+FFFD, and \\t \\r \\b \\f)")
|
||||
feat_case("lang/alloc_fence", "", "frames 300 kept 0 bad 0", "alloc_fence.ludic (25.1: a frame that makes only what it frees or reuses keeps nothing, and the fence passes it)")
|
||||
feat_case("lang/alloc_fence_declared", "", "bad 0 kept 0", "alloc_fence_declared.ludic (25.1: @alloc_ok on a function, a statement and a generic's statement is declared at run time; the failing fence lets it through)")
|
||||
alloc_fence_leak_case()
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue