Make sure the json module decodes UTF16 correctly
Javascript uses UTF-16 as its internal representation of strings, so JSON does so as well. This means that we could have surrogate pairs, with codepoints above 0xFFFF that take 2 ecape codes to decode.
This commit is contained in:
parent
7f4f37eaa2
commit
e5bcd287f8
1 changed files with 25 additions and 5 deletions
|
|
@ -203,6 +203,15 @@ proc handleHexChar(c: char, x: var int): bool =
|
||||||
of 'A'..'F': x = (x shl 4) or (ord(c) - ord('A') + 10)
|
of 'A'..'F': x = (x shl 4) or (ord(c) - ord('A') + 10)
|
||||||
else: result = false # error
|
else: result = false # error
|
||||||
|
|
||||||
|
proc parseEscapedUTF16(buf: cstring, pos: var int): int =
|
||||||
|
result = 0
|
||||||
|
#UTF-16 escape is always 4 bytes.
|
||||||
|
for _ in 0..3:
|
||||||
|
if handleHexChar(buf[pos], result):
|
||||||
|
inc(pos)
|
||||||
|
else:
|
||||||
|
return -1
|
||||||
|
|
||||||
proc parseString(my: var JsonParser): TokKind =
|
proc parseString(my: var JsonParser): TokKind =
|
||||||
result = tkString
|
result = tkString
|
||||||
var pos = my.bufpos + 1
|
var pos = my.bufpos + 1
|
||||||
|
|
@ -238,11 +247,22 @@ proc parseString(my: var JsonParser): TokKind =
|
||||||
inc(pos, 2)
|
inc(pos, 2)
|
||||||
of 'u':
|
of 'u':
|
||||||
inc(pos, 2)
|
inc(pos, 2)
|
||||||
var r: int
|
var r = parseEscapedUTF16(buf, pos)
|
||||||
if handleHexChar(buf[pos], r): inc(pos)
|
if r < 0:
|
||||||
if handleHexChar(buf[pos], r): inc(pos)
|
my.err = errInvalidToken
|
||||||
if handleHexChar(buf[pos], r): inc(pos)
|
break
|
||||||
if handleHexChar(buf[pos], r): inc(pos)
|
# Deal with surrogates
|
||||||
|
if (r and 0xfc00) == 0xd800:
|
||||||
|
if buf[pos] & buf[pos+1] != "\\u":
|
||||||
|
my.err = errInvalidToken
|
||||||
|
break
|
||||||
|
inc(pos, 2)
|
||||||
|
var s = parseEscapedUTF16(buf, pos)
|
||||||
|
if (s and 0xfc00) == 0xdc00 and s > 0:
|
||||||
|
r = 0x10000 + (((r - 0xd800) shl 10) or (s - 0xdc00))
|
||||||
|
else:
|
||||||
|
my.err = errInvalidToken
|
||||||
|
break
|
||||||
add(my.a, toUTF8(Rune(r)))
|
add(my.a, toUTF8(Rune(r)))
|
||||||
else:
|
else:
|
||||||
# don't bother with the error
|
# don't bother with the error
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue