From 493dbc8932326c5f830c0396b528d433a0bdf314 Mon Sep 17 00:00:00 2001 From: jangko Date: Thu, 20 Aug 2015 14:17:46 +0700 Subject: [PATCH 1/3] fixed UTF-16 to UTF-8 conversion in widestrs.nim the source of problem for issue #3228 --- lib/system/widestrs.nim | 44 +++++++++++++++++------------------------ 1 file changed, 18 insertions(+), 26 deletions(-) diff --git a/lib/system/widestrs.nim b/lib/system/widestrs.nim index e7b7f3972..e782e2452 100644 --- a/lib/system/widestrs.nim +++ b/lib/system/widestrs.nim @@ -119,36 +119,28 @@ proc `$`*(w: WideCString, estimate: int): string = var i = 0 while w[i].int16 != 0'i16: - var ch = w[i].int + var ch = uint32(cast[uint16](w[i])) inc i - if ch >=% UNI_SUR_HIGH_START and ch <=% UNI_SUR_HIGH_END: + if ch >= uint32(UNI_SUR_HIGH_START) and ch <= uint32(UNI_SUR_HIGH_END): # If the 16 bits following the high surrogate are in the source buffer... - let ch2 = w[i].int - # If it's a low surrogate, convert to UTF32: - if ch2 >=% UNI_SUR_LOW_START and ch2 <=% UNI_SUR_LOW_END: - ch = ((ch -% UNI_SUR_HIGH_START) shr halfShift) +% - (ch2 -% UNI_SUR_LOW_START) +% halfBase - inc i - - if ch <=% 127: + let ch2 = uint32(cast[uint16](w[i])) + ch = (ch shl halfShift) + ch2 + halfBase + inc i + + if ch < 0x80'u32: result.add chr(ch) - elif ch <=% 0x07FF: - result.add chr((ch shr 6) or 0b110_00000) - result.add chr((ch and ones(6)) or 0b10_000000) - elif ch <=% 0xFFFF: - result.add chr(ch shr 12 or 0b1110_0000) - result.add chr(ch shr 6 and ones(6) or 0b10_0000_00) - result.add chr(ch and ones(6) or 0b10_0000_00) - elif ch <=% 0x0010FFFF: - result.add chr(ch shr 18 or 0b1111_0000) - result.add chr(ch shr 12 and ones(6) or 0b10_0000_00) - result.add chr(ch shr 6 and ones(6) or 0b10_0000_00) - result.add chr(ch and ones(6) or 0b10_0000_00) + elif ch < 0x800'u32: + result.add chr((ch shr 6) or 0xc0) + result.add chr((ch and 0x3f) or 0x80) + elif ch < 0x10000'u32: + result.add chr((ch shr 12) or 0xe0) + result.add chr(((ch shr 6) and 0x3f) or 0x80) + result.add chr((ch and 0x3f) or 0x80) else: - # replacement char: - result.add chr(0xFFFD shr 12 or 0b1110_0000) - result.add chr(0xFFFD shr 6 and ones(6) or 0b10_0000_00) - result.add chr(0xFFFD and ones(6) or 0b10_0000_00) + result.add chr((ch shr 18) or 0xf0) + result.add chr(((ch shr 12) and 0x3f) or 0x80) + result.add chr(((ch shr 6) and 0x3f) or 0x80) + result.add chr((ch and 0x3f) or 0x80) proc `$`*(s: WideCString): string = result = s $ 80 From c103eddc737a48d3de04e64c2086549b7ec33d6d Mon Sep 17 00:00:00 2001 From: jangko Date: Thu, 20 Aug 2015 20:30:14 +0700 Subject: [PATCH 2/3] fixed UTF-16 to UTF-8 conversion in widestrs.nim the source of problem for issue #3228 also add test for entire range of valid UTF-16 --- lib/system/widestrs.nim | 14 +++---- tests/stdlib/twchartoutf8.nim | 78 +++++++++++++++++++++++++++++++++++ 2 files changed, 85 insertions(+), 7 deletions(-) create mode 100644 tests/stdlib/twchartoutf8.nim diff --git a/lib/system/widestrs.nim b/lib/system/widestrs.nim index e782e2452..94ae3e26b 100644 --- a/lib/system/widestrs.nim +++ b/lib/system/widestrs.nim @@ -119,20 +119,20 @@ proc `$`*(w: WideCString, estimate: int): string = var i = 0 while w[i].int16 != 0'i16: - var ch = uint32(cast[uint16](w[i])) + var ch = int(cast[uint16](w[i])) inc i - if ch >= uint32(UNI_SUR_HIGH_START) and ch <= uint32(UNI_SUR_HIGH_END): + if ch >= UNI_SUR_HIGH_START and ch <= UNI_SUR_HIGH_END: # If the 16 bits following the high surrogate are in the source buffer... - let ch2 = uint32(cast[uint16](w[i])) - ch = (ch shl halfShift) + ch2 + halfBase + let ch2 = int(cast[uint16](w[i])) + ch = (((ch and halfMask) shl halfShift) + (ch2 and halfMask)) + halfBase inc i - if ch < 0x80'u32: + if ch < 0x80: result.add chr(ch) - elif ch < 0x800'u32: + elif ch < 0x800: result.add chr((ch shr 6) or 0xc0) result.add chr((ch and 0x3f) or 0x80) - elif ch < 0x10000'u32: + elif ch < 0x10000: result.add chr((ch shr 12) or 0xe0) result.add chr(((ch shr 6) and 0x3f) or 0x80) result.add chr((ch and 0x3f) or 0x80) diff --git a/tests/stdlib/twchartoutf8.nim b/tests/stdlib/twchartoutf8.nim new file mode 100644 index 000000000..806a222b6 --- /dev/null +++ b/tests/stdlib/twchartoutf8.nim @@ -0,0 +1,78 @@ +#assume WideCharToMultiByte always produce correct result +#windows only + +when not defined(windows): + {.error: "windows only".} + +{.push gcsafe.} + +const CP_UTF8 = 65001'i32 + +type + LPBOOL = ptr int32 + LPWCSTR = ptr uint16 + +proc WideCharToMultiByte*(CodePage: int32, dwFlags: int32, + lpWideCharStr: LPWCSTR, cchWideChar: int32, + lpMultiByteStr: cstring, cchMultiByte: int32, + lpDefaultChar: cstring, lpUsedDefaultChar: LPBOOL): int32{. + stdcall, dynlib: "kernel32", importc: "WideCharToMultiByte".} + +{.pop.} + +proc convertToUTF8(wc: WideCString, wclen: int32): string = + let size = WideCharToMultiByte(CP_UTF8, 0'i32, cast[LPWCSTR](addr(wc[0])), wclen, + cstring(nil), 0'i32, cstring(nil), LPBOOL(nil)) + result = newString(size) + let res = WideCharToMultiByte(CP_UTF8, 0'i32, cast[LPWCSTR](addr(wc[0])), wclen, + cstring(result), size, cstring(nil), LPBOOL(nil)) + result[size] = chr(0) + assert size == res + +proc testCP(wc: WideCString, lo, hi: int) = + var x = 0 + let chunk = 1024 + for i in lo..hi: + wc[x] = cast[TUtf16Char](i) + if (x >= chunk) or (i >= hi): + wc[x] = TUtf16Char(0) + var a = convertToUTF8(wc, int32(x)) + var b = wc $ chunk + assert a == b + x = 0 + inc x + +proc testCP2(wc: WideCString, lo, hi: int) = + assert ((lo >=0x10000) and (hi <= 0x10FFFF)) + var x = 0 + let chunk = 1024 + for i in lo..hi: + let ch = i - 0x10000 + let W1 = 0xD800 or (ch shr 10) + let W2 = 0xDC00 or (0x3FF and ch) + wc[x] = cast[TUtf16Char](W1) + wc[x+1] = cast[TUtf16Char](W2) + inc(x, 2) + + if (x >= chunk) or (i >= hi): + wc[x] = TUtf16Char(0) + var a = convertToUTF8(wc, int32(x)) + var b = wc $ chunk + assert a == b + x = 0 + +#RFC-2781 "UTF-16, an encoding of ISO 10646" + +var wc: WideCString +unsafeNew(wc, 1024 * 4 + 2) + +#U+0000 to U+D7FF +#skip the U+0000 +wc.testCP(1, 0xD7FF) + +#U+E000 to U+FFFF +wc.testCP(0xE000, 0xFFFF) + +#U+10000 to U+10FFFF +wc.testCP2(0x10000, 0x10FFFF) +echo "OK" \ No newline at end of file From 7c757599f1c9157a65e8e2238d4b11eedeeb01bf Mon Sep 17 00:00:00 2001 From: jangko Date: Fri, 21 Aug 2015 10:43:31 +0700 Subject: [PATCH 3/3] fixed UTF-16 to UTF-8 conversion in widestrs.nim the source of problem for issue #3228 also add test for entire range of valid UTF-16 and test for invalid UTF-16 sequence --- lib/system/widestrs.nim | 26 +++++++++++++++++----- tests/stdlib/twchartoutf8.nim | 42 +++++++++++++++++++++++++++++------ 2 files changed, 55 insertions(+), 13 deletions(-) diff --git a/lib/system/widestrs.nim b/lib/system/widestrs.nim index 94ae3e26b..77310b289 100644 --- a/lib/system/widestrs.nim +++ b/lib/system/widestrs.nim @@ -114,7 +114,7 @@ proc newWideCString*(s: cstring): WideCString = proc newWideCString*(s: string): WideCString = result = newWideCString(s, s.len) -proc `$`*(w: WideCString, estimate: int): string = +proc `$`*(w: WideCString, estimate: int, replacement: int = 0xFFFD): string = result = newStringOfCap(estimate + estimate shr 2) var i = 0 @@ -124,9 +124,18 @@ proc `$`*(w: WideCString, estimate: int): string = if ch >= UNI_SUR_HIGH_START and ch <= UNI_SUR_HIGH_END: # If the 16 bits following the high surrogate are in the source buffer... let ch2 = int(cast[uint16](w[i])) - ch = (((ch and halfMask) shl halfShift) + (ch2 and halfMask)) + halfBase - inc i - + + # If it's a low surrogate, convert to UTF32: + if ch2 >= UNI_SUR_LOW_START and ch2 <= UNI_SUR_LOW_END: + ch = (((ch and halfMask) shl halfShift) + (ch2 and halfMask)) + halfBase + inc i + else: + #invalid UTF-16 + ch = replacement + elif ch >= UNI_SUR_LOW_START and ch <= UNI_SUR_LOW_END: + #invalid UTF-16 + ch = replacement + if ch < 0x80: result.add chr(ch) elif ch < 0x800: @@ -136,11 +145,16 @@ proc `$`*(w: WideCString, estimate: int): string = result.add chr((ch shr 12) or 0xe0) result.add chr(((ch shr 6) and 0x3f) or 0x80) result.add chr((ch and 0x3f) or 0x80) - else: + elif ch <= 0x10FFFF: result.add chr((ch shr 18) or 0xf0) result.add chr(((ch shr 12) and 0x3f) or 0x80) result.add chr(((ch shr 6) and 0x3f) or 0x80) result.add chr((ch and 0x3f) or 0x80) - + else: + # replacement char(in case user give very large number): + result.add chr(0xFFFD shr 12 or 0b1110_0000) + result.add chr(0xFFFD shr 6 and ones(6) or 0b10_0000_00) + result.add chr(0xFFFD and ones(6) or 0b10_0000_00) + proc `$`*(s: WideCString): string = result = s $ 80 diff --git a/tests/stdlib/twchartoutf8.nim b/tests/stdlib/twchartoutf8.nim index 806a222b6..9838bbfe7 100644 --- a/tests/stdlib/twchartoutf8.nim +++ b/tests/stdlib/twchartoutf8.nim @@ -33,9 +33,9 @@ proc testCP(wc: WideCString, lo, hi: int) = var x = 0 let chunk = 1024 for i in lo..hi: - wc[x] = cast[TUtf16Char](i) + wc[x] = cast[Utf16Char](i) if (x >= chunk) or (i >= hi): - wc[x] = TUtf16Char(0) + wc[x] = Utf16Char(0) var a = convertToUTF8(wc, int32(x)) var b = wc $ chunk assert a == b @@ -43,26 +43,26 @@ proc testCP(wc: WideCString, lo, hi: int) = inc x proc testCP2(wc: WideCString, lo, hi: int) = - assert ((lo >=0x10000) and (hi <= 0x10FFFF)) + assert((lo >= 0x10000) and (hi <= 0x10FFFF)) var x = 0 let chunk = 1024 for i in lo..hi: let ch = i - 0x10000 let W1 = 0xD800 or (ch shr 10) let W2 = 0xDC00 or (0x3FF and ch) - wc[x] = cast[TUtf16Char](W1) - wc[x+1] = cast[TUtf16Char](W2) + wc[x] = cast[Utf16Char](W1) + wc[x+1] = cast[Utf16Char](W2) inc(x, 2) if (x >= chunk) or (i >= hi): - wc[x] = TUtf16Char(0) + wc[x] = Utf16Char(0) var a = convertToUTF8(wc, int32(x)) var b = wc $ chunk assert a == b x = 0 #RFC-2781 "UTF-16, an encoding of ISO 10646" - + var wc: WideCString unsafeNew(wc, 1024 * 4 + 2) @@ -75,4 +75,32 @@ wc.testCP(0xE000, 0xFFFF) #U+10000 to U+10FFFF wc.testCP2(0x10000, 0x10FFFF) + +#invalid UTF-16 +const + b = "\xEF\xBF\xBD" + c = "\xEF\xBF\xBF" + +wc[0] = cast[Utf16Char](0xDC00) +wc[1] = Utf16Char(0) +var a = $wc +assert a == b + +wc[0] = cast[Utf16Char](0xFFFF) +wc[1] = cast[Utf16Char](0xDC00) +wc[2] = Utf16Char(0) +a = $wc +assert a == c & b + +wc[0] = cast[Utf16Char](0xD800) +wc[1] = Utf16Char(0) +a = $wc +assert a == b + +wc[0] = cast[Utf16Char](0xD800) +wc[1] = cast[Utf16Char](0xFFFF) +wc[2] = Utf16Char(0) +a = $wc +assert a == b & c + echo "OK" \ No newline at end of file