diff --git a/lib/system/widestrs.nim b/lib/system/widestrs.nim index e7b7f3972..77310b289 100644 --- a/lib/system/widestrs.nim +++ b/lib/system/widestrs.nim @@ -114,41 +114,47 @@ proc newWideCString*(s: cstring): WideCString = proc newWideCString*(s: string): WideCString = result = newWideCString(s, s.len) -proc `$`*(w: WideCString, estimate: int): string = +proc `$`*(w: WideCString, estimate: int, replacement: int = 0xFFFD): string = result = newStringOfCap(estimate + estimate shr 2) var i = 0 while w[i].int16 != 0'i16: - var ch = w[i].int + var ch = int(cast[uint16](w[i])) inc i - if ch >=% UNI_SUR_HIGH_START and ch <=% UNI_SUR_HIGH_END: + if ch >= UNI_SUR_HIGH_START and ch <= UNI_SUR_HIGH_END: # If the 16 bits following the high surrogate are in the source buffer... - let ch2 = w[i].int + let ch2 = int(cast[uint16](w[i])) + # If it's a low surrogate, convert to UTF32: - if ch2 >=% UNI_SUR_LOW_START and ch2 <=% UNI_SUR_LOW_END: - ch = ((ch -% UNI_SUR_HIGH_START) shr halfShift) +% - (ch2 -% UNI_SUR_LOW_START) +% halfBase + if ch2 >= UNI_SUR_LOW_START and ch2 <= UNI_SUR_LOW_END: + ch = (((ch and halfMask) shl halfShift) + (ch2 and halfMask)) + halfBase inc i - - if ch <=% 127: + else: + #invalid UTF-16 + ch = replacement + elif ch >= UNI_SUR_LOW_START and ch <= UNI_SUR_LOW_END: + #invalid UTF-16 + ch = replacement + + if ch < 0x80: result.add chr(ch) - elif ch <=% 0x07FF: - result.add chr((ch shr 6) or 0b110_00000) - result.add chr((ch and ones(6)) or 0b10_000000) - elif ch <=% 0xFFFF: - result.add chr(ch shr 12 or 0b1110_0000) - result.add chr(ch shr 6 and ones(6) or 0b10_0000_00) - result.add chr(ch and ones(6) or 0b10_0000_00) - elif ch <=% 0x0010FFFF: - result.add chr(ch shr 18 or 0b1111_0000) - result.add chr(ch shr 12 and ones(6) or 0b10_0000_00) - result.add chr(ch shr 6 and ones(6) or 0b10_0000_00) - result.add chr(ch and ones(6) or 0b10_0000_00) + elif ch < 0x800: + result.add chr((ch shr 6) or 0xc0) + result.add chr((ch and 0x3f) or 0x80) + elif ch < 0x10000: + result.add chr((ch shr 12) or 0xe0) + result.add chr(((ch shr 6) and 0x3f) or 0x80) + result.add chr((ch and 0x3f) or 0x80) + elif ch <= 0x10FFFF: + result.add chr((ch shr 18) or 0xf0) + result.add chr(((ch shr 12) and 0x3f) or 0x80) + result.add chr(((ch shr 6) and 0x3f) or 0x80) + result.add chr((ch and 0x3f) or 0x80) else: - # replacement char: + # replacement char(in case user give very large number): result.add chr(0xFFFD shr 12 or 0b1110_0000) result.add chr(0xFFFD shr 6 and ones(6) or 0b10_0000_00) result.add chr(0xFFFD and ones(6) or 0b10_0000_00) - + proc `$`*(s: WideCString): string = result = s $ 80 diff --git a/tests/stdlib/twchartoutf8.nim b/tests/stdlib/twchartoutf8.nim new file mode 100644 index 000000000..9838bbfe7 --- /dev/null +++ b/tests/stdlib/twchartoutf8.nim @@ -0,0 +1,106 @@ +#assume WideCharToMultiByte always produce correct result +#windows only + +when not defined(windows): + {.error: "windows only".} + +{.push gcsafe.} + +const CP_UTF8 = 65001'i32 + +type + LPBOOL = ptr int32 + LPWCSTR = ptr uint16 + +proc WideCharToMultiByte*(CodePage: int32, dwFlags: int32, + lpWideCharStr: LPWCSTR, cchWideChar: int32, + lpMultiByteStr: cstring, cchMultiByte: int32, + lpDefaultChar: cstring, lpUsedDefaultChar: LPBOOL): int32{. + stdcall, dynlib: "kernel32", importc: "WideCharToMultiByte".} + +{.pop.} + +proc convertToUTF8(wc: WideCString, wclen: int32): string = + let size = WideCharToMultiByte(CP_UTF8, 0'i32, cast[LPWCSTR](addr(wc[0])), wclen, + cstring(nil), 0'i32, cstring(nil), LPBOOL(nil)) + result = newString(size) + let res = WideCharToMultiByte(CP_UTF8, 0'i32, cast[LPWCSTR](addr(wc[0])), wclen, + cstring(result), size, cstring(nil), LPBOOL(nil)) + result[size] = chr(0) + assert size == res + +proc testCP(wc: WideCString, lo, hi: int) = + var x = 0 + let chunk = 1024 + for i in lo..hi: + wc[x] = cast[Utf16Char](i) + if (x >= chunk) or (i >= hi): + wc[x] = Utf16Char(0) + var a = convertToUTF8(wc, int32(x)) + var b = wc $ chunk + assert a == b + x = 0 + inc x + +proc testCP2(wc: WideCString, lo, hi: int) = + assert((lo >= 0x10000) and (hi <= 0x10FFFF)) + var x = 0 + let chunk = 1024 + for i in lo..hi: + let ch = i - 0x10000 + let W1 = 0xD800 or (ch shr 10) + let W2 = 0xDC00 or (0x3FF and ch) + wc[x] = cast[Utf16Char](W1) + wc[x+1] = cast[Utf16Char](W2) + inc(x, 2) + + if (x >= chunk) or (i >= hi): + wc[x] = Utf16Char(0) + var a = convertToUTF8(wc, int32(x)) + var b = wc $ chunk + assert a == b + x = 0 + +#RFC-2781 "UTF-16, an encoding of ISO 10646" + +var wc: WideCString +unsafeNew(wc, 1024 * 4 + 2) + +#U+0000 to U+D7FF +#skip the U+0000 +wc.testCP(1, 0xD7FF) + +#U+E000 to U+FFFF +wc.testCP(0xE000, 0xFFFF) + +#U+10000 to U+10FFFF +wc.testCP2(0x10000, 0x10FFFF) + +#invalid UTF-16 +const + b = "\xEF\xBF\xBD" + c = "\xEF\xBF\xBF" + +wc[0] = cast[Utf16Char](0xDC00) +wc[1] = Utf16Char(0) +var a = $wc +assert a == b + +wc[0] = cast[Utf16Char](0xFFFF) +wc[1] = cast[Utf16Char](0xDC00) +wc[2] = Utf16Char(0) +a = $wc +assert a == c & b + +wc[0] = cast[Utf16Char](0xD800) +wc[1] = Utf16Char(0) +a = $wc +assert a == b + +wc[0] = cast[Utf16Char](0xD800) +wc[1] = cast[Utf16Char](0xFFFF) +wc[2] = Utf16Char(0) +a = $wc +assert a == b & c + +echo "OK" \ No newline at end of file