Don't escape multibyte characters (#7570)
This commit is contained in:
parent
72dfe176f5
commit
8caf257607
4 changed files with 60 additions and 42 deletions
|
|
@ -172,32 +172,6 @@ proc put(g: var TSrcGen, kind: TTokType, s: string) =
|
||||||
else:
|
else:
|
||||||
g.pendingWhitespace = s.len
|
g.pendingWhitespace = s.len
|
||||||
|
|
||||||
proc addNimChar(dst: var string; c: char): void =
|
|
||||||
case c
|
|
||||||
of '\0': dst.add "\\x00" # not "\\0" to avoid ambiguous cases like "\\012".
|
|
||||||
of '\a': dst.add "\\a" # \x07
|
|
||||||
of '\b': dst.add "\\b" # \x08
|
|
||||||
of '\t': dst.add "\\t" # \x09
|
|
||||||
of '\L': dst.add "\\L" # \x0A
|
|
||||||
of '\v': dst.add "\\v" # \x0B
|
|
||||||
of '\f': dst.add "\\f" # \x0C
|
|
||||||
of '\c': dst.add "\\c" # \x0D
|
|
||||||
of '\e': dst.add "\\e" # \x1B
|
|
||||||
of '\x01'..'\x06', '\x0E'..'\x1A', '\x1C'..'\x1F', '\x80'..'\xFF':
|
|
||||||
dst.add "\\x"
|
|
||||||
dst.add strutils.toHex(ord(c), 2)
|
|
||||||
of '\'', '\"', '\\':
|
|
||||||
dst.add '\\'
|
|
||||||
dst.add c
|
|
||||||
else:
|
|
||||||
dst.add c
|
|
||||||
|
|
||||||
proc makeNimString(s: string): string =
|
|
||||||
result = "\""
|
|
||||||
for c in s:
|
|
||||||
result.addNimChar c
|
|
||||||
add(result, '\"')
|
|
||||||
|
|
||||||
proc putComment(g: var TSrcGen, s: string) =
|
proc putComment(g: var TSrcGen, s: string) =
|
||||||
if s.isNil: return
|
if s.isNil: return
|
||||||
var i = 0
|
var i = 0
|
||||||
|
|
@ -365,10 +339,13 @@ proc atom(g: TSrcGen; n: PNode): string =
|
||||||
of nkEmpty: result = ""
|
of nkEmpty: result = ""
|
||||||
of nkIdent: result = n.ident.s
|
of nkIdent: result = n.ident.s
|
||||||
of nkSym: result = n.sym.name.s
|
of nkSym: result = n.sym.name.s
|
||||||
of nkStrLit: result = makeNimString(n.strVal)
|
of nkStrLit: result = ""; result.addQuoted(n.strVal)
|
||||||
of nkRStrLit: result = "r\"" & replace(n.strVal, "\"", "\"\"") & '\"'
|
of nkRStrLit: result = "r\"" & replace(n.strVal, "\"", "\"\"") & '\"'
|
||||||
of nkTripleStrLit: result = "\"\"\"" & n.strVal & "\"\"\""
|
of nkTripleStrLit: result = "\"\"\"" & n.strVal & "\"\"\""
|
||||||
of nkCharLit: result = "\'"; result.addNimChar chr(int(n.intVal)); result.add '\''
|
of nkCharLit:
|
||||||
|
result = "\'"
|
||||||
|
result.addEscapedChar(chr(int(n.intVal)));
|
||||||
|
result.add '\''
|
||||||
of nkIntLit: result = litAux(g, n, n.intVal, 4)
|
of nkIntLit: result = litAux(g, n, n.intVal, 4)
|
||||||
of nkInt8Lit: result = litAux(g, n, n.intVal, 1) & "\'i8"
|
of nkInt8Lit: result = litAux(g, n, n.intVal, 1) & "\'i8"
|
||||||
of nkInt16Lit: result = litAux(g, n, n.intVal, 2) & "\'i16"
|
of nkInt16Lit: result = litAux(g, n, n.intVal, 2) & "\'i16"
|
||||||
|
|
|
||||||
|
|
@ -1818,20 +1818,29 @@ proc insertSep*(s: string, sep = '_', digits = 3): string {.noSideEffect,
|
||||||
dec(L)
|
dec(L)
|
||||||
|
|
||||||
proc escape*(s: string, prefix = "\"", suffix = "\""): string {.noSideEffect,
|
proc escape*(s: string, prefix = "\"", suffix = "\""): string {.noSideEffect,
|
||||||
rtl, extern: "nsuEscape".} =
|
rtl, extern: "nsuEscape", deprecated.} =
|
||||||
## Escapes a string `s`. See `system.addEscapedChar <system.html#addEscapedChar>`_
|
## Escapes a string `s`. See `system.addEscapedChar <system.html#addEscapedChar>`_
|
||||||
## for the escaping scheme.
|
## for the escaping scheme.
|
||||||
##
|
##
|
||||||
## The resulting string is prefixed with `prefix` and suffixed with `suffix`.
|
## The resulting string is prefixed with `prefix` and suffixed with `suffix`.
|
||||||
## Both may be empty strings.
|
## Both may be empty strings.
|
||||||
|
##
|
||||||
|
## **Warning:** This procedure is deprecated because it's to easy to missuse.
|
||||||
result = newStringOfCap(s.len + s.len shr 2)
|
result = newStringOfCap(s.len + s.len shr 2)
|
||||||
result.add(prefix)
|
result.add(prefix)
|
||||||
for c in items(s):
|
for c in items(s):
|
||||||
result.addEscapedChar(c)
|
case c
|
||||||
|
of '\0'..'\31', '\127'..'\255':
|
||||||
|
add(result, "\\x")
|
||||||
|
add(result, toHex(ord(c), 2))
|
||||||
|
of '\\': add(result, "\\\\")
|
||||||
|
of '\'': add(result, "\\'")
|
||||||
|
of '\"': add(result, "\\\"")
|
||||||
|
else: add(result, c)
|
||||||
add(result, suffix)
|
add(result, suffix)
|
||||||
|
|
||||||
proc unescape*(s: string, prefix = "\"", suffix = "\""): string {.noSideEffect,
|
proc unescape*(s: string, prefix = "\"", suffix = "\""): string {.noSideEffect,
|
||||||
rtl, extern: "nsuUnescape".} =
|
rtl, extern: "nsuUnescape", deprecated.} =
|
||||||
## Unescapes a string `s`.
|
## Unescapes a string `s`.
|
||||||
##
|
##
|
||||||
## This complements `escape <#escape>`_ as it performs the opposite
|
## This complements `escape <#escape>`_ as it performs the opposite
|
||||||
|
|
@ -1839,6 +1848,8 @@ proc unescape*(s: string, prefix = "\"", suffix = "\""): string {.noSideEffect,
|
||||||
##
|
##
|
||||||
## If `s` does not begin with ``prefix`` and end with ``suffix`` a
|
## If `s` does not begin with ``prefix`` and end with ``suffix`` a
|
||||||
## ValueError exception will be raised.
|
## ValueError exception will be raised.
|
||||||
|
##
|
||||||
|
## **Warning:** This procedure is deprecated because it's to easy to missuse.
|
||||||
result = newStringOfCap(s.len)
|
result = newStringOfCap(s.len)
|
||||||
var i = prefix.len
|
var i = prefix.len
|
||||||
if not s.startsWith(prefix):
|
if not s.startsWith(prefix):
|
||||||
|
|
|
||||||
|
|
@ -3933,29 +3933,48 @@ proc addEscapedChar*(s: var string, c: char) {.noSideEffect, inline.} =
|
||||||
## * replaces any ``\`` by ``\\``
|
## * replaces any ``\`` by ``\\``
|
||||||
## * replaces any ``'`` by ``\'``
|
## * replaces any ``'`` by ``\'``
|
||||||
## * replaces any ``"`` by ``\"``
|
## * replaces any ``"`` by ``\"``
|
||||||
## * replaces any other character in the set ``{'\0'..'\31', '\127'..'\255'}``
|
## * replaces any ``\a`` by ``\\a``
|
||||||
|
## * replaces any ``\b`` by ``\\b``
|
||||||
|
## * replaces any ``\t`` by ``\\t``
|
||||||
|
## * replaces any ``\n`` by ``\\n``
|
||||||
|
## * replaces any ``\v`` by ``\\v``
|
||||||
|
## * replaces any ``\f`` by ``\\f``
|
||||||
|
## * replaces any ``\c`` by ``\\c``
|
||||||
|
## * replaces any ``\e`` by ``\\e``
|
||||||
|
## * replaces any other character not in the set ``{'\21..'\126'}
|
||||||
## by ``\xHH`` where ``HH`` is its hexadecimal value.
|
## by ``\xHH`` where ``HH`` is its hexadecimal value.
|
||||||
##
|
##
|
||||||
## The procedure has been designed so that its output is usable for many
|
## The procedure has been designed so that its output is usable for many
|
||||||
## different common syntaxes.
|
## different common syntaxes.
|
||||||
## **Note**: This is not correct for producing Ansi C code!
|
## **Note**: This is not correct for producing Ansi C code!
|
||||||
case c
|
case c
|
||||||
of '\0'..'\31', '\127'..'\255':
|
of '\a': s.add "\\a" # \x07
|
||||||
add(s, "\\x")
|
of '\b': s.add "\\b" # \x08
|
||||||
|
of '\t': s.add "\\t" # \x09
|
||||||
|
of '\n': s.add "\\n" # \x0A
|
||||||
|
of '\v': s.add "\\v" # \x0B
|
||||||
|
of '\f': s.add "\\f" # \x0C
|
||||||
|
of '\c': s.add "\\c" # \x0D
|
||||||
|
of '\e': s.add "\\e" # \x1B
|
||||||
|
of '\\': s.add("\\\\")
|
||||||
|
of '\'': s.add("\\'")
|
||||||
|
of '\"': s.add("\\\"")
|
||||||
|
of {'\32'..'\126'} - {'\\', '\'', '\"'}: s.add(c)
|
||||||
|
else:
|
||||||
|
s.add("\\x")
|
||||||
const HexChars = "0123456789ABCDEF"
|
const HexChars = "0123456789ABCDEF"
|
||||||
let n = ord(c)
|
let n = ord(c)
|
||||||
s.add(HexChars[int((n and 0xF0) shr 4)])
|
s.add(HexChars[int((n and 0xF0) shr 4)])
|
||||||
s.add(HexChars[int(n and 0xF)])
|
s.add(HexChars[int(n and 0xF)])
|
||||||
of '\\': add(s, "\\\\")
|
|
||||||
of '\'': add(s, "\\'")
|
|
||||||
of '\"': add(s, "\\\"")
|
|
||||||
else: add(s, c)
|
|
||||||
|
|
||||||
proc addQuoted*[T](s: var string, x: T) =
|
proc addQuoted*[T](s: var string, x: T) =
|
||||||
## Appends `x` to string `s` in place, applying quoting and escaping
|
## Appends `x` to string `s` in place, applying quoting and escaping
|
||||||
## if `x` is a string or char. See
|
## if `x` is a string or char. See
|
||||||
## `addEscapedChar <system.html#addEscapedChar>`_
|
## `addEscapedChar <system.html#addEscapedChar>`_
|
||||||
## for the escaping scheme.
|
## for the escaping scheme. When `x` is a string, characters in the
|
||||||
|
## range ``{\128..\255}`` are never escaped so that multibyte UTF-8
|
||||||
|
## characters are untouched (note that this behavior is different from
|
||||||
|
## ``addEscapedChar``).
|
||||||
##
|
##
|
||||||
## The Nim standard library uses this function on the elements of
|
## The Nim standard library uses this function on the elements of
|
||||||
## collections when producing a string representation of a collection.
|
## collections when producing a string representation of a collection.
|
||||||
|
|
@ -3974,7 +3993,12 @@ proc addQuoted*[T](s: var string, x: T) =
|
||||||
when T is string:
|
when T is string:
|
||||||
s.add("\"")
|
s.add("\"")
|
||||||
for c in x:
|
for c in x:
|
||||||
s.addEscapedChar(c)
|
# Only ASCII chars are escaped to avoid butchering
|
||||||
|
# multibyte UTF-8 characters.
|
||||||
|
if c <= 127.char:
|
||||||
|
s.addEscapedChar(c)
|
||||||
|
else:
|
||||||
|
s.add c
|
||||||
s.add("\"")
|
s.add("\"")
|
||||||
elif T is char:
|
elif T is char:
|
||||||
s.add("'")
|
s.add("'")
|
||||||
|
|
|
||||||
|
|
@ -85,14 +85,20 @@ block:
|
||||||
s.addQuoted('\0')
|
s.addQuoted('\0')
|
||||||
s.addQuoted('\31')
|
s.addQuoted('\31')
|
||||||
s.addQuoted('\127')
|
s.addQuoted('\127')
|
||||||
s.addQuoted('\255')
|
doAssert s == "'\\x00''\\x1F''\\x7F'"
|
||||||
doAssert s == "'\\x00''\\x1F''\\x7F''\\xFF'"
|
|
||||||
block:
|
block:
|
||||||
var s = ""
|
var s = ""
|
||||||
s.addQuoted('\\')
|
s.addQuoted('\\')
|
||||||
s.addQuoted('\'')
|
s.addQuoted('\'')
|
||||||
s.addQuoted('\"')
|
s.addQuoted('\"')
|
||||||
doAssert s == """'\\''\'''\"'"""
|
doAssert s == """'\\''\'''\"'"""
|
||||||
|
block:
|
||||||
|
var s = ""
|
||||||
|
s.addQuoted("å")
|
||||||
|
s.addQuoted("ä")
|
||||||
|
s.addQuoted("ö")
|
||||||
|
s.addEscapedChar('\xFF')
|
||||||
|
doAssert s == """"å""ä""ö"\xFF"""
|
||||||
|
|
||||||
# Test customized element representation
|
# Test customized element representation
|
||||||
type CustomString = object
|
type CustomString = object
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue