make the lexer more forgiving so that nim-regex compiles again
This commit is contained in:
parent
10219584ee
commit
23c11987b4
1 changed files with 17 additions and 11 deletions
|
|
@ -618,7 +618,12 @@ proc getNumber(L: var TLexer, result: var TToken) =
|
||||||
tokenEnd(result, postPos-1)
|
tokenEnd(result, postPos-1)
|
||||||
L.bufpos = postPos
|
L.bufpos = postPos
|
||||||
|
|
||||||
proc handleHexChar(L: var TLexer, xi: var int) =
|
proc handleHexChar(L: var TLexer, xi: var int; position: range[1..4]) =
|
||||||
|
template invalid() =
|
||||||
|
lexMessage(L, errGenerated,
|
||||||
|
"expected a hex digit, but found: " & L.buf[L.bufpos] &
|
||||||
|
"; maybe prepend with 0")
|
||||||
|
|
||||||
case L.buf[L.bufpos]
|
case L.buf[L.bufpos]
|
||||||
of '0'..'9':
|
of '0'..'9':
|
||||||
xi = (xi shl 4) or (ord(L.buf[L.bufpos]) - ord('0'))
|
xi = (xi shl 4) or (ord(L.buf[L.bufpos]) - ord('0'))
|
||||||
|
|
@ -629,10 +634,11 @@ proc handleHexChar(L: var TLexer, xi: var int) =
|
||||||
of 'A'..'F':
|
of 'A'..'F':
|
||||||
xi = (xi shl 4) or (ord(L.buf[L.bufpos]) - ord('A') + 10)
|
xi = (xi shl 4) or (ord(L.buf[L.bufpos]) - ord('A') + 10)
|
||||||
inc(L.bufpos)
|
inc(L.bufpos)
|
||||||
|
of '"', '\'':
|
||||||
|
if position == 1: invalid()
|
||||||
|
# do not progress the bufpos here.
|
||||||
else:
|
else:
|
||||||
lexMessage(L, errGenerated,
|
invalid()
|
||||||
"expected a hex digit, but found: " & L.buf[L.bufpos] &
|
|
||||||
" ; maybe prepend with 0")
|
|
||||||
# Need to progress for `nim check`
|
# Need to progress for `nim check`
|
||||||
inc(L.bufpos)
|
inc(L.bufpos)
|
||||||
|
|
||||||
|
|
@ -727,8 +733,8 @@ proc getEscapedChar(L: var TLexer, tok: var TToken) =
|
||||||
of 'x', 'X':
|
of 'x', 'X':
|
||||||
inc(L.bufpos)
|
inc(L.bufpos)
|
||||||
var xi = 0
|
var xi = 0
|
||||||
handleHexChar(L, xi)
|
handleHexChar(L, xi, 1)
|
||||||
handleHexChar(L, xi)
|
handleHexChar(L, xi, 2)
|
||||||
add(tok.literal, chr(xi))
|
add(tok.literal, chr(xi))
|
||||||
of 'u', 'U':
|
of 'u', 'U':
|
||||||
if tok.tokType == tkCharLit:
|
if tok.tokType == tkCharLit:
|
||||||
|
|
@ -739,7 +745,7 @@ proc getEscapedChar(L: var TLexer, tok: var TToken) =
|
||||||
inc(L.bufpos)
|
inc(L.bufpos)
|
||||||
var start = L.bufpos
|
var start = L.bufpos
|
||||||
while L.buf[L.bufpos] != '}':
|
while L.buf[L.bufpos] != '}':
|
||||||
handleHexChar(L, xi)
|
handleHexChar(L, xi, 1)
|
||||||
if start == L.bufpos:
|
if start == L.bufpos:
|
||||||
lexMessage(L, errGenerated,
|
lexMessage(L, errGenerated,
|
||||||
"Unicode codepoint cannot be empty")
|
"Unicode codepoint cannot be empty")
|
||||||
|
|
@ -749,10 +755,10 @@ proc getEscapedChar(L: var TLexer, tok: var TToken) =
|
||||||
lexMessage(L, errGenerated,
|
lexMessage(L, errGenerated,
|
||||||
"Unicode codepoint must be lower than 0x10FFFF, but was: " & hex)
|
"Unicode codepoint must be lower than 0x10FFFF, but was: " & hex)
|
||||||
else:
|
else:
|
||||||
handleHexChar(L, xi)
|
handleHexChar(L, xi, 1)
|
||||||
handleHexChar(L, xi)
|
handleHexChar(L, xi, 2)
|
||||||
handleHexChar(L, xi)
|
handleHexChar(L, xi, 3)
|
||||||
handleHexChar(L, xi)
|
handleHexChar(L, xi, 4)
|
||||||
addUnicodeCodePoint(tok.literal, xi)
|
addUnicodeCodePoint(tok.literal, xi)
|
||||||
of '0'..'9':
|
of '0'..'9':
|
||||||
if matchTwoChars(L, '0', {'0'..'9'}):
|
if matchTwoChars(L, '0', {'0'..'9'}):
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue