make the lexer more forgiving so that nim-regex compiles again

This commit is contained in:
Araq 2019-02-04 15:49:36 +01:00
commit 23c11987b4

View file

@ -618,7 +618,12 @@ proc getNumber(L: var TLexer, result: var TToken) =
tokenEnd(result, postPos-1) tokenEnd(result, postPos-1)
L.bufpos = postPos L.bufpos = postPos
proc handleHexChar(L: var TLexer, xi: var int) = proc handleHexChar(L: var TLexer, xi: var int; position: range[1..4]) =
template invalid() =
lexMessage(L, errGenerated,
"expected a hex digit, but found: " & L.buf[L.bufpos] &
"; maybe prepend with 0")
case L.buf[L.bufpos] case L.buf[L.bufpos]
of '0'..'9': of '0'..'9':
xi = (xi shl 4) or (ord(L.buf[L.bufpos]) - ord('0')) xi = (xi shl 4) or (ord(L.buf[L.bufpos]) - ord('0'))
@ -629,10 +634,11 @@ proc handleHexChar(L: var TLexer, xi: var int) =
of 'A'..'F': of 'A'..'F':
xi = (xi shl 4) or (ord(L.buf[L.bufpos]) - ord('A') + 10) xi = (xi shl 4) or (ord(L.buf[L.bufpos]) - ord('A') + 10)
inc(L.bufpos) inc(L.bufpos)
of '"', '\'':
if position == 1: invalid()
# do not progress the bufpos here.
else: else:
lexMessage(L, errGenerated, invalid()
"expected a hex digit, but found: " & L.buf[L.bufpos] &
" ; maybe prepend with 0")
# Need to progress for `nim check` # Need to progress for `nim check`
inc(L.bufpos) inc(L.bufpos)
@ -727,8 +733,8 @@ proc getEscapedChar(L: var TLexer, tok: var TToken) =
of 'x', 'X': of 'x', 'X':
inc(L.bufpos) inc(L.bufpos)
var xi = 0 var xi = 0
handleHexChar(L, xi) handleHexChar(L, xi, 1)
handleHexChar(L, xi) handleHexChar(L, xi, 2)
add(tok.literal, chr(xi)) add(tok.literal, chr(xi))
of 'u', 'U': of 'u', 'U':
if tok.tokType == tkCharLit: if tok.tokType == tkCharLit:
@ -739,7 +745,7 @@ proc getEscapedChar(L: var TLexer, tok: var TToken) =
inc(L.bufpos) inc(L.bufpos)
var start = L.bufpos var start = L.bufpos
while L.buf[L.bufpos] != '}': while L.buf[L.bufpos] != '}':
handleHexChar(L, xi) handleHexChar(L, xi, 1)
if start == L.bufpos: if start == L.bufpos:
lexMessage(L, errGenerated, lexMessage(L, errGenerated,
"Unicode codepoint cannot be empty") "Unicode codepoint cannot be empty")
@ -749,10 +755,10 @@ proc getEscapedChar(L: var TLexer, tok: var TToken) =
lexMessage(L, errGenerated, lexMessage(L, errGenerated,
"Unicode codepoint must be lower than 0x10FFFF, but was: " & hex) "Unicode codepoint must be lower than 0x10FFFF, but was: " & hex)
else: else:
handleHexChar(L, xi) handleHexChar(L, xi, 1)
handleHexChar(L, xi) handleHexChar(L, xi, 2)
handleHexChar(L, xi) handleHexChar(L, xi, 3)
handleHexChar(L, xi) handleHexChar(L, xi, 4)
addUnicodeCodePoint(tok.literal, xi) addUnicodeCodePoint(tok.literal, xi)
of '0'..'9': of '0'..'9':
if matchTwoChars(L, '0', {'0'..'9'}): if matchTwoChars(L, '0', {'0'..'9'}):