Update pegs.nim to work at compiletime. No range errors. (#13459)

This commit is contained in:
solo989 2020-02-27 02:08:57 -08:00 • committed by GitHub
commit e84e01cb8c
No known key found for this signature in database
GPG key ID: 4AEE18F83AFDEB23

View file

@ -1420,7 +1420,7 @@ type
PegLexer {.inheritable.} = object ## the lexer object. PegLexer {.inheritable.} = object ## the lexer object.
bufpos: int ## the current position within the buffer bufpos: int ## the current position within the buffer
buf: cstring ## the buffer itself buf: string ## the buffer itself
lineNumber: int ## the current line number lineNumber: int ## the current line number
lineStart: int ## index of last line start in buffer lineStart: int ## index of last line start in buffer
colOffset: int ## column to add colOffset: int ## column to add
@ -1481,6 +1481,9 @@ proc handleHexChar(c: var PegLexer, xi: var int) =
proc getEscapedChar(c: var PegLexer, tok: var Token) = proc getEscapedChar(c: var PegLexer, tok: var Token) =
inc(c.bufpos) inc(c.bufpos)
if c.bufpos >= len(c.buf):
tok.kind = tkInvalid
return
case c.buf[c.bufpos] case c.buf[c.bufpos]
of 'r', 'R', 'c', 'C': of 'r', 'R', 'c', 'C':
add(tok.literal, '\c') add(tok.literal, '\c')
@ -1508,6 +1511,9 @@ proc getEscapedChar(c: var PegLexer, tok: var Token) =
inc(c.bufpos) inc(c.bufpos)
of 'x', 'X': of 'x', 'X':
inc(c.bufpos) inc(c.bufpos)
if c.bufpos >= len(c.buf):
tok.kind = tkInvalid
return
var xi = 0 var xi = 0
handleHexChar(c, xi) handleHexChar(c, xi)
handleHexChar(c, xi) handleHexChar(c, xi)
@ -1517,7 +1523,7 @@ proc getEscapedChar(c: var PegLexer, tok: var Token) =
var val = ord(c.buf[c.bufpos]) - ord('0') var val = ord(c.buf[c.bufpos]) - ord('0')
inc(c.bufpos) inc(c.bufpos)
var i = 1 var i = 1
while (i <= 3) and (c.buf[c.bufpos] in {'0'..'9'}): while (c.bufpos < len(c.buf)) and (i <= 3) and (c.buf[c.bufpos] in {'0'..'9'}):
val = val * 10 + ord(c.buf[c.bufpos]) - ord('0') val = val * 10 + ord(c.buf[c.bufpos]) - ord('0')
inc(c.bufpos) inc(c.bufpos)
inc(i) inc(i)
@ -1571,7 +1577,7 @@ proc getString(c: var PegLexer, tok: var Token) =
proc getDollar(c: var PegLexer, tok: var Token) = proc getDollar(c: var PegLexer, tok: var Token) =
var pos = c.bufpos + 1 var pos = c.bufpos + 1
if c.buf[pos] in {'0'..'9'}: if pos < c.buf.len and c.buf[pos] in {'0'..'9'}:
tok.kind = tkBackref tok.kind = tkBackref
tok.index = 0 tok.index = 0
while pos < c.buf.len and c.buf[pos] in {'0'..'9'}: while pos < c.buf.len and c.buf[pos] in {'0'..'9'}:
@ -1586,6 +1592,7 @@ proc getCharSet(c: var PegLexer, tok: var Token) =
tok.charset = {} tok.charset = {}
var pos = c.bufpos + 1 var pos = c.bufpos + 1
var caret = false var caret = false
if pos < c.buf.len:
if c.buf[pos] == '^': if c.buf[pos] == '^':
inc(pos) inc(pos)
caret = true caret = true
@ -1661,6 +1668,13 @@ proc getTok(c: var PegLexer, tok: var Token) =
setLen(tok.literal, 0) setLen(tok.literal, 0)
skip(c) skip(c)
if c.bufpos >= c.buf.len:
tok.kind = tkEof
tok.literal = "[EOF]"
add(tok.literal, '\0')
inc(c.bufpos)
return
case c.buf[c.bufpos] case c.buf[c.bufpos]
of '{': of '{':
inc(c.bufpos) inc(c.bufpos)
@ -1700,6 +1714,8 @@ proc getTok(c: var PegLexer, tok: var Token) =
of '$': getDollar(c, tok) of '$': getDollar(c, tok)
of 'a'..'z', 'A'..'Z', '\128'..'\255': of 'a'..'z', 'A'..'Z', '\128'..'\255':
getSymbol(c, tok) getSymbol(c, tok)
if c.bufpos >= c.buf.len:
return
if c.buf[c.bufpos] in {'\'', '"'} or if c.buf[c.bufpos] in {'\'', '"'} or
c.buf[c.bufpos] == '$' and c.bufpos+1 < c.buf.len and c.buf[c.bufpos] == '$' and c.bufpos+1 < c.buf.len and
c.buf[c.bufpos+1] in {'0'..'9'}: c.buf[c.bufpos+1] in {'0'..'9'}:
@ -1768,7 +1784,9 @@ proc arrowIsNextTok(c: PegLexer): bool =
# the only look ahead we need # the only look ahead we need
var pos = c.bufpos var pos = c.bufpos
while pos < c.buf.len and c.buf[pos] in {'\t', ' '}: inc(pos) while pos < c.buf.len and c.buf[pos] in {'\t', ' '}: inc(pos)
result = c.buf[pos] == '<' and (pos+1 < c.buf.len) and c.buf[pos+1] == '-' if pos+1 >= c.buf.len:
return
result = c.buf[pos] == '<' and c.buf[pos+1] == '-'
# ----------------------------- parser ---------------------------------------- # ----------------------------- parser ----------------------------------------
@ -2038,6 +2056,7 @@ proc escapePeg*(s: string): string =
if inQuote: result.add('\'') if inQuote: result.add('\'')
when isMainModule: when isMainModule:
proc pegsTest() =
assert escapePeg("abc''def'") == r"'abc'\x27\x27'def'\x27" assert escapePeg("abc''def'") == r"'abc'\x27\x27'def'\x27"
assert match("(a b c)", peg"'(' @ ')'") assert match("(a b c)", peg"'(' @ ')'")
assert match("W_HI_Le", peg"\y 'while'") assert match("W_HI_Le", peg"\y 'while'")
@ -2158,7 +2177,7 @@ when isMainModule:
assert(str.find(empty_test) == 0) assert(str.find(empty_test) == 0)
assert(str.match(empty_test)) assert(str.match(empty_test))
proc handleMatches*(m: int, n: int, c: openArray[string]): string = proc handleMatches(m: int, n: int, c: openArray[string]): string =
result = "" result = ""
if m > 0: if m > 0:
@ -2176,3 +2195,6 @@ when isMainModule:
doAssert "test1".match(peg"""{@}$""") doAssert "test1".match(peg"""{@}$""")
doAssert "test2".match(peg"""{(!$ .)*} $""") doAssert "test2".match(peg"""{(!$ .)*} $""")
pegsTest()
static:
pegsTest()