pegs: captured search loop

This commit is contained in:
Araq 2010-11-07 23:52:41 +01:00
commit 8ee63f9836
3 changed files with 55 additions and 5 deletions

View file

@ -66,6 +66,10 @@ notation meaning
failure. failure.
``@E`` Search: Shorthand for ``(!E .)* E``. (Search loop for the ``@E`` Search: Shorthand for ``(!E .)* E``. (Search loop for the
pattern `E`.) pattern `E`.)
``{@} E`` Captured Search: Shorthand for ``{(!E .)*} E``. (Search
loop for the pattern `E`.) Everything until and exluding
`E` is captured.
``@@ E`` Same as ``{@} E``.
``A <- E`` Rule: Bind the expression `E` to the *nonterminal symbol* ``A <- E`` Rule: Bind the expression `E` to the *nonterminal symbol*
`A`. **Left recursive rules are not possible and crash the `A`. **Left recursive rules are not possible and crash the
matching engine.** matching engine.**
@ -131,7 +135,7 @@ The PEG parser implements this grammar (written in PEG syntax)::
rule <- identifier \s* "<-" expr ig rule <- identifier \s* "<-" expr ig
identNoArrow <- identifier !(\s* "<-") identNoArrow <- identifier !(\s* "<-")
prefixOpr <- ig '&' / ig '!' / ig '@' prefixOpr <- ig '&' / ig '!' / ig '@' / ig '{@}' / ig '@@'
literal <- ig identifier? '$' [0-9]+ literal <- ig identifier? '$' [0-9]+
ig identNoArrow / ig identNoArrow /
ig charset / ig charset /

View file

@ -58,6 +58,7 @@ type
pkBackRefIgnoreCase, pkBackRefIgnoreCase,
pkBackRefIgnoreStyle, pkBackRefIgnoreStyle,
pkSearch, ## @a --> Internal DSL: @a pkSearch, ## @a --> Internal DSL: @a
pkCapturedSearch, ## {@} a --> Internal DSL: @@a
pkRule, ## a <- b pkRule, ## a <- b
pkList ## a, b pkList ## a, b
TNonTerminalFlag = enum TNonTerminalFlag = enum
@ -193,6 +194,11 @@ proc `@`*(a: TPeg): TPeg {.nosideEffect, rtl, extern: "npegsSearch".} =
result.kind = pkSearch result.kind = pkSearch
result.sons = @[a] result.sons = @[a]
proc `@@`*(a: TPeg): TPeg {.noSideEffect, rtl,
extern: "npgegsCapturedSearch".} =
result.kind = pkCapturedSearch
result.sons = @[a]
when false: when false:
proc contains(a: TPeg, k: TPegKind): bool = proc contains(a: TPeg, k: TPegKind): bool =
if a.kind == k: return true if a.kind == k: return true
@ -421,6 +427,9 @@ proc toStrAux(r: TPeg, res: var string) =
of pkSearch: of pkSearch:
add(res, '@') add(res, '@')
toStrAux(r.sons[0], res) toStrAux(r.sons[0], res)
of pkCapturedSearch:
add(res, "{@}")
toStrAux(r.sons[0], res)
of pkCapture: of pkCapture:
add(res, '{') add(res, '{')
toStrAux(r.sons[0], res) toStrAux(r.sons[0], res)
@ -558,6 +567,21 @@ proc m(s: string, p: TPeg, start: int, c: var TMatchClosure): int =
inc(result) inc(result)
result = -1 result = -1
c.ml = oldMl c.ml = oldMl
of pkCapturedSearch:
var idx = c.ml # reserve a slot for the subpattern
inc(c.ml)
result = 0
while start+result < s.len:
var x = m(s, p.sons[0], start+result, c)
if x >= 0:
if idx < maxSubpatterns:
c.matches[idx] = (start, start+result-1)
#else: silently ignore the capture
inc(result, x)
return
inc(result)
result = -1
c.ml = idx
of pkGreedyRep: of pkGreedyRep:
result = 0 result = 0
while true: while true:
@ -850,6 +874,7 @@ type
tkParRi, ## ')' tkParRi, ## ')'
tkCurlyLe, ## '{' tkCurlyLe, ## '{'
tkCurlyRi, ## '}' tkCurlyRi, ## '}'
tkCurlyAt, ## '{@}'
tkArrow, ## '<-' tkArrow, ## '<-'
tkBar, ## '/' tkBar, ## '/'
tkStar, ## '*' tkStar, ## '*'
@ -880,7 +905,8 @@ type
const const
tokKindToStr: array[TTokKind, string] = [ tokKindToStr: array[TTokKind, string] = [
"invalid", "[EOF]", ".", "_", "identifier", "string literal", "invalid", "[EOF]", ".", "_", "identifier", "string literal",
"character set", "(", ")", "{", "}", "<-", "/", "*", "+", "&", "!", "?", "character set", "(", ")", "{", "}", "{@}",
"<-", "/", "*", "+", "&", "!", "?",
"@", "built-in", "escaped", "$" "@", "built-in", "escaped", "$"
] ]
@ -1112,8 +1138,13 @@ proc getTok(c: var TPegLexer, tok: var TToken) =
skip(c) skip(c)
case c.buf[c.bufpos] case c.buf[c.bufpos]
of '{': of '{':
tok.kind = tkCurlyLe
inc(c.bufpos) inc(c.bufpos)
if c.buf[c.bufpos] == '@' and c.buf[c.bufpos+1] == '}':
tok.kind = tkCurlyAt
inc(c.bufpos, 2)
add(tok.literal, "{@}")
else:
tok.kind = tkCurlyLe
add(tok.literal, '{') add(tok.literal, '{')
of '}': of '}':
tok.kind = tkCurlyRi tok.kind = tkCurlyRi
@ -1193,6 +1224,10 @@ proc getTok(c: var TPegLexer, tok: var TToken) =
tok.kind = tkAt tok.kind = tkAt
inc(c.bufpos) inc(c.bufpos)
add(tok.literal, '@') add(tok.literal, '@')
if c.buf[c.bufpos] == '@':
tok.kind = tkCurlyAt
inc(c.bufpos)
add(tok.literal, '@')
else: else:
add(tok.literal, c.buf[c.bufpos]) add(tok.literal, c.buf[c.bufpos])
inc(c.bufpos) inc(c.bufpos)
@ -1261,6 +1296,9 @@ proc primary(p: var TPegParser): TPeg =
of tkAt: of tkAt:
getTok(p) getTok(p)
return @primary(p) return @primary(p)
of tkCurlyAt:
getTok(p)
return @@primary(p)
else: nil else: nil
case p.tok.kind case p.tok.kind
of tkIdentifier: of tkIdentifier:
@ -1346,7 +1384,7 @@ proc seqExpr(p: var TPegParser): TPeg =
while true: while true:
case p.tok.kind case p.tok.kind
of tkAmp, tkNot, tkAt, tkStringLit, tkCharset, tkParLe, tkCurlyLe, of tkAmp, tkNot, tkAt, tkStringLit, tkCharset, tkParLe, tkCurlyLe,
tkAny, tkAnyRune, tkBuiltin, tkEscaped, tkDollar: tkAny, tkAnyRune, tkBuiltin, tkEscaped, tkDollar, tkCurlyAt:
result = sequence(result, primary(p)) result = sequence(result, primary(p))
of tkIdentifier: of tkIdentifier:
if not arrowIsNextTok(p): if not arrowIsNextTok(p):
@ -1515,3 +1553,10 @@ when isMainModule:
for x in findAll("abcdef", peg"{.}", 3): for x in findAll("abcdef", peg"{.}", 3):
echo x echo x
if "f(a, b)" =~ peg"{[0-9]+} / ({\ident} '(' {@} ')')":
assert matches[0] == "f"
assert matches[1] == "a, b"
else:
assert false

View file

@ -21,6 +21,7 @@ Additions
- Added ``re.findAll``, ``pegs.findAll``. - Added ``re.findAll``, ``pegs.findAll``.
- Added ``os.findExe``. - Added ``os.findExe``.
- The Pegs module supports a *captured search loop operator* ``{@}``.
2010-10-20 Version 0.8.10 released 2010-10-20 Version 0.8.10 released