fixes #1286; object case transitions are now sound

This commit is contained in:
Andreas Rumpf 2019-05-26 23:10:34 +02:00
commit 49e686ab4e
14 changed files with 123 additions and 166 deletions

View file

@ -142,34 +142,29 @@ proc rule*(nt: NonTerminal): Peg = nt.rule
proc term*(t: string): Peg {.nosideEffect, rtl, extern: "npegs$1Str".} =
## constructs a PEG from a terminal string
if t.len != 1:
result.kind = pkTerminal
result.term = t
result = Peg(kind: pkTerminal, term: t)
else:
result.kind = pkChar
result.ch = t[0]
result = Peg(kind: pkChar, ch: t[0])
proc termIgnoreCase*(t: string): Peg {.
nosideEffect, rtl, extern: "npegs$1".} =
## constructs a PEG from a terminal string; ignore case for matching
result.kind = pkTerminalIgnoreCase
result.term = t
result = Peg(kind: pkTerminalIgnoreCase, term: t)
proc termIgnoreStyle*(t: string): Peg {.
nosideEffect, rtl, extern: "npegs$1".} =
## constructs a PEG from a terminal string; ignore style for matching
result.kind = pkTerminalIgnoreStyle
result.term = t
result = Peg(kind: pkTerminalIgnoreStyle, term: t)
proc term*(t: char): Peg {.nosideEffect, rtl, extern: "npegs$1Char".} =
## constructs a PEG from a terminal char
assert t != '\0'
result.kind = pkChar
result.ch = t
result = Peg(kind: pkChar, ch: t)
proc charSet*(s: set[char]): Peg {.nosideEffect, rtl, extern: "npegs$1".} =
## constructs a PEG from a character set `s`
assert '\0' notin s
result.kind = pkCharChoice
result = Peg(kind: pkCharChoice)
new(result.charChoice)
result.charChoice[] = s
@ -189,8 +184,7 @@ proc addChoice(dest: var Peg, elem: Peg) =
else: add(dest, elem)
template multipleOp(k: PegKind, localOpt: untyped) =
result.kind = k
result.sons = @[]
result = Peg(kind: k, sons: @[])
for x in items(a):
if x.kind == k:
for y in items(x.sons):
@ -230,8 +224,7 @@ proc `?`*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsOptional".} =
# a? ? --> a?
result = a
else:
result.kind = pkOption
result.sons = @[a]
result = Peg(kind: pkOption, sons: @[a])
proc `*`*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsGreedyRep".} =
## constructs a "greedy repetition" for the PEG `a`
@ -240,27 +233,22 @@ proc `*`*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsGreedyRep".} =
assert false
# produces endless loop!
of pkChar:
result.kind = pkGreedyRepChar
result.ch = a.ch
result = Peg(kind: pkGreedyRepChar, ch: a.ch)
of pkCharChoice:
result.kind = pkGreedyRepSet
result.charChoice = a.charChoice # copying a reference suffices!
result = Peg(kind: pkGreedyRepSet, charChoice: a.charChoice)
of pkAny, pkAnyRune:
result.kind = pkGreedyAny
result = Peg(kind: pkGreedyAny)
else:
result.kind = pkGreedyRep
result.sons = @[a]
result = Peg(kind: pkGreedyRep, sons: @[a])
proc `!*`*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsSearch".} =
## constructs a "search" for the PEG `a`
result.kind = pkSearch
result.sons = @[a]
result = Peg(kind: pkSearch, sons: @[a])
proc `!*\`*(a: Peg): Peg {.noSideEffect, rtl,
extern: "npgegsCapturedSearch".} =
## constructs a "captured search" for the PEG `a`
result.kind = pkCapturedSearch
result.sons = @[a]
result = Peg(kind: pkCapturedSearch, sons: @[a])
proc `+`*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsGreedyPosRep".} =
## constructs a "greedy positive repetition" with the PEG `a`
@ -268,50 +256,48 @@ proc `+`*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsGreedyPosRep".} =
proc `&`*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsAndPredicate".} =
## constructs an "and predicate" with the PEG `a`
result.kind = pkAndPredicate
result.sons = @[a]
result = Peg(kind: pkAndPredicate, sons: @[a])
proc `!`*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsNotPredicate".} =
## constructs a "not predicate" with the PEG `a`
result.kind = pkNotPredicate
result.sons = @[a]
result = Peg(kind: pkNotPredicate, sons: @[a])
proc any*: Peg {.inline.} =
## constructs the PEG `any character`:idx: (``.``)
result.kind = pkAny
result = Peg(kind: pkAny)
proc anyRune*: Peg {.inline.} =
## constructs the PEG `any rune`:idx: (``_``)
result.kind = pkAnyRune
result = Peg(kind: pkAnyRune)
proc newLine*: Peg {.inline.} =
## constructs the PEG `newline`:idx: (``\n``)
result.kind = pkNewLine
result = Peg(kind: pkNewLine)
proc unicodeLetter*: Peg {.inline.} =
## constructs the PEG ``\letter`` which matches any Unicode letter.
result.kind = pkLetter
result = Peg(kind: pkLetter)
proc unicodeLower*: Peg {.inline.} =
## constructs the PEG ``\lower`` which matches any Unicode lowercase letter.
result.kind = pkLower
result = Peg(kind: pkLower)
proc unicodeUpper*: Peg {.inline.} =
## constructs the PEG ``\upper`` which matches any Unicode uppercase letter.
result.kind = pkUpper
result = Peg(kind: pkUpper)
proc unicodeTitle*: Peg {.inline.} =
## constructs the PEG ``\title`` which matches any Unicode title letter.
result.kind = pkTitle
result = Peg(kind: pkTitle)
proc unicodeWhitespace*: Peg {.inline.} =
## constructs the PEG ``\white`` which matches any Unicode
## whitespace character.
result.kind = pkWhitespace
result = Peg(kind: pkWhitespace)
proc startAnchor*: Peg {.inline.} =
## constructs the PEG ``^`` which matches the start of the input.
result.kind = pkStartAnchor
result = Peg(kind: pkStartAnchor)
proc endAnchor*: Peg {.inline.} =
## constructs the PEG ``$`` which matches the end of the input.
@ -319,29 +305,25 @@ proc endAnchor*: Peg {.inline.} =
proc capture*(a: Peg): Peg {.nosideEffect, rtl, extern: "npegsCapture".} =
## constructs a capture with the PEG `a`
result.kind = pkCapture
result.sons = @[a]
result = Peg(kind: pkCapture, sons: @[a])
proc backref*(index: range[1..MaxSubpatterns]): Peg {.
nosideEffect, rtl, extern: "npegs$1".} =
## constructs a back reference of the given `index`. `index` starts counting
## from 1.
result.kind = pkBackRef
result.index = index-1
result = Peg(kind: pkBackRef, index: index-1)
proc backrefIgnoreCase*(index: range[1..MaxSubpatterns]): Peg {.
nosideEffect, rtl, extern: "npegs$1".} =
## constructs a back reference of the given `index`. `index` starts counting
## from 1. Ignores case for matching.
result.kind = pkBackRefIgnoreCase
result.index = index-1
result = Peg(kind: pkBackRefIgnoreCase, index: index-1)
proc backrefIgnoreStyle*(index: range[1..MaxSubpatterns]): Peg {.
nosideEffect, rtl, extern: "npegs$1".}=
## constructs a back reference of the given `index`. `index` starts counting
## from 1. Ignores style for matching.
result.kind = pkBackRefIgnoreStyle
result.index = index-1
result = Peg(kind: pkBackRefIgnoreStyle, index: index-1)
proc spaceCost(n: Peg): int =
case n.kind
@ -366,16 +348,12 @@ proc nonterminal*(n: NonTerminal): Peg {.
when false: echo "inlining symbol: ", n.name
result = n.rule # inlining of rule enables better optimizations
else:
result.kind = pkNonTerminal
result.nt = n
result = Peg(kind: pkNonTerminal, nt: n)
proc newNonTerminal*(name: string, line, column: int): NonTerminal {.
nosideEffect, rtl, extern: "npegs$1".} =
## constructs a nonterminal symbol
new(result)
result.name = name
result.line = line
result.col = column
result = NonTerminal(name: name, line: line, col: column)
template letters*: Peg =
## expands to ``charset({'A'..'Z', 'a'..'z'})``
@ -585,8 +563,14 @@ template matchOrParse(mopProc: untyped) =
if p.index >= c.ml: return -1
var (a, b) = c.matches[p.index]
var n: Peg
n.kind = succ(pkTerminal, ord(p.kind)-ord(pkBackRef))
n.term = s.substr(a, b)
case p.kind
of pkBackRef:
n = Peg(kind: pkTerminal, term: s.substr(a, b))
of pkBackRefIgnoreStyle:
n = Peg(kind: pkTerminalIgnoreStyle, term: s.substr(a, b))
of pkBackRefIgnoreCase:
n = Peg(kind: pkTerminalIgnoreCase, term: s.substr(a, b))
else: assert(false, "impossible case")
mopProc(s, n, start, c)
case p.kind