Change endpos to inclusive

This commit is contained in:
Oleh Prypin 2015-04-09 23:49:26 +03:00
commit 2f0375c4c8
3 changed files with 11 additions and 10 deletions

View file

@ -36,7 +36,7 @@ Finds the given pattern in the string between the end and start positions.
`start` :: The start point at which to start matching. `|abc` is `0`; `a|bc` `start` :: The start point at which to start matching. `|abc` is `0`; `a|bc`
is `1` is `1`
`endpos` :: The maximum index for a match; `int.high` means the end of the `endpos` :: The maximum index for a match; `int.high` means the end of the
string, otherwise it's an exclusive upper bound. string, otherwise it's an inclusive upper bound.
[[proc-match]] [[proc-match]]
==== match(string, Regex, start = 0, endpos = int.high): RegexMatch ==== match(string, Regex, start = 0, endpos = int.high): RegexMatch

View file

@ -311,7 +311,7 @@ proc matchImpl(str: string, pattern: Regex, start, endpos: int, flags: int): Opt
result.pcreMatchBounds = newSeq[Slice[cint]](ceil(vecsize / 2).int) result.pcreMatchBounds = newSeq[Slice[cint]](ceil(vecsize / 2).int)
result.pcreMatchBounds.setLen(vecsize div 3) result.pcreMatchBounds.setLen(vecsize div 3)
let strlen = if endpos == int.high: str.len else: endpos let strlen = if endpos == int.high: str.len else: endpos+1
let execRet = pcre.exec(pattern.pcreObj, let execRet = pcre.exec(pattern.pcreObj,
pattern.pcreExtra, pattern.pcreExtra,
@ -335,7 +335,7 @@ iterator findIter*(str: string, pattern: Regex, start = 0, endpos = int.high): R
# see pcredemo for explaination # see pcredemo for explaination
let matchesCrLf = pattern.matchesCrLf() let matchesCrLf = pattern.matchesCrLf()
let unicode = (getinfo[cint](pattern, pcre.INFO_OPTIONS) and pcre.UTF8) > 0 let unicode = (getinfo[cint](pattern, pcre.INFO_OPTIONS) and pcre.UTF8) > 0
let endpos = if endpos == int.high: str.len else: endpos let strlen = if endpos == int.high: str.len else: endpos+1
var offset = start var offset = start
var match: Option[RegexMatch] var match: Option[RegexMatch]
@ -361,13 +361,13 @@ iterator findIter*(str: string, pattern: Regex, start = 0, endpos = int.high): R
elif unicode: elif unicode:
# XXX what about invalid unicode? # XXX what about invalid unicode?
offset += str.runeLenAt(offset) offset += str.runeLenAt(offset)
assert(offset <= endpos) assert(offset <= strlen)
else: else:
offset = match.get.matchBounds.b + 1 offset = match.get.matchBounds.b + 1
yield match.get yield match.get
if offset >= endpos: if offset >= strlen:
# do while # do while
break break
@ -390,11 +390,11 @@ proc split*(str: string, pattern: Regex, maxSplit = -1, start = 0): seq[string]
var bounds = 0 .. -1 var bounds = 0 .. -1
for match in str.findIter(pattern, start = start): for match in str.findIter(pattern, start = start):
# upper bound is exclusive, lower is inclusive: # bounds are inclusive:
# #
# 0123456 # 0123456
# ^^^ # ^^^
# (1, 4) # (1, 3)
bounds = match.matchBounds bounds = match.matchBounds
# "12".split("") would be @["", "1", "2"], but # "12".split("") would be @["", "1", "2"], but

View file

@ -1,9 +1,10 @@
include nre, unittest, optional_t.nonstrict include nre, unittest, optional_t.nonstrict
suite "match": suite "match":
test "upper bound must be exclusive": test "upper bound must be inclusive":
check("abc".match(re"abc", endpos = 0) == None[RegexMatch]()) check("abc".match(re"abc", endpos = -1) == None[RegexMatch]())
check("abc".match(re"abc", endpos = 3) != None[RegexMatch]()) check("abc".match(re"abc", endpos = 1) == None[RegexMatch]())
check("abc".match(re"abc", endpos = 2) != None[RegexMatch]())
test "match examples": test "match examples":
check("abc".match(re"(\w)").captures[0] == "a") check("abc".match(re"(\w)").captures[0] == "a")