make re.split consistent with strutils.split and other programming languages; refs #7278
This commit is contained in:
parent
5c8332d871
commit
e2094bc6f4
2 changed files with 33 additions and 13 deletions
|
|
@ -4,10 +4,16 @@
|
||||||
|
|
||||||
#### Breaking changes in the standard library
|
#### Breaking changes in the standard library
|
||||||
|
|
||||||
|
- ``re.split`` for empty regular expressions now yields every character in
|
||||||
|
the string which is what other programming languages chose to do.
|
||||||
|
|
||||||
#### Breaking changes in the compiler
|
#### Breaking changes in the compiler
|
||||||
|
|
||||||
### Library additions
|
### Library additions
|
||||||
|
|
||||||
|
- ``re.split`` now also supports the ``maxsplit`` parameter for consistency
|
||||||
|
with ``strutils.split``.
|
||||||
|
|
||||||
### Library changes
|
### Library changes
|
||||||
|
|
||||||
### Language additions
|
### Language additions
|
||||||
|
|
|
||||||
|
|
@ -498,7 +498,7 @@ proc transformFile*(infile, outfile: string,
|
||||||
var x = readFile(infile).string
|
var x = readFile(infile).string
|
||||||
writeFile(outfile, x.multiReplace(subs))
|
writeFile(outfile, x.multiReplace(subs))
|
||||||
|
|
||||||
iterator split*(s: string, sep: Regex): string =
|
iterator split*(s: string, sep: Regex; maxsplit = -1): string =
|
||||||
## Splits the string ``s`` into substrings.
|
## Splits the string ``s`` into substrings.
|
||||||
##
|
##
|
||||||
## Substrings are separated by the regular expression ``sep``
|
## Substrings are separated by the regular expression ``sep``
|
||||||
|
|
@ -520,22 +520,28 @@ iterator split*(s: string, sep: Regex): string =
|
||||||
## "example"
|
## "example"
|
||||||
## ""
|
## ""
|
||||||
##
|
##
|
||||||
var
|
var last = 0
|
||||||
first = -1
|
var splits = maxsplit
|
||||||
last = -1
|
var x: int
|
||||||
while last < len(s):
|
while last <= len(s):
|
||||||
var x = matchLen(s, sep, last)
|
var first = last
|
||||||
if x > 0: inc(last, x)
|
var sepLen = 1
|
||||||
first = last
|
|
||||||
if x == 0: inc(last)
|
|
||||||
while last < len(s):
|
while last < len(s):
|
||||||
x = matchLen(s, sep, last)
|
x = matchLen(s, sep, last)
|
||||||
if x >= 0: break
|
if x >= 0:
|
||||||
|
sepLen = x
|
||||||
|
break
|
||||||
inc(last)
|
inc(last)
|
||||||
if first <= last:
|
if x == 0:
|
||||||
yield substr(s, first, last-1)
|
if last >= len(s): break
|
||||||
|
inc last
|
||||||
|
if splits == 0: last = len(s)
|
||||||
|
yield substr(s, first, last-1)
|
||||||
|
if splits == 0: break
|
||||||
|
dec(splits)
|
||||||
|
inc(last, sepLen)
|
||||||
|
|
||||||
proc split*(s: string, sep: Regex): seq[string] {.inline.} =
|
proc split*(s: string, sep: Regex, maxsplit = -1): seq[string] {.inline.} =
|
||||||
## Splits the string ``s`` into a seq of substrings.
|
## Splits the string ``s`` into a seq of substrings.
|
||||||
##
|
##
|
||||||
## The portion matched by ``sep`` is not returned.
|
## The portion matched by ``sep`` is not returned.
|
||||||
|
|
@ -632,6 +638,14 @@ when isMainModule:
|
||||||
accum.add(word)
|
accum.add(word)
|
||||||
doAssert(accum == @["AAA", "", "BBB"])
|
doAssert(accum == @["AAA", "", "BBB"])
|
||||||
|
|
||||||
|
doAssert(split("abc", re"") == @["a", "b", "c"])
|
||||||
|
doAssert(split("", re"") == @[])
|
||||||
|
|
||||||
|
doAssert(split("a;b;c", re";") == @["a", "b", "c"])
|
||||||
|
doAssert(split(";a;b;c", re";") == @["", "a", "b", "c"])
|
||||||
|
doAssert(split(";a;b;c;", re";") == @["", "a", "b", "c", ""])
|
||||||
|
doAssert(split("a;b;c;", re";") == @["a", "b", "c", ""])
|
||||||
|
|
||||||
for x in findAll("abcdef", re"^{.}", 3):
|
for x in findAll("abcdef", re"^{.}", 3):
|
||||||
doAssert x == "d"
|
doAssert x == "d"
|
||||||
accum = @[]
|
accum = @[]
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue