diff --git a/lib/pure/lexbase.nim b/lib/pure/lexbase.nim
index 585ba87f5..bfecf6a58 100644
--- a/lib/pure/lexbase.nim
+++ b/lib/pure/lexbase.nim
@@ -34,37 +34,15 @@ type
lineNumber*: int ## the current line number
sentinel: int
lineStart: int # index of last line start in buffer
- fileOpened: bool
+ refillChars: set[char]
{.deprecated: [TBaseLexer: BaseLexer].}
-proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192)
- ## inits the BaseLexer with a stream to read from
-
-proc close*(L: var BaseLexer)
- ## closes the base lexer. This closes `L`'s associated stream too.
-
-proc getCurrentLine*(L: BaseLexer, marker: bool = true): string
- ## retrieves the current line.
-
-proc getColNumber*(L: BaseLexer, pos: int): int
- ## retrieves the current column.
-
-proc handleCR*(L: var BaseLexer, pos: int): int
- ## Call this if you scanned over '\c' in the buffer; it returns the the
- ## position to continue the scanning from. `pos` must be the position
- ## of the '\c'.
-proc handleLF*(L: var BaseLexer, pos: int): int
- ## Call this if you scanned over '\L' in the buffer; it returns the the
- ## position to continue the scanning from. `pos` must be the position
- ## of the '\L'.
-
-# implementation
-
const
chrSize = sizeof(char)
-proc close(L: var BaseLexer) =
+proc close*(L: var BaseLexer) =
+ ## closes the base lexer. This closes `L`'s associated stream too.
dealloc(L.buf)
close(L.input)
@@ -80,7 +58,7 @@ proc fillBuffer(L: var BaseLexer) =
toCopy = L.bufLen - L.sentinel - 1
assert(toCopy >= 0)
if toCopy > 0:
- moveMem(L.buf, addr(L.buf[L.sentinel + 1]), toCopy * chrSize)
+ moveMem(L.buf, addr(L.buf[L.sentinel + 1]), toCopy * chrSize)
# "moveMem" handles overlapping regions
charsRead = readData(L.input, addr(L.buf[toCopy]),
(L.sentinel + 1) * chrSize) div chrSize
@@ -93,7 +71,7 @@ proc fillBuffer(L: var BaseLexer) =
dec(s) # BUGFIX (valgrind)
while true:
assert(s < L.bufLen)
- while (s >= 0) and not (L.buf[s] in NewLines): dec(s)
+ while s >= 0 and L.buf[s] notin L.refillChars: dec(s)
if s >= 0:
# we found an appropriate character for a sentinel:
L.sentinel = s
@@ -121,31 +99,46 @@ proc fillBaseLexer(L: var BaseLexer, pos: int): int =
fillBuffer(L)
L.bufpos = 0 # XXX: is this really correct?
result = 0
- L.lineStart = result
-proc handleCR(L: var BaseLexer, pos: int): int =
+proc handleCR*(L: var BaseLexer, pos: int): int =
+ ## Call this if you scanned over '\c' in the buffer; it returns the the
+ ## position to continue the scanning from. `pos` must be the position
+ ## of the '\c'.
assert(L.buf[pos] == '\c')
inc(L.lineNumber)
result = fillBaseLexer(L, pos)
if L.buf[result] == '\L':
result = fillBaseLexer(L, result)
+ L.lineStart = result
-proc handleLF(L: var BaseLexer, pos: int): int =
+proc handleLF*(L: var BaseLexer, pos: int): int =
+ ## Call this if you scanned over '\L' in the buffer; it returns the the
+ ## position to continue the scanning from. `pos` must be the position
+ ## of the '\L'.
assert(L.buf[pos] == '\L')
inc(L.lineNumber)
result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result;
+ L.lineStart = result
+
+proc handleRefillChar*(L: var BaseLexer, pos: int): int =
+ ## To be documented.
+ assert(L.buf[pos] in L.refillChars)
+ result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result;
proc skipUtf8Bom(L: var BaseLexer) =
if (L.buf[0] == '\xEF') and (L.buf[1] == '\xBB') and (L.buf[2] == '\xBF'):
inc(L.bufpos, 3)
inc(L.lineStart, 3)
-proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) =
+proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192;
+ refillChars: set[char] = NewLines) =
+ ## inits the BaseLexer with a stream to read from.
assert(bufLen > 0)
assert(input != nil)
L.input = input
L.bufpos = 0
L.bufLen = bufLen
+ L.refillChars = refillChars
L.buf = cast[cstring](alloc(bufLen * chrSize))
L.sentinel = bufLen - 1
L.lineStart = 0
@@ -153,10 +146,12 @@ proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) =
fillBuffer(L)
skipUtf8Bom(L)
-proc getColNumber(L: BaseLexer, pos: int): int =
+proc getColNumber*(L: BaseLexer, pos: int): int =
+ ## retrieves the current column.
result = abs(pos - L.lineStart)
-proc getCurrentLine(L: BaseLexer, marker: bool = true): string =
+proc getCurrentLine*(L: BaseLexer, marker: bool = true): string =
+ ## retrieves the current line.
var i: int
result = ""
i = L.lineStart
@@ -166,4 +161,3 @@ proc getCurrentLine(L: BaseLexer, marker: bool = true): string =
add(result, "\n")
if marker:
add(result, spaces(getColNumber(L, L.bufpos)) & "^\n")
-
diff --git a/lib/pure/parsexml.nim b/lib/pure/parsexml.nim
index eb792f086..e1abb0a4f 100644
--- a/lib/pure/parsexml.nim
+++ b/lib/pure/parsexml.nim
@@ -8,19 +8,19 @@
#
## This module implements a simple high performance `XML`:idx: / `HTML`:idx:
-## parser.
+## parser.
## The only encoding that is supported is UTF-8. The parser has been designed
-## to be somewhat error correcting, so that even most "wild HTML" found on the
+## to be somewhat error correcting, so that even most "wild HTML" found on the
## web can be parsed with it. **Note:** This parser does not check that each
-## ```` has a corresponding ````! These checks have do be
-## implemented by the client code for various reasons:
+## ```` has a corresponding ````! These checks have do be
+## implemented by the client code for various reasons:
##
## * Old HTML contains tags that have no end tag: ``
`` for example.
## * HTML tags are case insensitive, XML tags are case sensitive. Since this
## library can parse both, only the client knows which comparison is to be
## used.
## * Thus the checks would have been very difficult to implement properly with
-## little benefit, especially since they are simple to implement in the
+## little benefit, especially since they are simple to implement in the
## client. The client should use the `errorMsgExpected` proc to generate
## a nice error message that fits the other error messages this library
## creates.
@@ -29,7 +29,7 @@
## Example 1: Retrieve HTML title
## ==============================
##
-## The file ``examples/htmltitle.nim`` demonstrates how to use the
+## The file ``examples/htmltitle.nim`` demonstrates how to use the
## XML parser to accomplish a simple task: To determine the title of an HTML
## document.
##
@@ -40,22 +40,22 @@
## Example 2: Retrieve all HTML links
## ==================================
##
-## The file ``examples/htmlrefs.nim`` demonstrates how to use the
-## XML parser to accomplish another simple task: To determine all the links
+## The file ``examples/htmlrefs.nim`` demonstrates how to use the
+## XML parser to accomplish another simple task: To determine all the links
## an HTML document contains.
##
## .. code-block:: nim
## :file: examples/htmlrefs.nim
##
-import
+import
hashes, strutils, lexbase, streams, unicode
# the parser treats ``
`` as ``
``
-# xmlElementCloseEnd, ## ``/>``
+# xmlElementCloseEnd, ## ``/>``
-type
+type
XmlEventKind* = enum ## enumation of all events that may occur when parsing
xmlError, ## an error occurred during parsing
xmlEof, ## end of file reached
@@ -65,13 +65,13 @@ type
xmlPI, ## processing instruction (````)
xmlElementStart, ## ````
xmlElementEnd, ## ````
- xmlElementOpen, ## ````
+ xmlElementClose, ## ``>``
xmlCData, ## ````
xmlEntity, ## &entity;
xmlSpecial ## ````
-
+
XmlErrorKind* = enum ## enumeration that lists all errors that can occur
errNone, ## no error
errEndOfCDataExpected, ## ``]]>`` expected
@@ -82,8 +82,8 @@ type
errEqExpected, ## ``=`` expected
errQuoteExpected, ## ``"`` or ``'`` expected
errEndOfCommentExpected ## ``-->`` expected
-
- ParserState = enum
+
+ ParserState = enum
stateStart, stateNormal, stateAttr, stateEmptyElementTag, stateError
XmlParseOption* = enum ## options for the XML parser
@@ -121,8 +121,8 @@ proc open*(my: var XmlParser, input: Stream, filename: string,
## the `options` parameter: If `options` contains ``reportWhitespace``
## a whitespace token is reported as an ``xmlWhitespace`` event.
## If `options` contains ``reportComments`` a comment token is reported as an
- ## ``xmlComment`` event.
- lexbase.open(my, input)
+ ## ``xmlComment`` event.
+ lexbase.open(my, input, 8192, {'\c', '\L', '/'})
my.filename = filename
my.state = stateStart
my.kind = xmlError
@@ -130,24 +130,24 @@ proc open*(my: var XmlParser, input: Stream, filename: string,
my.b = ""
my.c = nil
my.options = options
-
-proc close*(my: var XmlParser) {.inline.} =
+
+proc close*(my: var XmlParser) {.inline.} =
## closes the parser `my` and its associated input stream.
lexbase.close(my)
-proc kind*(my: XmlParser): XmlEventKind {.inline.} =
+proc kind*(my: XmlParser): XmlEventKind {.inline.} =
## returns the current event type for the XML parser
return my.kind
template charData*(my: XmlParser): string =
- ## returns the character data for the events: ``xmlCharData``,
+ ## returns the character data for the events: ``xmlCharData``,
## ``xmlWhitespace``, ``xmlComment``, ``xmlCData``, ``xmlSpecial``
- assert(my.kind in {xmlCharData, xmlWhitespace, xmlComment, xmlCData,
+ assert(my.kind in {xmlCharData, xmlWhitespace, xmlComment, xmlCData,
xmlSpecial})
my.a
template elementName*(my: XmlParser): string =
- ## returns the element name for the events: ``xmlElementStart``,
+ ## returns the element name for the events: ``xmlElementStart``,
## ``xmlElementEnd``, ``xmlElementOpen``
assert(my.kind in {xmlElementStart, xmlElementEnd, xmlElementOpen})
my.a
@@ -156,12 +156,12 @@ template entityName*(my: XmlParser): string =
## returns the entity name for the event: ``xmlEntity``
assert(my.kind == xmlEntity)
my.a
-
+
template attrKey*(my: XmlParser): string =
## returns the attribute key for the event ``xmlAttribute``
assert(my.kind == xmlAttribute)
my.a
-
+
template attrValue*(my: XmlParser): string =
## returns the attribute value for the event ``xmlAttribute``
assert(my.kind == xmlAttribute)
@@ -187,110 +187,118 @@ proc rawData2*(my: XmlParser): string {.inline.} =
## This is only used for speed hacks.
shallowCopy(result, my.b)
-proc getColumn*(my: XmlParser): int {.inline.} =
+proc getColumn*(my: XmlParser): int {.inline.} =
## get the current column the parser has arrived at.
result = getColNumber(my, my.bufpos)
-proc getLine*(my: XmlParser): int {.inline.} =
+proc getLine*(my: XmlParser): int {.inline.} =
## get the current line the parser has arrived at.
result = my.lineNumber
-proc getFilename*(my: XmlParser): string {.inline.} =
+proc getFilename*(my: XmlParser): string {.inline.} =
## get the filename of the file that the parser processes.
result = my.filename
-
-proc errorMsg*(my: XmlParser): string =
+
+proc errorMsg*(my: XmlParser): string =
## returns a helpful error message for the event ``xmlError``
assert(my.kind == xmlError)
result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), errorMessages[my.err]]
-proc errorMsgExpected*(my: XmlParser, tag: string): string =
+proc errorMsgExpected*(my: XmlParser, tag: string): string =
## returns an error message " expected" in the same format as the
- ## other error messages
+ ## other error messages
result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), "<$1> expected" % tag]
-proc errorMsg*(my: XmlParser, msg: string): string =
+proc errorMsg*(my: XmlParser, msg: string): string =
## returns an error message with text `msg` in the same format as the
- ## other error messages
+ ## other error messages
result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), msg]
-
-proc markError(my: var XmlParser, kind: XmlErrorKind) {.inline.} =
+
+proc markError(my: var XmlParser, kind: XmlErrorKind) {.inline.} =
my.err = kind
my.state = stateError
-proc parseCDATA(my: var XmlParser) =
+proc parseCDATA(my: var XmlParser) =
var pos = my.bufpos + len("':
inc(pos, 3)
break
add(my.a, ']')
inc(pos)
- of '\0':
+ of '\0':
markError(my, errEndOfCDataExpected)
break
- of '\c':
+ of '\c':
pos = lexbase.handleCR(my, pos)
buf = my.buf
add(my.a, '\L')
- of '\L':
+ of '\L':
pos = lexbase.handleLF(my, pos)
buf = my.buf
add(my.a, '\L')
+ of '/':
+ pos = lexbase.handleRefillChar(my, pos)
+ buf = my.buf
+ add(my.a, '/')
else:
add(my.a, buf[pos])
- inc(pos)
+ inc(pos)
my.bufpos = pos # store back
my.kind = xmlCData
-proc parseComment(my: var XmlParser) =
+proc parseComment(my: var XmlParser) =
var pos = my.bufpos + len("