This commit is contained in:
Araq 2015-07-01 15:47:15 +02:00
commit 0d7e0e1b4f
2 changed files with 178 additions and 156 deletions

View file

@ -34,37 +34,15 @@ type
lineNumber*: int ## the current line number lineNumber*: int ## the current line number
sentinel: int sentinel: int
lineStart: int # index of last line start in buffer lineStart: int # index of last line start in buffer
fileOpened: bool refillChars: set[char]
{.deprecated: [TBaseLexer: BaseLexer].} {.deprecated: [TBaseLexer: BaseLexer].}
proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192)
## inits the BaseLexer with a stream to read from
proc close*(L: var BaseLexer)
## closes the base lexer. This closes `L`'s associated stream too.
proc getCurrentLine*(L: BaseLexer, marker: bool = true): string
## retrieves the current line.
proc getColNumber*(L: BaseLexer, pos: int): int
## retrieves the current column.
proc handleCR*(L: var BaseLexer, pos: int): int
## Call this if you scanned over '\c' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position
## of the '\c'.
proc handleLF*(L: var BaseLexer, pos: int): int
## Call this if you scanned over '\L' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position
## of the '\L'.
# implementation
const const
chrSize = sizeof(char) chrSize = sizeof(char)
proc close(L: var BaseLexer) = proc close*(L: var BaseLexer) =
## closes the base lexer. This closes `L`'s associated stream too.
dealloc(L.buf) dealloc(L.buf)
close(L.input) close(L.input)
@ -80,7 +58,7 @@ proc fillBuffer(L: var BaseLexer) =
toCopy = L.bufLen - L.sentinel - 1 toCopy = L.bufLen - L.sentinel - 1
assert(toCopy >= 0) assert(toCopy >= 0)
if toCopy > 0: if toCopy > 0:
moveMem(L.buf, addr(L.buf[L.sentinel + 1]), toCopy * chrSize) moveMem(L.buf, addr(L.buf[L.sentinel + 1]), toCopy * chrSize)
# "moveMem" handles overlapping regions # "moveMem" handles overlapping regions
charsRead = readData(L.input, addr(L.buf[toCopy]), charsRead = readData(L.input, addr(L.buf[toCopy]),
(L.sentinel + 1) * chrSize) div chrSize (L.sentinel + 1) * chrSize) div chrSize
@ -93,7 +71,7 @@ proc fillBuffer(L: var BaseLexer) =
dec(s) # BUGFIX (valgrind) dec(s) # BUGFIX (valgrind)
while true: while true:
assert(s < L.bufLen) assert(s < L.bufLen)
while (s >= 0) and not (L.buf[s] in NewLines): dec(s) while s >= 0 and L.buf[s] notin L.refillChars: dec(s)
if s >= 0: if s >= 0:
# we found an appropriate character for a sentinel: # we found an appropriate character for a sentinel:
L.sentinel = s L.sentinel = s
@ -121,31 +99,46 @@ proc fillBaseLexer(L: var BaseLexer, pos: int): int =
fillBuffer(L) fillBuffer(L)
L.bufpos = 0 # XXX: is this really correct? L.bufpos = 0 # XXX: is this really correct?
result = 0 result = 0
L.lineStart = result
proc handleCR(L: var BaseLexer, pos: int): int = proc handleCR*(L: var BaseLexer, pos: int): int =
## Call this if you scanned over '\c' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position
## of the '\c'.
assert(L.buf[pos] == '\c') assert(L.buf[pos] == '\c')
inc(L.lineNumber) inc(L.lineNumber)
result = fillBaseLexer(L, pos) result = fillBaseLexer(L, pos)
if L.buf[result] == '\L': if L.buf[result] == '\L':
result = fillBaseLexer(L, result) result = fillBaseLexer(L, result)
L.lineStart = result
proc handleLF(L: var BaseLexer, pos: int): int = proc handleLF*(L: var BaseLexer, pos: int): int =
## Call this if you scanned over '\L' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position
## of the '\L'.
assert(L.buf[pos] == '\L') assert(L.buf[pos] == '\L')
inc(L.lineNumber) inc(L.lineNumber)
result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result; result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result;
L.lineStart = result
proc handleRefillChar*(L: var BaseLexer, pos: int): int =
## To be documented.
assert(L.buf[pos] in L.refillChars)
result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result;
proc skipUtf8Bom(L: var BaseLexer) = proc skipUtf8Bom(L: var BaseLexer) =
if (L.buf[0] == '\xEF') and (L.buf[1] == '\xBB') and (L.buf[2] == '\xBF'): if (L.buf[0] == '\xEF') and (L.buf[1] == '\xBB') and (L.buf[2] == '\xBF'):
inc(L.bufpos, 3) inc(L.bufpos, 3)
inc(L.lineStart, 3) inc(L.lineStart, 3)
proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) = proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192;
refillChars: set[char] = NewLines) =
## inits the BaseLexer with a stream to read from.
assert(bufLen > 0) assert(bufLen > 0)
assert(input != nil) assert(input != nil)
L.input = input L.input = input
L.bufpos = 0 L.bufpos = 0
L.bufLen = bufLen L.bufLen = bufLen
L.refillChars = refillChars
L.buf = cast[cstring](alloc(bufLen * chrSize)) L.buf = cast[cstring](alloc(bufLen * chrSize))
L.sentinel = bufLen - 1 L.sentinel = bufLen - 1
L.lineStart = 0 L.lineStart = 0
@ -153,10 +146,12 @@ proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) =
fillBuffer(L) fillBuffer(L)
skipUtf8Bom(L) skipUtf8Bom(L)
proc getColNumber(L: BaseLexer, pos: int): int = proc getColNumber*(L: BaseLexer, pos: int): int =
## retrieves the current column.
result = abs(pos - L.lineStart) result = abs(pos - L.lineStart)
proc getCurrentLine(L: BaseLexer, marker: bool = true): string = proc getCurrentLine*(L: BaseLexer, marker: bool = true): string =
## retrieves the current line.
var i: int var i: int
result = "" result = ""
i = L.lineStart i = L.lineStart
@ -166,4 +161,3 @@ proc getCurrentLine(L: BaseLexer, marker: bool = true): string =
add(result, "\n") add(result, "\n")
if marker: if marker:
add(result, spaces(getColNumber(L, L.bufpos)) & "^\n") add(result, spaces(getColNumber(L, L.bufpos)) & "^\n")

View file

@ -8,19 +8,19 @@
# #
## This module implements a simple high performance `XML`:idx: / `HTML`:idx: ## This module implements a simple high performance `XML`:idx: / `HTML`:idx:
## parser. ## parser.
## The only encoding that is supported is UTF-8. The parser has been designed ## The only encoding that is supported is UTF-8. The parser has been designed
## to be somewhat error correcting, so that even most "wild HTML" found on the ## to be somewhat error correcting, so that even most "wild HTML" found on the
## web can be parsed with it. **Note:** This parser does not check that each ## web can be parsed with it. **Note:** This parser does not check that each
## ``<tag>`` has a corresponding ``</tag>``! These checks have do be ## ``<tag>`` has a corresponding ``</tag>``! These checks have do be
## implemented by the client code for various reasons: ## implemented by the client code for various reasons:
## ##
## * Old HTML contains tags that have no end tag: ``<br>`` for example. ## * Old HTML contains tags that have no end tag: ``<br>`` for example.
## * HTML tags are case insensitive, XML tags are case sensitive. Since this ## * HTML tags are case insensitive, XML tags are case sensitive. Since this
## library can parse both, only the client knows which comparison is to be ## library can parse both, only the client knows which comparison is to be
## used. ## used.
## * Thus the checks would have been very difficult to implement properly with ## * Thus the checks would have been very difficult to implement properly with
## little benefit, especially since they are simple to implement in the ## little benefit, especially since they are simple to implement in the
## client. The client should use the `errorMsgExpected` proc to generate ## client. The client should use the `errorMsgExpected` proc to generate
## a nice error message that fits the other error messages this library ## a nice error message that fits the other error messages this library
## creates. ## creates.
@ -29,7 +29,7 @@
## Example 1: Retrieve HTML title ## Example 1: Retrieve HTML title
## ============================== ## ==============================
## ##
## The file ``examples/htmltitle.nim`` demonstrates how to use the ## The file ``examples/htmltitle.nim`` demonstrates how to use the
## XML parser to accomplish a simple task: To determine the title of an HTML ## XML parser to accomplish a simple task: To determine the title of an HTML
## document. ## document.
## ##
@ -40,22 +40,22 @@
## Example 2: Retrieve all HTML links ## Example 2: Retrieve all HTML links
## ================================== ## ==================================
## ##
## The file ``examples/htmlrefs.nim`` demonstrates how to use the ## The file ``examples/htmlrefs.nim`` demonstrates how to use the
## XML parser to accomplish another simple task: To determine all the links ## XML parser to accomplish another simple task: To determine all the links
## an HTML document contains. ## an HTML document contains.
## ##
## .. code-block:: nim ## .. code-block:: nim
## :file: examples/htmlrefs.nim ## :file: examples/htmlrefs.nim
## ##
import import
hashes, strutils, lexbase, streams, unicode hashes, strutils, lexbase, streams, unicode
# the parser treats ``<br />`` as ``<br></br>`` # the parser treats ``<br />`` as ``<br></br>``
# xmlElementCloseEnd, ## ``/>`` # xmlElementCloseEnd, ## ``/>``
type type
XmlEventKind* = enum ## enumation of all events that may occur when parsing XmlEventKind* = enum ## enumation of all events that may occur when parsing
xmlError, ## an error occurred during parsing xmlError, ## an error occurred during parsing
xmlEof, ## end of file reached xmlEof, ## end of file reached
@ -65,13 +65,13 @@ type
xmlPI, ## processing instruction (``<?name something ?>``) xmlPI, ## processing instruction (``<?name something ?>``)
xmlElementStart, ## ``<elem>`` xmlElementStart, ## ``<elem>``
xmlElementEnd, ## ``</elem>`` xmlElementEnd, ## ``</elem>``
xmlElementOpen, ## ``<elem xmlElementOpen, ## ``<elem
xmlAttribute, ## ``key = "value"`` pair xmlAttribute, ## ``key = "value"`` pair
xmlElementClose, ## ``>`` xmlElementClose, ## ``>``
xmlCData, ## ``<![CDATA[`` ... data ... ``]]>`` xmlCData, ## ``<![CDATA[`` ... data ... ``]]>``
xmlEntity, ## &entity; xmlEntity, ## &entity;
xmlSpecial ## ``<! ... data ... >`` xmlSpecial ## ``<! ... data ... >``
XmlErrorKind* = enum ## enumeration that lists all errors that can occur XmlErrorKind* = enum ## enumeration that lists all errors that can occur
errNone, ## no error errNone, ## no error
errEndOfCDataExpected, ## ``]]>`` expected errEndOfCDataExpected, ## ``]]>`` expected
@ -82,8 +82,8 @@ type
errEqExpected, ## ``=`` expected errEqExpected, ## ``=`` expected
errQuoteExpected, ## ``"`` or ``'`` expected errQuoteExpected, ## ``"`` or ``'`` expected
errEndOfCommentExpected ## ``-->`` expected errEndOfCommentExpected ## ``-->`` expected
ParserState = enum ParserState = enum
stateStart, stateNormal, stateAttr, stateEmptyElementTag, stateError stateStart, stateNormal, stateAttr, stateEmptyElementTag, stateError
XmlParseOption* = enum ## options for the XML parser XmlParseOption* = enum ## options for the XML parser
@ -121,8 +121,8 @@ proc open*(my: var XmlParser, input: Stream, filename: string,
## the `options` parameter: If `options` contains ``reportWhitespace`` ## the `options` parameter: If `options` contains ``reportWhitespace``
## a whitespace token is reported as an ``xmlWhitespace`` event. ## a whitespace token is reported as an ``xmlWhitespace`` event.
## If `options` contains ``reportComments`` a comment token is reported as an ## If `options` contains ``reportComments`` a comment token is reported as an
## ``xmlComment`` event. ## ``xmlComment`` event.
lexbase.open(my, input) lexbase.open(my, input, 8192, {'\c', '\L', '/'})
my.filename = filename my.filename = filename
my.state = stateStart my.state = stateStart
my.kind = xmlError my.kind = xmlError
@ -130,24 +130,24 @@ proc open*(my: var XmlParser, input: Stream, filename: string,
my.b = "" my.b = ""
my.c = nil my.c = nil
my.options = options my.options = options
proc close*(my: var XmlParser) {.inline.} = proc close*(my: var XmlParser) {.inline.} =
## closes the parser `my` and its associated input stream. ## closes the parser `my` and its associated input stream.
lexbase.close(my) lexbase.close(my)
proc kind*(my: XmlParser): XmlEventKind {.inline.} = proc kind*(my: XmlParser): XmlEventKind {.inline.} =
## returns the current event type for the XML parser ## returns the current event type for the XML parser
return my.kind return my.kind
template charData*(my: XmlParser): string = template charData*(my: XmlParser): string =
## returns the character data for the events: ``xmlCharData``, ## returns the character data for the events: ``xmlCharData``,
## ``xmlWhitespace``, ``xmlComment``, ``xmlCData``, ``xmlSpecial`` ## ``xmlWhitespace``, ``xmlComment``, ``xmlCData``, ``xmlSpecial``
assert(my.kind in {xmlCharData, xmlWhitespace, xmlComment, xmlCData, assert(my.kind in {xmlCharData, xmlWhitespace, xmlComment, xmlCData,
xmlSpecial}) xmlSpecial})
my.a my.a
template elementName*(my: XmlParser): string = template elementName*(my: XmlParser): string =
## returns the element name for the events: ``xmlElementStart``, ## returns the element name for the events: ``xmlElementStart``,
## ``xmlElementEnd``, ``xmlElementOpen`` ## ``xmlElementEnd``, ``xmlElementOpen``
assert(my.kind in {xmlElementStart, xmlElementEnd, xmlElementOpen}) assert(my.kind in {xmlElementStart, xmlElementEnd, xmlElementOpen})
my.a my.a
@ -156,12 +156,12 @@ template entityName*(my: XmlParser): string =
## returns the entity name for the event: ``xmlEntity`` ## returns the entity name for the event: ``xmlEntity``
assert(my.kind == xmlEntity) assert(my.kind == xmlEntity)
my.a my.a
template attrKey*(my: XmlParser): string = template attrKey*(my: XmlParser): string =
## returns the attribute key for the event ``xmlAttribute`` ## returns the attribute key for the event ``xmlAttribute``
assert(my.kind == xmlAttribute) assert(my.kind == xmlAttribute)
my.a my.a
template attrValue*(my: XmlParser): string = template attrValue*(my: XmlParser): string =
## returns the attribute value for the event ``xmlAttribute`` ## returns the attribute value for the event ``xmlAttribute``
assert(my.kind == xmlAttribute) assert(my.kind == xmlAttribute)
@ -187,110 +187,118 @@ proc rawData2*(my: XmlParser): string {.inline.} =
## This is only used for speed hacks. ## This is only used for speed hacks.
shallowCopy(result, my.b) shallowCopy(result, my.b)
proc getColumn*(my: XmlParser): int {.inline.} = proc getColumn*(my: XmlParser): int {.inline.} =
## get the current column the parser has arrived at. ## get the current column the parser has arrived at.
result = getColNumber(my, my.bufpos) result = getColNumber(my, my.bufpos)
proc getLine*(my: XmlParser): int {.inline.} = proc getLine*(my: XmlParser): int {.inline.} =
## get the current line the parser has arrived at. ## get the current line the parser has arrived at.
result = my.lineNumber result = my.lineNumber
proc getFilename*(my: XmlParser): string {.inline.} = proc getFilename*(my: XmlParser): string {.inline.} =
## get the filename of the file that the parser processes. ## get the filename of the file that the parser processes.
result = my.filename result = my.filename
proc errorMsg*(my: XmlParser): string = proc errorMsg*(my: XmlParser): string =
## returns a helpful error message for the event ``xmlError`` ## returns a helpful error message for the event ``xmlError``
assert(my.kind == xmlError) assert(my.kind == xmlError)
result = "$1($2, $3) Error: $4" % [ result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), errorMessages[my.err]] my.filename, $getLine(my), $getColumn(my), errorMessages[my.err]]
proc errorMsgExpected*(my: XmlParser, tag: string): string = proc errorMsgExpected*(my: XmlParser, tag: string): string =
## returns an error message "<tag> expected" in the same format as the ## returns an error message "<tag> expected" in the same format as the
## other error messages ## other error messages
result = "$1($2, $3) Error: $4" % [ result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), "<$1> expected" % tag] my.filename, $getLine(my), $getColumn(my), "<$1> expected" % tag]
proc errorMsg*(my: XmlParser, msg: string): string = proc errorMsg*(my: XmlParser, msg: string): string =
## returns an error message with text `msg` in the same format as the ## returns an error message with text `msg` in the same format as the
## other error messages ## other error messages
result = "$1($2, $3) Error: $4" % [ result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), msg] my.filename, $getLine(my), $getColumn(my), msg]
proc markError(my: var XmlParser, kind: XmlErrorKind) {.inline.} = proc markError(my: var XmlParser, kind: XmlErrorKind) {.inline.} =
my.err = kind my.err = kind
my.state = stateError my.state = stateError
proc parseCDATA(my: var XmlParser) = proc parseCDATA(my: var XmlParser) =
var pos = my.bufpos + len("<![CDATA[") var pos = my.bufpos + len("<![CDATA[")
var buf = my.buf var buf = my.buf
while true: while true:
case buf[pos] case buf[pos]
of ']': of ']':
if buf[pos+1] == ']' and buf[pos+2] == '>': if buf[pos+1] == ']' and buf[pos+2] == '>':
inc(pos, 3) inc(pos, 3)
break break
add(my.a, ']') add(my.a, ']')
inc(pos) inc(pos)
of '\0': of '\0':
markError(my, errEndOfCDataExpected) markError(my, errEndOfCDataExpected)
break break
of '\c': of '\c':
pos = lexbase.handleCR(my, pos) pos = lexbase.handleCR(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '\L': of '\L':
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
add(my.a, '/')
else: else:
add(my.a, buf[pos]) add(my.a, buf[pos])
inc(pos) inc(pos)
my.bufpos = pos # store back my.bufpos = pos # store back
my.kind = xmlCData my.kind = xmlCData
proc parseComment(my: var XmlParser) = proc parseComment(my: var XmlParser) =
var pos = my.bufpos + len("<!--") var pos = my.bufpos + len("<!--")
var buf = my.buf var buf = my.buf
while true: while true:
case buf[pos] case buf[pos]
of '-': of '-':
if buf[pos+1] == '-' and buf[pos+2] == '>': if buf[pos+1] == '-' and buf[pos+2] == '>':
inc(pos, 3) inc(pos, 3)
break break
if my.options.contains(reportComments): add(my.a, '-') if my.options.contains(reportComments): add(my.a, '-')
inc(pos) inc(pos)
of '\0': of '\0':
markError(my, errEndOfCommentExpected) markError(my, errEndOfCommentExpected)
break break
of '\c': of '\c':
pos = lexbase.handleCR(my, pos) pos = lexbase.handleCR(my, pos)
buf = my.buf buf = my.buf
if my.options.contains(reportComments): add(my.a, '\L') if my.options.contains(reportComments): add(my.a, '\L')
of '\L': of '\L':
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
if my.options.contains(reportComments): add(my.a, '\L') if my.options.contains(reportComments): add(my.a, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
if my.options.contains(reportComments): add(my.a, '/')
else: else:
if my.options.contains(reportComments): add(my.a, buf[pos]) if my.options.contains(reportComments): add(my.a, buf[pos])
inc(pos) inc(pos)
my.bufpos = pos my.bufpos = pos
my.kind = xmlComment my.kind = xmlComment
proc parseWhitespace(my: var XmlParser, skip=false) = proc parseWhitespace(my: var XmlParser, skip=false) =
var pos = my.bufpos var pos = my.bufpos
var buf = my.buf var buf = my.buf
while true: while true:
case buf[pos] case buf[pos]
of ' ', '\t': of ' ', '\t':
if not skip: add(my.a, buf[pos]) if not skip: add(my.a, buf[pos])
inc(pos) inc(pos)
of '\c': of '\c':
# the specification says that CR-LF, CR are to be transformed to LF # the specification says that CR-LF, CR are to be transformed to LF
pos = lexbase.handleCR(my, pos) pos = lexbase.handleCR(my, pos)
buf = my.buf buf = my.buf
if not skip: add(my.a, '\L') if not skip: add(my.a, '\L')
of '\L': of '\L':
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
if not skip: add(my.a, '\L') if not skip: add(my.a, '\L')
@ -302,10 +310,10 @@ const
NameStartChar = {'A'..'Z', 'a'..'z', '_', ':', '\128'..'\255'} NameStartChar = {'A'..'Z', 'a'..'z', '_', ':', '\128'..'\255'}
NameChar = {'A'..'Z', 'a'..'z', '0'..'9', '.', '-', '_', ':', '\128'..'\255'} NameChar = {'A'..'Z', 'a'..'z', '0'..'9', '.', '-', '_', ':', '\128'..'\255'}
proc parseName(my: var XmlParser, dest: var string) = proc parseName(my: var XmlParser, dest: var string) =
var pos = my.bufpos var pos = my.bufpos
var buf = my.buf var buf = my.buf
if buf[pos] in NameStartChar: if buf[pos] in NameStartChar:
while true: while true:
add(dest, buf[pos]) add(dest, buf[pos])
inc(pos) inc(pos)
@ -314,14 +322,14 @@ proc parseName(my: var XmlParser, dest: var string) =
else: else:
markError(my, errNameExpected) markError(my, errNameExpected)
proc parseEntity(my: var XmlParser, dest: var string) = proc parseEntity(my: var XmlParser, dest: var string) =
var pos = my.bufpos+1 var pos = my.bufpos+1
var buf = my.buf var buf = my.buf
my.kind = xmlCharData my.kind = xmlCharData
if buf[pos] == '#': if buf[pos] == '#':
var r: int var r: int
inc(pos) inc(pos)
if buf[pos] == 'x': if buf[pos] == 'x':
inc(pos) inc(pos)
while true: while true:
case buf[pos] case buf[pos]
@ -331,7 +339,7 @@ proc parseEntity(my: var XmlParser, dest: var string) =
else: break else: break
inc(pos) inc(pos)
else: else:
while buf[pos] in {'0'..'9'}: while buf[pos] in {'0'..'9'}:
r = r * 10 + (ord(buf[pos]) - ord('0')) r = r * 10 + (ord(buf[pos]) - ord('0'))
inc(pos) inc(pos)
add(dest, toUTF8(Rune(r))) add(dest, toUTF8(Rune(r)))
@ -345,11 +353,11 @@ proc parseEntity(my: var XmlParser, dest: var string) =
buf[pos+3] == ';': buf[pos+3] == ';':
add(dest, '&') add(dest, '&')
inc(pos, 3) inc(pos, 3)
elif buf[pos] == 'a' and buf[pos+1] == 'p' and buf[pos+2] == 'o' and elif buf[pos] == 'a' and buf[pos+1] == 'p' and buf[pos+2] == 'o' and
buf[pos+3] == 's' and buf[pos+4] == ';': buf[pos+3] == 's' and buf[pos+4] == ';':
add(dest, '\'') add(dest, '\'')
inc(pos, 4) inc(pos, 4)
elif buf[pos] == 'q' and buf[pos+1] == 'u' and buf[pos+2] == 'o' and elif buf[pos] == 'q' and buf[pos+1] == 'u' and buf[pos+2] == 'o' and
buf[pos+3] == 't' and buf[pos+4] == ';': buf[pos+3] == 't' and buf[pos+4] == ';':
add(dest, '"') add(dest, '"')
inc(pos, 4) inc(pos, 4)
@ -357,23 +365,23 @@ proc parseEntity(my: var XmlParser, dest: var string) =
my.bufpos = pos my.bufpos = pos
parseName(my, dest) parseName(my, dest)
pos = my.bufpos pos = my.bufpos
if my.err != errNameExpected: if my.err != errNameExpected:
my.kind = xmlEntity my.kind = xmlEntity
else: else:
add(dest, '&') add(dest, '&')
if buf[pos] == ';': if buf[pos] == ';':
inc(pos) inc(pos)
else: else:
markError(my, errSemicolonExpected) markError(my, errSemicolonExpected)
my.bufpos = pos my.bufpos = pos
proc parsePI(my: var XmlParser) = proc parsePI(my: var XmlParser) =
inc(my.bufpos, "<?".len) inc(my.bufpos, "<?".len)
parseName(my, my.a) parseName(my, my.a)
var pos = my.bufpos var pos = my.bufpos
var buf = my.buf var buf = my.buf
setLen(my.b, 0) setLen(my.b, 0)
while true: while true:
case buf[pos] case buf[pos]
of '\0': of '\0':
markError(my, errQmGtExpected) markError(my, errQmGtExpected)
@ -387,29 +395,33 @@ proc parsePI(my: var XmlParser) =
of '\c': of '\c':
# the specification says that CR-LF, CR are to be transformed to LF # the specification says that CR-LF, CR are to be transformed to LF
pos = lexbase.handleCR(my, pos) pos = lexbase.handleCR(my, pos)
buf = my.buf buf = my.buf
add(my.b, '\L') add(my.b, '\L')
of '\L': of '\L':
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
add(my.b, '\L') add(my.b, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
add(my.b, '/')
else: else:
add(my.b, buf[pos]) add(my.b, buf[pos])
inc(pos) inc(pos)
my.bufpos = pos my.bufpos = pos
my.kind = xmlPI my.kind = xmlPI
proc parseSpecial(my: var XmlParser) = proc parseSpecial(my: var XmlParser) =
# things that start with <! # things that start with <!
var pos = my.bufpos + 2 var pos = my.bufpos + 2
var buf = my.buf var buf = my.buf
var opentags = 0 var opentags = 0
while true: while true:
case buf[pos] case buf[pos]
of '\0': of '\0':
markError(my, errGtExpected) markError(my, errGtExpected)
break break
of '<': of '<':
inc(opentags) inc(opentags)
inc(pos) inc(pos)
add(my.a, '<') add(my.a, '<')
@ -420,47 +432,55 @@ proc parseSpecial(my: var XmlParser) =
dec(opentags) dec(opentags)
inc(pos) inc(pos)
add(my.a, '>') add(my.a, '>')
of '\c': of '\c':
pos = lexbase.handleCR(my, pos) pos = lexbase.handleCR(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '\L': of '\L':
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
add(my.b, '/')
else: else:
add(my.a, buf[pos]) add(my.a, buf[pos])
inc(pos) inc(pos)
my.bufpos = pos my.bufpos = pos
my.kind = xmlSpecial my.kind = xmlSpecial
proc parseTag(my: var XmlParser) = proc parseTag(my: var XmlParser) =
inc(my.bufpos) inc(my.bufpos)
parseName(my, my.a) parseName(my, my.a)
# if we have no name, do not interpret the '<': # if we have no name, do not interpret the '<':
if my.a.len == 0: if my.a.len == 0:
my.kind = xmlCharData my.kind = xmlCharData
add(my.a, '<') add(my.a, '<')
return return
parseWhitespace(my, skip=true) parseWhitespace(my, skip=true)
if my.buf[my.bufpos] in NameStartChar: if my.buf[my.bufpos] in NameStartChar:
# an attribute follows: # an attribute follows:
my.kind = xmlElementOpen my.kind = xmlElementOpen
my.state = stateAttr my.state = stateAttr
my.c = my.a # save for later my.c = my.a # save for later
else: else:
my.kind = xmlElementStart my.kind = xmlElementStart
if my.buf[my.bufpos] == '/' and my.buf[my.bufpos+1] == '>': let slash = my.buf[my.bufpos] == '/'
inc(my.bufpos, 2) if slash:
my.bufpos = lexbase.handleRefillChar(my, my.bufpos)
if slash and my.buf[my.bufpos] == '>':
inc(my.bufpos)
my.state = stateEmptyElementTag my.state = stateEmptyElementTag
my.c = nil my.c = nil
elif my.buf[my.bufpos] == '>': elif my.buf[my.bufpos] == '>':
inc(my.bufpos) inc(my.bufpos)
else: else:
markError(my, errGtExpected) markError(my, errGtExpected)
proc parseEndTag(my: var XmlParser) = proc parseEndTag(my: var XmlParser) =
inc(my.bufpos, 2) my.bufpos = lexbase.handleRefillChar(my, my.bufpos+1)
#inc(my.bufpos, 2)
parseName(my, my.a) parseName(my, my.a)
parseWhitespace(my, skip=true) parseWhitespace(my, skip=true)
if my.buf[my.bufpos] == '>': if my.buf[my.bufpos] == '>':
@ -469,13 +489,13 @@ proc parseEndTag(my: var XmlParser) =
markError(my, errGtExpected) markError(my, errGtExpected)
my.kind = xmlElementEnd my.kind = xmlElementEnd
proc parseAttribute(my: var XmlParser) = proc parseAttribute(my: var XmlParser) =
my.kind = xmlAttribute my.kind = xmlAttribute
setLen(my.a, 0) setLen(my.a, 0)
setLen(my.b, 0) setLen(my.b, 0)
parseName(my, my.a) parseName(my, my.a)
# if we have no name, we have '<tag attr= key %&$$%': # if we have no name, we have '<tag attr= key %&$$%':
if my.a.len == 0: if my.a.len == 0:
markError(my, errGtExpected) markError(my, errGtExpected)
return return
parseWhitespace(my, skip=true) parseWhitespace(my, skip=true)
@ -491,27 +511,27 @@ proc parseAttribute(my: var XmlParser) =
var quote = buf[pos] var quote = buf[pos]
var pendingSpace = false var pendingSpace = false
inc(pos) inc(pos)
while true: while true:
case buf[pos] case buf[pos]
of '\0': of '\0':
markError(my, errQuoteExpected) markError(my, errQuoteExpected)
break break
of '&': of '&':
if pendingSpace: if pendingSpace:
add(my.b, ' ') add(my.b, ' ')
pendingSpace = false pendingSpace = false
my.bufpos = pos my.bufpos = pos
parseEntity(my, my.b) parseEntity(my, my.b)
my.kind = xmlAttribute # parseEntity overwrites my.kind! my.kind = xmlAttribute # parseEntity overwrites my.kind!
pos = my.bufpos pos = my.bufpos
of ' ', '\t': of ' ', '\t':
pendingSpace = true pendingSpace = true
inc(pos) inc(pos)
of '\c': of '\c':
pos = lexbase.handleCR(my, pos) pos = lexbase.handleCR(my, pos)
buf = my.buf buf = my.buf
pendingSpace = true pendingSpace = true
of '\L': of '\L':
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
pendingSpace = true pendingSpace = true
@ -520,44 +540,48 @@ proc parseAttribute(my: var XmlParser) =
inc(pos) inc(pos)
break break
else: else:
if pendingSpace: if pendingSpace:
add(my.b, ' ') add(my.b, ' ')
pendingSpace = false pendingSpace = false
add(my.b, buf[pos]) add(my.b, buf[pos])
inc(pos) inc(pos)
else: else:
markError(my, errQuoteExpected) markError(my, errQuoteExpected)
my.bufpos = pos my.bufpos = pos
parseWhitespace(my, skip=true) parseWhitespace(my, skip=true)
proc parseCharData(my: var XmlParser) = proc parseCharData(my: var XmlParser) =
var pos = my.bufpos var pos = my.bufpos
var buf = my.buf var buf = my.buf
while true: while true:
case buf[pos] case buf[pos]
of '\0', '<', '&': break of '\0', '<', '&': break
of '\c': of '\c':
# the specification says that CR-LF, CR are to be transformed to LF # the specification says that CR-LF, CR are to be transformed to LF
pos = lexbase.handleCR(my, pos) pos = lexbase.handleCR(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '\L': of '\L':
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
add(my.a, '/')
else: else:
add(my.a, buf[pos]) add(my.a, buf[pos])
inc(pos) inc(pos)
my.bufpos = pos my.bufpos = pos
my.kind = xmlCharData my.kind = xmlCharData
proc rawGetTok(my: var XmlParser) = proc rawGetTok(my: var XmlParser) =
my.kind = xmlError my.kind = xmlError
setLen(my.a, 0) setLen(my.a, 0)
var pos = my.bufpos var pos = my.bufpos
var buf = my.buf var buf = my.buf
case buf[pos] case buf[pos]
of '<': of '<':
case buf[pos+1] case buf[pos+1]
of '/': of '/':
parseEndTag(my) parseEndTag(my)
@ -566,44 +590,44 @@ proc rawGetTok(my: var XmlParser) =
buf[pos+5] == 'A' and buf[pos+6] == 'T' and buf[pos+7] == 'A' and buf[pos+5] == 'A' and buf[pos+6] == 'T' and buf[pos+7] == 'A' and
buf[pos+8] == '[': buf[pos+8] == '[':
parseCDATA(my) parseCDATA(my)
elif buf[pos+2] == '-' and buf[pos+3] == '-': elif buf[pos+2] == '-' and buf[pos+3] == '-':
parseComment(my) parseComment(my)
else: else:
parseSpecial(my) parseSpecial(my)
of '?': of '?':
parsePI(my) parsePI(my)
else: else:
parseTag(my) parseTag(my)
of ' ', '\t', '\c', '\l': of ' ', '\t', '\c', '\l':
parseWhitespace(my) parseWhitespace(my)
my.kind = xmlWhitespace my.kind = xmlWhitespace
of '\0': of '\0':
my.kind = xmlEof my.kind = xmlEof
of '&': of '&':
parseEntity(my, my.a) parseEntity(my, my.a)
else: else:
parseCharData(my) parseCharData(my)
assert my.kind != xmlError assert my.kind != xmlError
proc getTok(my: var XmlParser) = proc getTok(my: var XmlParser) =
while true: while true:
rawGetTok(my) rawGetTok(my)
case my.kind case my.kind
of xmlComment: of xmlComment:
if my.options.contains(reportComments): break if my.options.contains(reportComments): break
of xmlWhitespace: of xmlWhitespace:
if my.options.contains(reportWhitespace): break if my.options.contains(reportWhitespace): break
else: break else: break
proc next*(my: var XmlParser) = proc next*(my: var XmlParser) =
## retrieves the first/next event. This controls the parser. ## retrieves the first/next event. This controls the parser.
case my.state case my.state
of stateNormal: of stateNormal:
getTok(my) getTok(my)
of stateStart: of stateStart:
my.state = stateNormal my.state = stateNormal
getTok(my) getTok(my)
if my.kind == xmlPI and my.a == "xml": if my.kind == xmlPI and my.a == "xml":
# just skip the first ``<?xml >`` processing instruction # just skip the first ``<?xml >`` processing instruction
getTok(my) getTok(my)
of stateAttr: of stateAttr:
@ -612,10 +636,14 @@ proc next*(my: var XmlParser) =
my.kind = xmlElementClose my.kind = xmlElementClose
inc(my.bufpos) inc(my.bufpos)
my.state = stateNormal my.state = stateNormal
elif my.buf[my.bufpos] == '/' and my.buf[my.bufpos+1] == '>': elif my.buf[my.bufpos] == '/':
my.kind = xmlElementClose my.bufpos = lexbase.handleRefillChar(my, my.bufpos)
inc(my.bufpos, 2) if my.buf[my.bufpos] == '>':
my.state = stateEmptyElementTag my.kind = xmlElementClose
inc(my.bufpos)
my.state = stateEmptyElementTag
else:
markError(my, errGtExpected)
else: else:
parseAttribute(my) parseAttribute(my)
# state remains the same # state remains the same
@ -624,10 +652,10 @@ proc next*(my: var XmlParser) =
my.kind = xmlElementEnd my.kind = xmlElementEnd
if not my.c.isNil: if not my.c.isNil:
my.a = my.c my.a = my.c
of stateError: of stateError:
my.kind = xmlError my.kind = xmlError
my.state = stateNormal my.state = stateNormal
when not defined(testing) and isMainModule: when not defined(testing) and isMainModule:
import os import os
var s = newFileStream(paramStr(1), fmRead) var s = newFileStream(paramStr(1), fmRead)
@ -645,13 +673,13 @@ when not defined(testing) and isMainModule:
of xmlPI: echo("<? $1 ## $2 ?>" % [x.piName, x.piRest]) of xmlPI: echo("<? $1 ## $2 ?>" % [x.piName, x.piRest])
of xmlElementStart: echo("<$1>" % x.elementName) of xmlElementStart: echo("<$1>" % x.elementName)
of xmlElementEnd: echo("</$1>" % x.elementName) of xmlElementEnd: echo("</$1>" % x.elementName)
of xmlElementOpen: echo("<$1" % x.elementName) of xmlElementOpen: echo("<$1" % x.elementName)
of xmlAttribute: of xmlAttribute:
echo("Key: " & x.attrKey) echo("Key: " & x.attrKey)
echo("Value: " & x.attrValue) echo("Value: " & x.attrValue)
of xmlElementClose: echo(">") of xmlElementClose: echo(">")
of xmlCData: of xmlCData:
echo("<![CDATA[$1]]>" % x.charData) echo("<![CDATA[$1]]>" % x.charData)
of xmlEntity: of xmlEntity: