diff --git a/lib/pure/lexbase.nim b/lib/pure/lexbase.nim index 585ba87f5..bfecf6a58 100644 --- a/lib/pure/lexbase.nim +++ b/lib/pure/lexbase.nim @@ -34,37 +34,15 @@ type lineNumber*: int ## the current line number sentinel: int lineStart: int # index of last line start in buffer - fileOpened: bool + refillChars: set[char] {.deprecated: [TBaseLexer: BaseLexer].} -proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192) - ## inits the BaseLexer with a stream to read from - -proc close*(L: var BaseLexer) - ## closes the base lexer. This closes `L`'s associated stream too. - -proc getCurrentLine*(L: BaseLexer, marker: bool = true): string - ## retrieves the current line. - -proc getColNumber*(L: BaseLexer, pos: int): int - ## retrieves the current column. - -proc handleCR*(L: var BaseLexer, pos: int): int - ## Call this if you scanned over '\c' in the buffer; it returns the the - ## position to continue the scanning from. `pos` must be the position - ## of the '\c'. -proc handleLF*(L: var BaseLexer, pos: int): int - ## Call this if you scanned over '\L' in the buffer; it returns the the - ## position to continue the scanning from. `pos` must be the position - ## of the '\L'. - -# implementation - const chrSize = sizeof(char) -proc close(L: var BaseLexer) = +proc close*(L: var BaseLexer) = + ## closes the base lexer. This closes `L`'s associated stream too. dealloc(L.buf) close(L.input) @@ -80,7 +58,7 @@ proc fillBuffer(L: var BaseLexer) = toCopy = L.bufLen - L.sentinel - 1 assert(toCopy >= 0) if toCopy > 0: - moveMem(L.buf, addr(L.buf[L.sentinel + 1]), toCopy * chrSize) + moveMem(L.buf, addr(L.buf[L.sentinel + 1]), toCopy * chrSize) # "moveMem" handles overlapping regions charsRead = readData(L.input, addr(L.buf[toCopy]), (L.sentinel + 1) * chrSize) div chrSize @@ -93,7 +71,7 @@ proc fillBuffer(L: var BaseLexer) = dec(s) # BUGFIX (valgrind) while true: assert(s < L.bufLen) - while (s >= 0) and not (L.buf[s] in NewLines): dec(s) + while s >= 0 and L.buf[s] notin L.refillChars: dec(s) if s >= 0: # we found an appropriate character for a sentinel: L.sentinel = s @@ -121,31 +99,46 @@ proc fillBaseLexer(L: var BaseLexer, pos: int): int = fillBuffer(L) L.bufpos = 0 # XXX: is this really correct? result = 0 - L.lineStart = result -proc handleCR(L: var BaseLexer, pos: int): int = +proc handleCR*(L: var BaseLexer, pos: int): int = + ## Call this if you scanned over '\c' in the buffer; it returns the the + ## position to continue the scanning from. `pos` must be the position + ## of the '\c'. assert(L.buf[pos] == '\c') inc(L.lineNumber) result = fillBaseLexer(L, pos) if L.buf[result] == '\L': result = fillBaseLexer(L, result) + L.lineStart = result -proc handleLF(L: var BaseLexer, pos: int): int = +proc handleLF*(L: var BaseLexer, pos: int): int = + ## Call this if you scanned over '\L' in the buffer; it returns the the + ## position to continue the scanning from. `pos` must be the position + ## of the '\L'. assert(L.buf[pos] == '\L') inc(L.lineNumber) result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result; + L.lineStart = result + +proc handleRefillChar*(L: var BaseLexer, pos: int): int = + ## To be documented. + assert(L.buf[pos] in L.refillChars) + result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result; proc skipUtf8Bom(L: var BaseLexer) = if (L.buf[0] == '\xEF') and (L.buf[1] == '\xBB') and (L.buf[2] == '\xBF'): inc(L.bufpos, 3) inc(L.lineStart, 3) -proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) = +proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192; + refillChars: set[char] = NewLines) = + ## inits the BaseLexer with a stream to read from. assert(bufLen > 0) assert(input != nil) L.input = input L.bufpos = 0 L.bufLen = bufLen + L.refillChars = refillChars L.buf = cast[cstring](alloc(bufLen * chrSize)) L.sentinel = bufLen - 1 L.lineStart = 0 @@ -153,10 +146,12 @@ proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) = fillBuffer(L) skipUtf8Bom(L) -proc getColNumber(L: BaseLexer, pos: int): int = +proc getColNumber*(L: BaseLexer, pos: int): int = + ## retrieves the current column. result = abs(pos - L.lineStart) -proc getCurrentLine(L: BaseLexer, marker: bool = true): string = +proc getCurrentLine*(L: BaseLexer, marker: bool = true): string = + ## retrieves the current line. var i: int result = "" i = L.lineStart @@ -166,4 +161,3 @@ proc getCurrentLine(L: BaseLexer, marker: bool = true): string = add(result, "\n") if marker: add(result, spaces(getColNumber(L, L.bufpos)) & "^\n") - diff --git a/lib/pure/parsexml.nim b/lib/pure/parsexml.nim index eb792f086..e1abb0a4f 100644 --- a/lib/pure/parsexml.nim +++ b/lib/pure/parsexml.nim @@ -8,19 +8,19 @@ # ## This module implements a simple high performance `XML`:idx: / `HTML`:idx: -## parser. +## parser. ## The only encoding that is supported is UTF-8. The parser has been designed -## to be somewhat error correcting, so that even most "wild HTML" found on the +## to be somewhat error correcting, so that even most "wild HTML" found on the ## web can be parsed with it. **Note:** This parser does not check that each -## ```` has a corresponding ````! These checks have do be -## implemented by the client code for various reasons: +## ```` has a corresponding ````! These checks have do be +## implemented by the client code for various reasons: ## ## * Old HTML contains tags that have no end tag: ``
`` for example. ## * HTML tags are case insensitive, XML tags are case sensitive. Since this ## library can parse both, only the client knows which comparison is to be ## used. ## * Thus the checks would have been very difficult to implement properly with -## little benefit, especially since they are simple to implement in the +## little benefit, especially since they are simple to implement in the ## client. The client should use the `errorMsgExpected` proc to generate ## a nice error message that fits the other error messages this library ## creates. @@ -29,7 +29,7 @@ ## Example 1: Retrieve HTML title ## ============================== ## -## The file ``examples/htmltitle.nim`` demonstrates how to use the +## The file ``examples/htmltitle.nim`` demonstrates how to use the ## XML parser to accomplish a simple task: To determine the title of an HTML ## document. ## @@ -40,22 +40,22 @@ ## Example 2: Retrieve all HTML links ## ================================== ## -## The file ``examples/htmlrefs.nim`` demonstrates how to use the -## XML parser to accomplish another simple task: To determine all the links +## The file ``examples/htmlrefs.nim`` demonstrates how to use the +## XML parser to accomplish another simple task: To determine all the links ## an HTML document contains. ## ## .. code-block:: nim ## :file: examples/htmlrefs.nim ## -import +import hashes, strutils, lexbase, streams, unicode # the parser treats ``
`` as ``

`` -# xmlElementCloseEnd, ## ``/>`` +# xmlElementCloseEnd, ## ``/>`` -type +type XmlEventKind* = enum ## enumation of all events that may occur when parsing xmlError, ## an error occurred during parsing xmlEof, ## end of file reached @@ -65,13 +65,13 @@ type xmlPI, ## processing instruction (````) xmlElementStart, ## ```` xmlElementEnd, ## ```` - xmlElementOpen, ## ```` + xmlElementClose, ## ``>`` xmlCData, ## ```` xmlEntity, ## &entity; xmlSpecial ## ```` - + XmlErrorKind* = enum ## enumeration that lists all errors that can occur errNone, ## no error errEndOfCDataExpected, ## ``]]>`` expected @@ -82,8 +82,8 @@ type errEqExpected, ## ``=`` expected errQuoteExpected, ## ``"`` or ``'`` expected errEndOfCommentExpected ## ``-->`` expected - - ParserState = enum + + ParserState = enum stateStart, stateNormal, stateAttr, stateEmptyElementTag, stateError XmlParseOption* = enum ## options for the XML parser @@ -121,8 +121,8 @@ proc open*(my: var XmlParser, input: Stream, filename: string, ## the `options` parameter: If `options` contains ``reportWhitespace`` ## a whitespace token is reported as an ``xmlWhitespace`` event. ## If `options` contains ``reportComments`` a comment token is reported as an - ## ``xmlComment`` event. - lexbase.open(my, input) + ## ``xmlComment`` event. + lexbase.open(my, input, 8192, {'\c', '\L', '/'}) my.filename = filename my.state = stateStart my.kind = xmlError @@ -130,24 +130,24 @@ proc open*(my: var XmlParser, input: Stream, filename: string, my.b = "" my.c = nil my.options = options - -proc close*(my: var XmlParser) {.inline.} = + +proc close*(my: var XmlParser) {.inline.} = ## closes the parser `my` and its associated input stream. lexbase.close(my) -proc kind*(my: XmlParser): XmlEventKind {.inline.} = +proc kind*(my: XmlParser): XmlEventKind {.inline.} = ## returns the current event type for the XML parser return my.kind template charData*(my: XmlParser): string = - ## returns the character data for the events: ``xmlCharData``, + ## returns the character data for the events: ``xmlCharData``, ## ``xmlWhitespace``, ``xmlComment``, ``xmlCData``, ``xmlSpecial`` - assert(my.kind in {xmlCharData, xmlWhitespace, xmlComment, xmlCData, + assert(my.kind in {xmlCharData, xmlWhitespace, xmlComment, xmlCData, xmlSpecial}) my.a template elementName*(my: XmlParser): string = - ## returns the element name for the events: ``xmlElementStart``, + ## returns the element name for the events: ``xmlElementStart``, ## ``xmlElementEnd``, ``xmlElementOpen`` assert(my.kind in {xmlElementStart, xmlElementEnd, xmlElementOpen}) my.a @@ -156,12 +156,12 @@ template entityName*(my: XmlParser): string = ## returns the entity name for the event: ``xmlEntity`` assert(my.kind == xmlEntity) my.a - + template attrKey*(my: XmlParser): string = ## returns the attribute key for the event ``xmlAttribute`` assert(my.kind == xmlAttribute) my.a - + template attrValue*(my: XmlParser): string = ## returns the attribute value for the event ``xmlAttribute`` assert(my.kind == xmlAttribute) @@ -187,110 +187,118 @@ proc rawData2*(my: XmlParser): string {.inline.} = ## This is only used for speed hacks. shallowCopy(result, my.b) -proc getColumn*(my: XmlParser): int {.inline.} = +proc getColumn*(my: XmlParser): int {.inline.} = ## get the current column the parser has arrived at. result = getColNumber(my, my.bufpos) -proc getLine*(my: XmlParser): int {.inline.} = +proc getLine*(my: XmlParser): int {.inline.} = ## get the current line the parser has arrived at. result = my.lineNumber -proc getFilename*(my: XmlParser): string {.inline.} = +proc getFilename*(my: XmlParser): string {.inline.} = ## get the filename of the file that the parser processes. result = my.filename - -proc errorMsg*(my: XmlParser): string = + +proc errorMsg*(my: XmlParser): string = ## returns a helpful error message for the event ``xmlError`` assert(my.kind == xmlError) result = "$1($2, $3) Error: $4" % [ my.filename, $getLine(my), $getColumn(my), errorMessages[my.err]] -proc errorMsgExpected*(my: XmlParser, tag: string): string = +proc errorMsgExpected*(my: XmlParser, tag: string): string = ## returns an error message " expected" in the same format as the - ## other error messages + ## other error messages result = "$1($2, $3) Error: $4" % [ my.filename, $getLine(my), $getColumn(my), "<$1> expected" % tag] -proc errorMsg*(my: XmlParser, msg: string): string = +proc errorMsg*(my: XmlParser, msg: string): string = ## returns an error message with text `msg` in the same format as the - ## other error messages + ## other error messages result = "$1($2, $3) Error: $4" % [ my.filename, $getLine(my), $getColumn(my), msg] - -proc markError(my: var XmlParser, kind: XmlErrorKind) {.inline.} = + +proc markError(my: var XmlParser, kind: XmlErrorKind) {.inline.} = my.err = kind my.state = stateError -proc parseCDATA(my: var XmlParser) = +proc parseCDATA(my: var XmlParser) = var pos = my.bufpos + len("': inc(pos, 3) break add(my.a, ']') inc(pos) - of '\0': + of '\0': markError(my, errEndOfCDataExpected) break - of '\c': + of '\c': pos = lexbase.handleCR(my, pos) buf = my.buf add(my.a, '\L') - of '\L': + of '\L': pos = lexbase.handleLF(my, pos) buf = my.buf add(my.a, '\L') + of '/': + pos = lexbase.handleRefillChar(my, pos) + buf = my.buf + add(my.a, '/') else: add(my.a, buf[pos]) - inc(pos) + inc(pos) my.bufpos = pos # store back my.kind = xmlCData -proc parseComment(my: var XmlParser) = +proc parseComment(my: var XmlParser) = var pos = my.bufpos + len("