more modules updated

This commit is contained in:
Araq 2014-08-28 01:41:41 +02:00
commit e07b833439
2 changed files with 59 additions and 59 deletions

View file

@ -30,7 +30,7 @@ type
bufpos*: int ## the current position within the buffer bufpos*: int ## the current position within the buffer
buf*: cstring ## the buffer itself buf*: cstring ## the buffer itself
bufLen*: int ## length of buffer in characters bufLen*: int ## length of buffer in characters
input: PStream ## the input stream input: Stream ## the input stream
lineNumber*: int ## the current line number lineNumber*: int ## the current line number
sentinel: int sentinel: int
lineStart: int # index of last line start in buffer lineStart: int # index of last line start in buffer
@ -38,23 +38,23 @@ type
{.deprecated: [TBaseLexer: BaseLexer].} {.deprecated: [TBaseLexer: BaseLexer].}
proc open*(L: var TBaseLexer, input: PStream, bufLen: int = 8192) proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192)
## inits the TBaseLexer with a stream to read from ## inits the TBaseLexer with a stream to read from
proc close*(L: var TBaseLexer) proc close*(L: var BaseLexer)
## closes the base lexer. This closes `L`'s associated stream too. ## closes the base lexer. This closes `L`'s associated stream too.
proc getCurrentLine*(L: TBaseLexer, marker: bool = true): string proc getCurrentLine*(L: BaseLexer, marker: bool = true): string
## retrieves the current line. ## retrieves the current line.
proc getColNumber*(L: TBaseLexer, pos: int): int proc getColNumber*(L: BaseLexer, pos: int): int
## retrieves the current column. ## retrieves the current column.
proc handleCR*(L: var TBaseLexer, pos: int): int proc handleCR*(L: var BaseLexer, pos: int): int
## Call this if you scanned over '\c' in the buffer; it returns the the ## Call this if you scanned over '\c' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position ## position to continue the scanning from. `pos` must be the position
## of the '\c'. ## of the '\c'.
proc handleLF*(L: var TBaseLexer, pos: int): int proc handleLF*(L: var BaseLexer, pos: int): int
## Call this if you scanned over '\L' in the buffer; it returns the the ## Call this if you scanned over '\L' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position ## position to continue the scanning from. `pos` must be the position
## of the '\L'. ## of the '\L'.
@ -64,11 +64,11 @@ proc handleLF*(L: var TBaseLexer, pos: int): int
const const
chrSize = sizeof(char) chrSize = sizeof(char)
proc close(L: var TBaseLexer) = proc close(L: var BaseLexer) =
dealloc(L.buf) dealloc(L.buf)
close(L.input) close(L.input)
proc fillBuffer(L: var TBaseLexer) = proc fillBuffer(L: var BaseLexer) =
var var
charsRead, toCopy, s: int # all are in characters, charsRead, toCopy, s: int # all are in characters,
# not bytes (in case this # not bytes (in case this
@ -113,7 +113,7 @@ proc fillBuffer(L: var TBaseLexer) =
break break
s = L.bufLen - 1 s = L.bufLen - 1
proc fillBaseLexer(L: var TBaseLexer, pos: int): int = proc fillBaseLexer(L: var BaseLexer, pos: int): int =
assert(pos <= L.sentinel) assert(pos <= L.sentinel)
if pos < L.sentinel: if pos < L.sentinel:
result = pos + 1 # nothing to do result = pos + 1 # nothing to do
@ -123,24 +123,24 @@ proc fillBaseLexer(L: var TBaseLexer, pos: int): int =
result = 0 result = 0
L.lineStart = result L.lineStart = result
proc handleCR(L: var TBaseLexer, pos: int): int = proc handleCR(L: var BaseLexer, pos: int): int =
assert(L.buf[pos] == '\c') assert(L.buf[pos] == '\c')
inc(L.lineNumber) inc(L.lineNumber)
result = fillBaseLexer(L, pos) result = fillBaseLexer(L, pos)
if L.buf[result] == '\L': if L.buf[result] == '\L':
result = fillBaseLexer(L, result) result = fillBaseLexer(L, result)
proc handleLF(L: var TBaseLexer, pos: int): int = proc handleLF(L: var BaseLexer, pos: int): int =
assert(L.buf[pos] == '\L') assert(L.buf[pos] == '\L')
inc(L.lineNumber) inc(L.lineNumber)
result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result; result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result;
proc skipUtf8Bom(L: var TBaseLexer) = proc skipUtf8Bom(L: var BaseLexer) =
if (L.buf[0] == '\xEF') and (L.buf[1] == '\xBB') and (L.buf[2] == '\xBF'): if (L.buf[0] == '\xEF') and (L.buf[1] == '\xBB') and (L.buf[2] == '\xBF'):
inc(L.bufpos, 3) inc(L.bufpos, 3)
inc(L.lineStart, 3) inc(L.lineStart, 3)
proc open(L: var TBaseLexer, input: PStream, bufLen: int = 8192) = proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) =
assert(bufLen > 0) assert(bufLen > 0)
assert(input != nil) assert(input != nil)
L.input = input L.input = input
@ -153,10 +153,10 @@ proc open(L: var TBaseLexer, input: PStream, bufLen: int = 8192) =
fillBuffer(L) fillBuffer(L)
skipUtf8Bom(L) skipUtf8Bom(L)
proc getColNumber(L: TBaseLexer, pos: int): int = proc getColNumber(L: BaseLexer, pos: int): int =
result = abs(pos - L.lineStart) result = abs(pos - L.lineStart)
proc getCurrentLine(L: TBaseLexer, marker: bool = true): string = proc getCurrentLine(L: BaseLexer, marker: bool = true): string =
var i: int var i: int
result = "" result = ""
i = L.lineStart i = L.lineStart

View file

@ -90,7 +90,7 @@ type
reportWhitespace, ## report whitespace reportWhitespace, ## report whitespace
reportComments ## report comments reportComments ## report comments
XmlParser* = object of TBaseLexer ## the parser object. XmlParser* = object of BaseLexer ## the parser object.
a, b, c: string a, b, c: string
kind: XmlEventKind kind: XmlEventKind
err: XmlErrorKind err: XmlErrorKind
@ -134,93 +134,93 @@ proc close*(my: var XmlParser) {.inline.} =
## closes the parser `my` and its associated input stream. ## closes the parser `my` and its associated input stream.
lexbase.close(my) lexbase.close(my)
proc kind*(my: TXmlParser): TXmlEventKind {.inline.} = proc kind*(my: XmlParser): XmlEventKind {.inline.} =
## returns the current event type for the XML parser ## returns the current event type for the XML parser
return my.kind return my.kind
proc charData*(my: TXmlParser): string {.inline.} = proc charData*(my: XmlParser): string {.inline.} =
## returns the character data for the events: ``xmlCharData``, ## returns the character data for the events: ``xmlCharData``,
## ``xmlWhitespace``, ``xmlComment``, ``xmlCData``, ``xmlSpecial`` ## ``xmlWhitespace``, ``xmlComment``, ``xmlCData``, ``xmlSpecial``
assert(my.kind in {xmlCharData, xmlWhitespace, xmlComment, xmlCData, assert(my.kind in {xmlCharData, xmlWhitespace, xmlComment, xmlCData,
xmlSpecial}) xmlSpecial})
return my.a return my.a
proc elementName*(my: TXmlParser): string {.inline.} = proc elementName*(my: XmlParser): string {.inline.} =
## returns the element name for the events: ``xmlElementStart``, ## returns the element name for the events: ``xmlElementStart``,
## ``xmlElementEnd``, ``xmlElementOpen`` ## ``xmlElementEnd``, ``xmlElementOpen``
assert(my.kind in {xmlElementStart, xmlElementEnd, xmlElementOpen}) assert(my.kind in {xmlElementStart, xmlElementEnd, xmlElementOpen})
return my.a return my.a
proc entityName*(my: TXmlParser): string {.inline.} = proc entityName*(my: XmlParser): string {.inline.} =
## returns the entity name for the event: ``xmlEntity`` ## returns the entity name for the event: ``xmlEntity``
assert(my.kind == xmlEntity) assert(my.kind == xmlEntity)
return my.a return my.a
proc attrKey*(my: TXmlParser): string {.inline.} = proc attrKey*(my: XmlParser): string {.inline.} =
## returns the attribute key for the event ``xmlAttribute`` ## returns the attribute key for the event ``xmlAttribute``
assert(my.kind == xmlAttribute) assert(my.kind == xmlAttribute)
return my.a return my.a
proc attrValue*(my: TXmlParser): string {.inline.} = proc attrValue*(my: XmlParser): string {.inline.} =
## returns the attribute value for the event ``xmlAttribute`` ## returns the attribute value for the event ``xmlAttribute``
assert(my.kind == xmlAttribute) assert(my.kind == xmlAttribute)
return my.b return my.b
proc PIName*(my: TXmlParser): string {.inline.} = proc PIName*(my: XmlParser): string {.inline.} =
## returns the processing instruction name for the event ``xmlPI`` ## returns the processing instruction name for the event ``xmlPI``
assert(my.kind == xmlPI) assert(my.kind == xmlPI)
return my.a return my.a
proc PIRest*(my: TXmlParser): string {.inline.} = proc PIRest*(my: XmlParser): string {.inline.} =
## returns the rest of the processing instruction for the event ``xmlPI`` ## returns the rest of the processing instruction for the event ``xmlPI``
assert(my.kind == xmlPI) assert(my.kind == xmlPI)
return my.b return my.b
proc rawData*(my: TXmlParser): string {.inline.} = proc rawData*(my: XmlParser): string {.inline.} =
## returns the underlying 'data' string by reference. ## returns the underlying 'data' string by reference.
## This is only used for speed hacks. ## This is only used for speed hacks.
shallowCopy(result, my.a) shallowCopy(result, my.a)
proc rawData2*(my: TXmlParser): string {.inline.} = proc rawData2*(my: XmlParser): string {.inline.} =
## returns the underlying second 'data' string by reference. ## returns the underlying second 'data' string by reference.
## This is only used for speed hacks. ## This is only used for speed hacks.
shallowCopy(result, my.b) shallowCopy(result, my.b)
proc getColumn*(my: TXmlParser): int {.inline.} = proc getColumn*(my: XmlParser): int {.inline.} =
## get the current column the parser has arrived at. ## get the current column the parser has arrived at.
result = getColNumber(my, my.bufPos) result = getColNumber(my, my.bufpos)
proc getLine*(my: TXmlParser): int {.inline.} = proc getLine*(my: XmlParser): int {.inline.} =
## get the current line the parser has arrived at. ## get the current line the parser has arrived at.
result = my.linenumber result = my.lineNumber
proc getFilename*(my: TXmlParser): string {.inline.} = proc getFilename*(my: XmlParser): string {.inline.} =
## get the filename of the file that the parser processes. ## get the filename of the file that the parser processes.
result = my.filename result = my.filename
proc errorMsg*(my: TXmlParser): string = proc errorMsg*(my: XmlParser): string =
## returns a helpful error message for the event ``xmlError`` ## returns a helpful error message for the event ``xmlError``
assert(my.kind == xmlError) assert(my.kind == xmlError)
result = "$1($2, $3) Error: $4" % [ result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), errorMessages[my.err]] my.filename, $getLine(my), $getColumn(my), errorMessages[my.err]]
proc errorMsgExpected*(my: TXmlParser, tag: string): string = proc errorMsgExpected*(my: XmlParser, tag: string): string =
## returns an error message "<tag> expected" in the same format as the ## returns an error message "<tag> expected" in the same format as the
## other error messages ## other error messages
result = "$1($2, $3) Error: $4" % [ result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), "<$1> expected" % tag] my.filename, $getLine(my), $getColumn(my), "<$1> expected" % tag]
proc errorMsg*(my: TXmlParser, msg: string): string = proc errorMsg*(my: XmlParser, msg: string): string =
## returns an error message with text `msg` in the same format as the ## returns an error message with text `msg` in the same format as the
## other error messages ## other error messages
result = "$1($2, $3) Error: $4" % [ result = "$1($2, $3) Error: $4" % [
my.filename, $getLine(my), $getColumn(my), msg] my.filename, $getLine(my), $getColumn(my), msg]
proc markError(my: var TXmlParser, kind: TXmlError) {.inline.} = proc markError(my: var XmlParser, kind: XmlErrorKind) {.inline.} =
my.err = kind my.err = kind
my.state = stateError my.state = stateError
proc parseCDATA(my: var TXMLParser) = proc parseCDATA(my: var XmlParser) =
var pos = my.bufpos + len("<![CDATA[") var pos = my.bufpos + len("<![CDATA[")
var buf = my.buf var buf = my.buf
while true: while true:
@ -246,9 +246,9 @@ proc parseCDATA(my: var TXMLParser) =
add(my.a, buf[pos]) add(my.a, buf[pos])
inc(pos) inc(pos)
my.bufpos = pos # store back my.bufpos = pos # store back
my.kind = xmlCDATA my.kind = xmlCData
proc parseComment(my: var TXMLParser) = proc parseComment(my: var XmlParser) =
var pos = my.bufpos + len("<!--") var pos = my.bufpos + len("<!--")
var buf = my.buf var buf = my.buf
while true: while true:
@ -276,7 +276,7 @@ proc parseComment(my: var TXMLParser) =
my.bufpos = pos my.bufpos = pos
my.kind = xmlComment my.kind = xmlComment
proc parseWhitespace(my: var TXmlParser, skip=False) = proc parseWhitespace(my: var XmlParser, skip=False) =
var pos = my.bufpos var pos = my.bufpos
var buf = my.buf var buf = my.buf
while true: while true:
@ -301,7 +301,7 @@ const
NameStartChar = {'A'..'Z', 'a'..'z', '_', ':', '\128'..'\255'} NameStartChar = {'A'..'Z', 'a'..'z', '_', ':', '\128'..'\255'}
NameChar = {'A'..'Z', 'a'..'z', '0'..'9', '.', '-', '_', ':', '\128'..'\255'} NameChar = {'A'..'Z', 'a'..'z', '0'..'9', '.', '-', '_', ':', '\128'..'\255'}
proc parseName(my: var TXmlParser, dest: var string) = proc parseName(my: var XmlParser, dest: var string) =
var pos = my.bufpos var pos = my.bufpos
var buf = my.buf var buf = my.buf
if buf[pos] in nameStartChar: if buf[pos] in nameStartChar:
@ -313,7 +313,7 @@ proc parseName(my: var TXmlParser, dest: var string) =
else: else:
markError(my, errNameExpected) markError(my, errNameExpected)
proc parseEntity(my: var TXmlParser, dest: var string) = proc parseEntity(my: var XmlParser, dest: var string) =
var pos = my.bufpos+1 var pos = my.bufpos+1
var buf = my.buf var buf = my.buf
my.kind = xmlCharData my.kind = xmlCharData
@ -333,7 +333,7 @@ proc parseEntity(my: var TXmlParser, dest: var string) =
while buf[pos] in {'0'..'9'}: while buf[pos] in {'0'..'9'}:
r = r * 10 + (ord(buf[pos]) - ord('0')) r = r * 10 + (ord(buf[pos]) - ord('0'))
inc(pos) inc(pos)
add(dest, toUTF8(TRune(r))) add(dest, toUTF8(Rune(r)))
elif buf[pos] == 'l' and buf[pos+1] == 't' and buf[pos+2] == ';': elif buf[pos] == 'l' and buf[pos+1] == 't' and buf[pos+2] == ';':
add(dest, '<') add(dest, '<')
inc(pos, 2) inc(pos, 2)
@ -363,10 +363,10 @@ proc parseEntity(my: var TXmlParser, dest: var string) =
if buf[pos] == ';': if buf[pos] == ';':
inc(pos) inc(pos)
else: else:
markError(my, errSemiColonExpected) markError(my, errSemicolonExpected)
my.bufpos = pos my.bufpos = pos
proc parsePI(my: var TXmlParser) = proc parsePI(my: var XmlParser) =
inc(my.bufpos, "<?".len) inc(my.bufpos, "<?".len)
parseName(my, my.a) parseName(my, my.a)
var pos = my.bufpos var pos = my.bufpos
@ -398,7 +398,7 @@ proc parsePI(my: var TXmlParser) =
my.bufpos = pos my.bufpos = pos
my.kind = xmlPI my.kind = xmlPI
proc parseSpecial(my: var TXmlParser) = proc parseSpecial(my: var XmlParser) =
# things that start with <! # things that start with <!
var pos = my.bufpos + 2 var pos = my.bufpos + 2
var buf = my.buf var buf = my.buf
@ -433,7 +433,7 @@ proc parseSpecial(my: var TXmlParser) =
my.bufpos = pos my.bufpos = pos
my.kind = xmlSpecial my.kind = xmlSpecial
proc parseTag(my: var TXmlParser) = proc parseTag(my: var XmlParser) =
inc(my.bufpos) inc(my.bufpos)
parseName(my, my.a) parseName(my, my.a)
# if we have no name, do not interpret the '<': # if we have no name, do not interpret the '<':
@ -458,7 +458,7 @@ proc parseTag(my: var TXmlParser) =
else: else:
markError(my, errGtExpected) markError(my, errGtExpected)
proc parseEndTag(my: var TXmlParser) = proc parseEndTag(my: var XmlParser) =
inc(my.bufpos, 2) inc(my.bufpos, 2)
parseName(my, my.a) parseName(my, my.a)
parseWhitespace(my, skip=True) parseWhitespace(my, skip=True)
@ -468,7 +468,7 @@ proc parseEndTag(my: var TXmlParser) =
markError(my, errGtExpected) markError(my, errGtExpected)
my.kind = xmlElementEnd my.kind = xmlElementEnd
proc parseAttribute(my: var TXmlParser) = proc parseAttribute(my: var XmlParser) =
my.kind = xmlAttribute my.kind = xmlAttribute
setLen(my.a, 0) setLen(my.a, 0)
setLen(my.b, 0) setLen(my.b, 0)
@ -529,7 +529,7 @@ proc parseAttribute(my: var TXmlParser) =
my.bufpos = pos my.bufpos = pos
parseWhitespace(my, skip=True) parseWhitespace(my, skip=True)
proc parseCharData(my: var TXmlParser) = proc parseCharData(my: var XmlParser) =
var pos = my.bufpos var pos = my.bufpos
var buf = my.buf var buf = my.buf
while true: while true:
@ -550,7 +550,7 @@ proc parseCharData(my: var TXmlParser) =
my.bufpos = pos my.bufpos = pos
my.kind = xmlCharData my.kind = xmlCharData
proc rawGetTok(my: var TXmlParser) = proc rawGetTok(my: var XmlParser) =
my.kind = xmlError my.kind = xmlError
setLen(my.a, 0) setLen(my.a, 0)
var pos = my.bufpos var pos = my.bufpos
@ -574,7 +574,7 @@ proc rawGetTok(my: var TXmlParser) =
else: else:
parseTag(my) parseTag(my)
of ' ', '\t', '\c', '\l': of ' ', '\t', '\c', '\l':
parseWhiteSpace(my) parseWhitespace(my)
my.kind = xmlWhitespace my.kind = xmlWhitespace
of '\0': of '\0':
my.kind = xmlEof my.kind = xmlEof
@ -584,7 +584,7 @@ proc rawGetTok(my: var TXmlParser) =
parseCharData(my) parseCharData(my)
assert my.kind != xmlError assert my.kind != xmlError
proc getTok(my: var TXmlParser) = proc getTok(my: var XmlParser) =
while true: while true:
rawGetTok(my) rawGetTok(my)
case my.kind case my.kind
@ -594,7 +594,7 @@ proc getTok(my: var TXmlParser) =
if my.options.contains(reportWhitespace): break if my.options.contains(reportWhitespace): break
else: break else: break
proc next*(my: var TXmlParser) = proc next*(my: var XmlParser) =
## retrieves the first/next event. This controls the parser. ## retrieves the first/next event. This controls the parser.
case my.state case my.state
of stateNormal: of stateNormal:
@ -629,14 +629,14 @@ proc next*(my: var TXmlParser) =
when isMainModule: when isMainModule:
import os import os
var s = newFileStream(ParamStr(1), fmRead) var s = newFileStream(paramStr(1), fmRead)
if s == nil: quit("cannot open the file" & ParamStr(1)) if s == nil: quit("cannot open the file" & paramStr(1))
var x: TXmlParser var x: XmlParser
open(x, s, ParamStr(1)) open(x, s, paramStr(1))
while true: while true:
next(x) next(x)
case x.kind case x.kind
of xmlError: Echo(x.errorMsg()) of xmlError: echo(x.errorMsg())
of xmlEof: break of xmlEof: break
of xmlCharData: echo(x.charData) of xmlCharData: echo(x.charData)
of xmlWhitespace: echo("|$1|" % x.charData) of xmlWhitespace: echo("|$1|" % x.charData)