This commit is contained in:
Araq 2015-07-01 15:47:15 +02:00
commit 0d7e0e1b4f
2 changed files with 178 additions and 156 deletions

View file

@ -34,37 +34,15 @@ type
lineNumber*: int ## the current line number lineNumber*: int ## the current line number
sentinel: int sentinel: int
lineStart: int # index of last line start in buffer lineStart: int # index of last line start in buffer
fileOpened: bool refillChars: set[char]
{.deprecated: [TBaseLexer: BaseLexer].} {.deprecated: [TBaseLexer: BaseLexer].}
proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192)
## inits the BaseLexer with a stream to read from
proc close*(L: var BaseLexer)
## closes the base lexer. This closes `L`'s associated stream too.
proc getCurrentLine*(L: BaseLexer, marker: bool = true): string
## retrieves the current line.
proc getColNumber*(L: BaseLexer, pos: int): int
## retrieves the current column.
proc handleCR*(L: var BaseLexer, pos: int): int
## Call this if you scanned over '\c' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position
## of the '\c'.
proc handleLF*(L: var BaseLexer, pos: int): int
## Call this if you scanned over '\L' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position
## of the '\L'.
# implementation
const const
chrSize = sizeof(char) chrSize = sizeof(char)
proc close(L: var BaseLexer) = proc close*(L: var BaseLexer) =
## closes the base lexer. This closes `L`'s associated stream too.
dealloc(L.buf) dealloc(L.buf)
close(L.input) close(L.input)
@ -93,7 +71,7 @@ proc fillBuffer(L: var BaseLexer) =
dec(s) # BUGFIX (valgrind) dec(s) # BUGFIX (valgrind)
while true: while true:
assert(s < L.bufLen) assert(s < L.bufLen)
while (s >= 0) and not (L.buf[s] in NewLines): dec(s) while s >= 0 and L.buf[s] notin L.refillChars: dec(s)
if s >= 0: if s >= 0:
# we found an appropriate character for a sentinel: # we found an appropriate character for a sentinel:
L.sentinel = s L.sentinel = s
@ -121,31 +99,46 @@ proc fillBaseLexer(L: var BaseLexer, pos: int): int =
fillBuffer(L) fillBuffer(L)
L.bufpos = 0 # XXX: is this really correct? L.bufpos = 0 # XXX: is this really correct?
result = 0 result = 0
L.lineStart = result
proc handleCR(L: var BaseLexer, pos: int): int = proc handleCR*(L: var BaseLexer, pos: int): int =
## Call this if you scanned over '\c' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position
## of the '\c'.
assert(L.buf[pos] == '\c') assert(L.buf[pos] == '\c')
inc(L.lineNumber) inc(L.lineNumber)
result = fillBaseLexer(L, pos) result = fillBaseLexer(L, pos)
if L.buf[result] == '\L': if L.buf[result] == '\L':
result = fillBaseLexer(L, result) result = fillBaseLexer(L, result)
L.lineStart = result
proc handleLF(L: var BaseLexer, pos: int): int = proc handleLF*(L: var BaseLexer, pos: int): int =
## Call this if you scanned over '\L' in the buffer; it returns the the
## position to continue the scanning from. `pos` must be the position
## of the '\L'.
assert(L.buf[pos] == '\L') assert(L.buf[pos] == '\L')
inc(L.lineNumber) inc(L.lineNumber)
result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result; result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result;
L.lineStart = result
proc handleRefillChar*(L: var BaseLexer, pos: int): int =
## To be documented.
assert(L.buf[pos] in L.refillChars)
result = fillBaseLexer(L, pos) #L.lastNL := result-1; // BUGFIX: was: result;
proc skipUtf8Bom(L: var BaseLexer) = proc skipUtf8Bom(L: var BaseLexer) =
if (L.buf[0] == '\xEF') and (L.buf[1] == '\xBB') and (L.buf[2] == '\xBF'): if (L.buf[0] == '\xEF') and (L.buf[1] == '\xBB') and (L.buf[2] == '\xBF'):
inc(L.bufpos, 3) inc(L.bufpos, 3)
inc(L.lineStart, 3) inc(L.lineStart, 3)
proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) = proc open*(L: var BaseLexer, input: Stream, bufLen: int = 8192;
refillChars: set[char] = NewLines) =
## inits the BaseLexer with a stream to read from.
assert(bufLen > 0) assert(bufLen > 0)
assert(input != nil) assert(input != nil)
L.input = input L.input = input
L.bufpos = 0 L.bufpos = 0
L.bufLen = bufLen L.bufLen = bufLen
L.refillChars = refillChars
L.buf = cast[cstring](alloc(bufLen * chrSize)) L.buf = cast[cstring](alloc(bufLen * chrSize))
L.sentinel = bufLen - 1 L.sentinel = bufLen - 1
L.lineStart = 0 L.lineStart = 0
@ -153,10 +146,12 @@ proc open(L: var BaseLexer, input: Stream, bufLen: int = 8192) =
fillBuffer(L) fillBuffer(L)
skipUtf8Bom(L) skipUtf8Bom(L)
proc getColNumber(L: BaseLexer, pos: int): int = proc getColNumber*(L: BaseLexer, pos: int): int =
## retrieves the current column.
result = abs(pos - L.lineStart) result = abs(pos - L.lineStart)
proc getCurrentLine(L: BaseLexer, marker: bool = true): string = proc getCurrentLine*(L: BaseLexer, marker: bool = true): string =
## retrieves the current line.
var i: int var i: int
result = "" result = ""
i = L.lineStart i = L.lineStart
@ -166,4 +161,3 @@ proc getCurrentLine(L: BaseLexer, marker: bool = true): string =
add(result, "\n") add(result, "\n")
if marker: if marker:
add(result, spaces(getColNumber(L, L.bufpos)) & "^\n") add(result, spaces(getColNumber(L, L.bufpos)) & "^\n")

View file

@ -122,7 +122,7 @@ proc open*(my: var XmlParser, input: Stream, filename: string,
## a whitespace token is reported as an ``xmlWhitespace`` event. ## a whitespace token is reported as an ``xmlWhitespace`` event.
## If `options` contains ``reportComments`` a comment token is reported as an ## If `options` contains ``reportComments`` a comment token is reported as an
## ``xmlComment`` event. ## ``xmlComment`` event.
lexbase.open(my, input) lexbase.open(my, input, 8192, {'\c', '\L', '/'})
my.filename = filename my.filename = filename
my.state = stateStart my.state = stateStart
my.kind = xmlError my.kind = xmlError
@ -243,6 +243,10 @@ proc parseCDATA(my: var XmlParser) =
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
add(my.a, '/')
else: else:
add(my.a, buf[pos]) add(my.a, buf[pos])
inc(pos) inc(pos)
@ -271,6 +275,10 @@ proc parseComment(my: var XmlParser) =
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
if my.options.contains(reportComments): add(my.a, '\L') if my.options.contains(reportComments): add(my.a, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
if my.options.contains(reportComments): add(my.a, '/')
else: else:
if my.options.contains(reportComments): add(my.a, buf[pos]) if my.options.contains(reportComments): add(my.a, buf[pos])
inc(pos) inc(pos)
@ -393,6 +401,10 @@ proc parsePI(my: var XmlParser) =
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
add(my.b, '\L') add(my.b, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
add(my.b, '/')
else: else:
add(my.b, buf[pos]) add(my.b, buf[pos])
inc(pos) inc(pos)
@ -428,6 +440,10 @@ proc parseSpecial(my: var XmlParser) =
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
add(my.b, '/')
else: else:
add(my.a, buf[pos]) add(my.a, buf[pos])
inc(pos) inc(pos)
@ -450,8 +466,11 @@ proc parseTag(my: var XmlParser) =
my.c = my.a # save for later my.c = my.a # save for later
else: else:
my.kind = xmlElementStart my.kind = xmlElementStart
if my.buf[my.bufpos] == '/' and my.buf[my.bufpos+1] == '>': let slash = my.buf[my.bufpos] == '/'
inc(my.bufpos, 2) if slash:
my.bufpos = lexbase.handleRefillChar(my, my.bufpos)
if slash and my.buf[my.bufpos] == '>':
inc(my.bufpos)
my.state = stateEmptyElementTag my.state = stateEmptyElementTag
my.c = nil my.c = nil
elif my.buf[my.bufpos] == '>': elif my.buf[my.bufpos] == '>':
@ -460,7 +479,8 @@ proc parseTag(my: var XmlParser) =
markError(my, errGtExpected) markError(my, errGtExpected)
proc parseEndTag(my: var XmlParser) = proc parseEndTag(my: var XmlParser) =
inc(my.bufpos, 2) my.bufpos = lexbase.handleRefillChar(my, my.bufpos+1)
#inc(my.bufpos, 2)
parseName(my, my.a) parseName(my, my.a)
parseWhitespace(my, skip=true) parseWhitespace(my, skip=true)
if my.buf[my.bufpos] == '>': if my.buf[my.bufpos] == '>':
@ -545,6 +565,10 @@ proc parseCharData(my: var XmlParser) =
pos = lexbase.handleLF(my, pos) pos = lexbase.handleLF(my, pos)
buf = my.buf buf = my.buf
add(my.a, '\L') add(my.a, '\L')
of '/':
pos = lexbase.handleRefillChar(my, pos)
buf = my.buf
add(my.a, '/')
else: else:
add(my.a, buf[pos]) add(my.a, buf[pos])
inc(pos) inc(pos)
@ -612,10 +636,14 @@ proc next*(my: var XmlParser) =
my.kind = xmlElementClose my.kind = xmlElementClose
inc(my.bufpos) inc(my.bufpos)
my.state = stateNormal my.state = stateNormal
elif my.buf[my.bufpos] == '/' and my.buf[my.bufpos+1] == '>': elif my.buf[my.bufpos] == '/':
my.kind = xmlElementClose my.bufpos = lexbase.handleRefillChar(my, my.bufpos)
inc(my.bufpos, 2) if my.buf[my.bufpos] == '>':
my.state = stateEmptyElementTag my.kind = xmlElementClose
inc(my.bufpos)
my.state = stateEmptyElementTag
else:
markError(my, errGtExpected)
else: else:
parseAttribute(my) parseAttribute(my)
# state remains the same # state remains the same