bugfixes for unicode; xmlparser; htmlparser; scanner
This commit is contained in:
parent
64da2f1681
commit
6bc16904ed
18 changed files with 226 additions and 74 deletions
|
|
@ -5,7 +5,7 @@ discard distinct div
|
||||||
elif else end enum except
|
elif else end enum except
|
||||||
finally for from generic
|
finally for from generic
|
||||||
if implies import in include is isnot iterator
|
if implies import in include is isnot iterator
|
||||||
lambda
|
lambda let
|
||||||
macro method mod
|
macro method mod
|
||||||
nil not notin
|
nil not notin
|
||||||
object of or out
|
object of or out
|
||||||
|
|
|
||||||
|
|
@ -177,7 +177,7 @@ XML Processing
|
||||||
* `xmltree <xmltree.html>`_
|
* `xmltree <xmltree.html>`_
|
||||||
A simple XML tree. More efficient and simpler than the DOM.
|
A simple XML tree. More efficient and simpler than the DOM.
|
||||||
|
|
||||||
* `xmltreeparser <xmltreeparser.html>`_
|
* `xmlparser <xmlparser.html>`_
|
||||||
This module parses an XML document and creates its XML tree representation.
|
This module parses an XML document and creates its XML tree representation.
|
||||||
|
|
||||||
* `htmlparser <htmlparser.html>`_
|
* `htmlparser <htmlparser.html>`_
|
||||||
|
|
|
||||||
4
koch.nim
4
koch.nim
|
|
@ -108,13 +108,13 @@ proc safeRemove(filename: string) =
|
||||||
proc bootIteration(args: string): bool =
|
proc bootIteration(args: string): bool =
|
||||||
var nimrod1 = "rod" / "nimrod1".exe
|
var nimrod1 = "rod" / "nimrod1".exe
|
||||||
safeRemove(nimrod1)
|
safeRemove(nimrod1)
|
||||||
moveFile(nimrod1, "rod" / "nimrod".exe)
|
moveFile(dest=nimrod1, source="rod" / "nimrod".exe)
|
||||||
exec "rod" / "nimrod1 cc $# $# rod/nimrod.nim" % [bootOptions, args]
|
exec "rod" / "nimrod1 cc $# $# rod/nimrod.nim" % [bootOptions, args]
|
||||||
# Nimrod does not produce an executable again if nothing changed. That's ok:
|
# Nimrod does not produce an executable again if nothing changed. That's ok:
|
||||||
result = sameFileContent("rod" / "nimrod".exe, nimrod1)
|
result = sameFileContent("rod" / "nimrod".exe, nimrod1)
|
||||||
safeRemove("bin" / "nimrod".exe)
|
safeRemove("bin" / "nimrod".exe)
|
||||||
var dest = "bin" / "nimrod".exe
|
var dest = "bin" / "nimrod".exe
|
||||||
copyFile(dest, "rod" / "nimrod".exe)
|
copyFile(dest=dest, source="rod" / "nimrod".exe)
|
||||||
inclFilePermissions(dest, {fpUserExec})
|
inclFilePermissions(dest, {fpUserExec})
|
||||||
safeRemove(nimrod1)
|
safeRemove(nimrod1)
|
||||||
if result: echo "executables are equal: SUCCESS!"
|
if result: echo "executables are equal: SUCCESS!"
|
||||||
|
|
|
||||||
|
|
@ -265,7 +265,7 @@ proc addNode(father, son: PXmlNode) =
|
||||||
proc parse(x: var TXmlParser, errors: var seq[string]): PXmlNode
|
proc parse(x: var TXmlParser, errors: var seq[string]): PXmlNode
|
||||||
|
|
||||||
proc expected(x: var TXmlParser, n: PXmlNode): string =
|
proc expected(x: var TXmlParser, n: PXmlNode): string =
|
||||||
result = errorMsg(x, "</" & n.tag & "$1> expected")
|
result = errorMsg(x, "</" & n.tag & "> expected")
|
||||||
|
|
||||||
proc untilElementEnd(x: var TXmlParser, result: PXmlNode,
|
proc untilElementEnd(x: var TXmlParser, result: PXmlNode,
|
||||||
errors: var seq[string]) =
|
errors: var seq[string]) =
|
||||||
|
|
@ -378,17 +378,19 @@ proc parseHtml*(s: PStream): PXmlNode =
|
||||||
var errors: seq[string] = @[]
|
var errors: seq[string] = @[]
|
||||||
result = parseHtml(s, "unknown_html_doc", errors)
|
result = parseHtml(s, "unknown_html_doc", errors)
|
||||||
|
|
||||||
proc loadHtml*(path: string, reportErrors = false): PXmlNode =
|
proc loadHtml*(path: string, errors: var seq[string]): PXmlNode =
|
||||||
## Loads and parses HTML from file specified by ``path``, and returns
|
## Loads and parses HTML from file specified by ``path``, and returns
|
||||||
## a ``PXmlNode``. If `reportErrors` is true, the parsing errors are
|
## a ``PXmlNode``. Every occured parsing error is added to
|
||||||
## ``echo``ed, otherwise they are ignored.
|
## the `errors` sequence.
|
||||||
var s = newFileStream(path, fmRead)
|
var s = newFileStream(path, fmRead)
|
||||||
if s == nil: raise newException(EIO, "Unable to read file: " & path)
|
if s == nil: raise newException(EIO, "Unable to read file: " & path)
|
||||||
|
|
||||||
var errors: seq[string] = @[]
|
|
||||||
result = parseHtml(s, path, errors)
|
result = parseHtml(s, path, errors)
|
||||||
if reportErrors:
|
|
||||||
for msg in items(errors): echo(msg)
|
proc loadHtml*(path: string): PXmlNode =
|
||||||
|
## Loads and parses HTML from file specified by ``path``, and returns
|
||||||
|
## a ``PXmlNode``. All parsing errors are ignored.
|
||||||
|
var errors: seq[string] = @[]
|
||||||
|
result = loadHtml(path, errors)
|
||||||
|
|
||||||
when true:
|
when true:
|
||||||
nil
|
nil
|
||||||
|
|
@ -403,3 +405,17 @@ else:
|
||||||
errors.add("<html> tag expected")
|
errors.add("<html> tag expected")
|
||||||
checkHtmlAux(n, errors)
|
checkHtmlAux(n, errors)
|
||||||
|
|
||||||
|
when isMainModule:
|
||||||
|
import os
|
||||||
|
|
||||||
|
var errors: seq[string] = @[]
|
||||||
|
var x = loadHtml(paramStr(1), errors)
|
||||||
|
for e in items(errors): echo e
|
||||||
|
|
||||||
|
var f: TFile
|
||||||
|
if open(f, "test.txt", fmWrite):
|
||||||
|
f.write($x)
|
||||||
|
f.close()
|
||||||
|
else:
|
||||||
|
quit("cannot write test.txt")
|
||||||
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
#
|
#
|
||||||
#
|
#
|
||||||
# Nimrod's Runtime Library
|
# Nimrod's Runtime Library
|
||||||
# (c) Copyright 2009 Andreas Rumpf
|
# (c) Copyright 2010 Andreas Rumpf
|
||||||
#
|
#
|
||||||
# See the file "copying.txt", included in this
|
# See the file "copying.txt", included in this
|
||||||
# distribution, for details about the copyright.
|
# distribution, for details about the copyright.
|
||||||
|
|
@ -619,9 +619,11 @@ proc sameFileContent*(path1, path2: string): bool =
|
||||||
close(a)
|
close(a)
|
||||||
close(b)
|
close(b)
|
||||||
|
|
||||||
proc copyFile*(dest, source: string) =
|
proc copyFile*(dest, source: string) {.deprecated.} =
|
||||||
## Copies a file from `source` to `dest`. If this fails,
|
## Copies a file from `source` to `dest`. If this fails,
|
||||||
## `EOS` is raised.
|
## `EOS` is raised.
|
||||||
|
## **Deprecated since version 0.8.8**: Use this proc with named arguments
|
||||||
|
## only, because the order will change!
|
||||||
when defined(Windows):
|
when defined(Windows):
|
||||||
if CopyFileA(source, dest, 0'i32) == 0'i32: OSError()
|
if CopyFileA(source, dest, 0'i32) == 0'i32: OSError()
|
||||||
else:
|
else:
|
||||||
|
|
@ -647,8 +649,10 @@ proc copyFile*(dest, source: string) =
|
||||||
close(s)
|
close(s)
|
||||||
close(d)
|
close(d)
|
||||||
|
|
||||||
proc moveFile*(dest, source: string) =
|
proc moveFile*(dest, source: string) {.deprecated.} =
|
||||||
## Moves a file from `source` to `dest`. If this fails, `EOS` is raised.
|
## Moves a file from `source` to `dest`. If this fails, `EOS` is raised.
|
||||||
|
## **Deprecated since version 0.8.8**: Use this proc with named arguments
|
||||||
|
## only, because the order will change!
|
||||||
if crename(source, dest) != 0'i32: OSError()
|
if crename(source, dest) != 0'i32: OSError()
|
||||||
|
|
||||||
proc removeFile*(file: string) =
|
proc removeFile*(file: string) =
|
||||||
|
|
|
||||||
|
|
@ -83,8 +83,8 @@ proc toUTF8*(c: TRune): string =
|
||||||
result[0] = chr(i)
|
result[0] = chr(i)
|
||||||
elif i <=% 0x07FF:
|
elif i <=% 0x07FF:
|
||||||
result = newString(2)
|
result = newString(2)
|
||||||
result[0] = chr(i shr 6 or 0b110_0000)
|
result[0] = chr((i shr 6) or 0b110_00000)
|
||||||
result[1] = chr(i and ones(6) or 0b10_000000)
|
result[1] = chr((i and ones(6)) or 0b10_000000)
|
||||||
elif i <=% 0xFFFF:
|
elif i <=% 0xFFFF:
|
||||||
result = newString(3)
|
result = newString(3)
|
||||||
result[0] = chr(i shr 12 or 0b1110_0000)
|
result[0] = chr(i shr 12 or 0b1110_0000)
|
||||||
|
|
|
||||||
|
|
@ -227,7 +227,7 @@ proc createAttributeNS*(doc: PDocument, namespaceURI: string, qualifiedName: str
|
||||||
raise newException(EInvalidCharacterErr, "Invalid character")
|
raise newException(EInvalidCharacterErr, "Invalid character")
|
||||||
# Exceptions
|
# Exceptions
|
||||||
if qualifiedName.contains(':'):
|
if qualifiedName.contains(':'):
|
||||||
if namespaceURI == nil or namespaceURI == "":
|
if namespaceURI == nil:
|
||||||
raise newException(ENamespaceErr, "When qualifiedName contains a prefix namespaceURI cannot be nil")
|
raise newException(ENamespaceErr, "When qualifiedName contains a prefix namespaceURI cannot be nil")
|
||||||
elif qualifiedName.split(':')[0].toLower() == "xml" and namespaceURI != "http://www.w3.org/XML/1998/namespace":
|
elif qualifiedName.split(':')[0].toLower() == "xml" and namespaceURI != "http://www.w3.org/XML/1998/namespace":
|
||||||
raise newException(ENamespaceErr,
|
raise newException(ENamespaceErr,
|
||||||
|
|
@ -303,7 +303,7 @@ proc createElement*(doc: PDocument, tagName: string): PElement =
|
||||||
proc createElementNS*(doc: PDocument, namespaceURI: string, qualifiedName: string): PElement =
|
proc createElementNS*(doc: PDocument, namespaceURI: string, qualifiedName: string): PElement =
|
||||||
## Creates an element of the given qualified name and namespace URI.
|
## Creates an element of the given qualified name and namespace URI.
|
||||||
if qualifiedName.contains(':'):
|
if qualifiedName.contains(':'):
|
||||||
if namespaceURI == nil or namespaceURI == "":
|
if namespaceURI == nil:
|
||||||
raise newException(ENamespaceErr, "When qualifiedName contains a prefix namespaceURI cannot be nil")
|
raise newException(ENamespaceErr, "When qualifiedName contains a prefix namespaceURI cannot be nil")
|
||||||
elif qualifiedName.split(':')[0].toLower() == "xml" and namespaceURI != "http://www.w3.org/XML/1998/namespace":
|
elif qualifiedName.split(':')[0].toLower() == "xml" and namespaceURI != "http://www.w3.org/XML/1998/namespace":
|
||||||
raise newException(ENamespaceErr,
|
raise newException(ENamespaceErr,
|
||||||
|
|
@ -467,6 +467,9 @@ proc namespaceURI*(n: PNode): string =
|
||||||
|
|
||||||
return n.FNamespaceURI
|
return n.FNamespaceURI
|
||||||
|
|
||||||
|
proc `namespaceURI=`*(n: PNode, value: string) =
|
||||||
|
n.FNamespaceURI = value
|
||||||
|
|
||||||
proc nextSibling*(n: PNode): PNode =
|
proc nextSibling*(n: PNode): PNode =
|
||||||
## Returns the next sibling of this node
|
## Returns the next sibling of this node
|
||||||
|
|
||||||
|
|
@ -507,7 +510,7 @@ proc previousSibling*(n: PNode): PNode =
|
||||||
return n.FParentNode.childNodes[i - 1]
|
return n.FParentNode.childNodes[i - 1]
|
||||||
return nil
|
return nil
|
||||||
|
|
||||||
proc `prefix=`*(n: var PNode, value: string) =
|
proc `prefix=`*(n: PNode, value: string) =
|
||||||
## Modifies the prefix of this node
|
## Modifies the prefix of this node
|
||||||
|
|
||||||
# Setter
|
# Setter
|
||||||
|
|
@ -530,11 +533,10 @@ proc `prefix=`*(n: var PNode, value: string) =
|
||||||
if n.nodeType == ElementNode:
|
if n.nodeType == ElementNode:
|
||||||
var el: PElement = PElement(n)
|
var el: PElement = PElement(n)
|
||||||
el.FTagName = value & ":" & n.FLocalName
|
el.FTagName = value & ":" & n.FLocalName
|
||||||
n = PNode(el)
|
|
||||||
elif n.nodeType == AttributeNode:
|
elif n.nodeType == AttributeNode:
|
||||||
var attr: PAttr = PAttr(n)
|
var attr: PAttr = PAttr(n)
|
||||||
attr.FName = value & ":" & n.FLocalName
|
attr.FName = value & ":" & n.FLocalName
|
||||||
n = PNode(attr)
|
|
||||||
|
|
||||||
# Procedures
|
# Procedures
|
||||||
proc appendChild*(n: PNode, newChild: PNode) =
|
proc appendChild*(n: PNode, newChild: PNode) =
|
||||||
|
|
|
||||||
|
|
@ -14,10 +14,35 @@ import xmldom, os, streams, parsexml, strutils
|
||||||
#XMLDom's Parser - Turns XML into a Document
|
#XMLDom's Parser - Turns XML into a Document
|
||||||
|
|
||||||
type
|
type
|
||||||
#Parsing errors
|
# Parsing errors
|
||||||
EMismatchedTag* = object of E_Base ## Raised when a tag is not properly closed
|
EMismatchedTag* = object of E_Base ## Raised when a tag is not properly closed
|
||||||
EParserError* = object of E_Base ## Raised when an unexpected XML Parser event occurs
|
EParserError* = object of E_Base ## Raised when an unexpected XML Parser event occurs
|
||||||
|
|
||||||
|
# For namespaces
|
||||||
|
xmlnsAttr = tuple[name, value: string, ownerElement: PElement]
|
||||||
|
|
||||||
|
var nsList: seq[xmlnsAttr] = @[] # Used for storing namespaces
|
||||||
|
|
||||||
|
proc getNS(prefix: string): string =
|
||||||
|
var defaultNS: seq[string] = @[]
|
||||||
|
|
||||||
|
for key, value, tag in items(nsList):
|
||||||
|
if ":" in key:
|
||||||
|
if key.split(':')[1] == prefix:
|
||||||
|
return value
|
||||||
|
|
||||||
|
if key == "xmlns":
|
||||||
|
defaultNS.add(value)
|
||||||
|
|
||||||
|
# Don't return the default namespaces
|
||||||
|
# in the loop, because then they would have a precedence
|
||||||
|
# over normal namespaces
|
||||||
|
if defaultNS.len() > 0:
|
||||||
|
return defaultNS[0] # Return the first found default namespace
|
||||||
|
# if none are specified for this prefix
|
||||||
|
|
||||||
|
return ""
|
||||||
|
|
||||||
proc parseText(x: var TXmlParser, doc: var PDocument): PText =
|
proc parseText(x: var TXmlParser, doc: var PDocument): PText =
|
||||||
result = doc.createTextNode(x.charData())
|
result = doc.createTextNode(x.charData())
|
||||||
|
|
||||||
|
|
@ -28,24 +53,33 @@ proc parseElement(x: var TXmlParser, doc: var PDocument): PElement =
|
||||||
case x.kind()
|
case x.kind()
|
||||||
of xmlEof:
|
of xmlEof:
|
||||||
break
|
break
|
||||||
of xmlElementStart:
|
of xmlElementStart, xmlElementOpen:
|
||||||
if n.tagName() != "":
|
if n.tagName() != "":
|
||||||
n.appendChild(parseElement(x, doc))
|
n.appendChild(parseElement(x, doc))
|
||||||
else:
|
else:
|
||||||
n = doc.createElement(x.elementName)
|
n = doc.createElementNS("", x.elementName)
|
||||||
of xmlElementOpen:
|
|
||||||
if n.tagName() != "":
|
|
||||||
n.appendChild(parseElement(x, doc))
|
|
||||||
else:
|
|
||||||
if x.elementName.contains(':'):
|
|
||||||
#TODO: NamespaceURI
|
|
||||||
n = doc.createElementNS("nil", x.elementName)
|
|
||||||
else:
|
|
||||||
n = doc.createElement(x.elementName)
|
|
||||||
|
|
||||||
of xmlElementEnd:
|
of xmlElementEnd:
|
||||||
if x.elementName == n.nodeName:
|
if x.elementName == n.nodeName:
|
||||||
# n.normalize() # Remove any whitespace etc.
|
# n.normalize() # Remove any whitespace etc.
|
||||||
|
|
||||||
|
var ns: string
|
||||||
|
if x.elementName.contains(':'):
|
||||||
|
ns = getNS(x.elementName.split(':')[0])
|
||||||
|
else:
|
||||||
|
ns = getNS("")
|
||||||
|
|
||||||
|
n.namespaceURI = ns
|
||||||
|
|
||||||
|
# Remove any namespaces this element declared
|
||||||
|
var count = 0 # Variable which keeps the index
|
||||||
|
# We need to edit it..
|
||||||
|
for i in low(nsList)..len(nsList)-1:
|
||||||
|
if nsList[count][2] == n:
|
||||||
|
nsList.delete(count)
|
||||||
|
dec(count)
|
||||||
|
inc(count)
|
||||||
|
|
||||||
return n
|
return n
|
||||||
else: #The wrong element is ended
|
else: #The wrong element is ended
|
||||||
raise newException(EMismatchedTag, "Mismatched tag at line " &
|
raise newException(EMismatchedTag, "Mismatched tag at line " &
|
||||||
|
|
@ -54,11 +88,15 @@ proc parseElement(x: var TXmlParser, doc: var PDocument): PElement =
|
||||||
of xmlCharData:
|
of xmlCharData:
|
||||||
n.appendChild(parseText(x, doc))
|
n.appendChild(parseText(x, doc))
|
||||||
of xmlAttribute:
|
of xmlAttribute:
|
||||||
|
if x.attrKey == "xmlns" or x.attrKey.startsWith("xmlns:"):
|
||||||
|
nsList.add((x.attrKey, x.attrValue, n))
|
||||||
|
|
||||||
if x.attrKey.contains(':'):
|
if x.attrKey.contains(':'):
|
||||||
#TODO: NamespaceURI
|
var ns = getNS(x.attrKey)
|
||||||
n.setAttributeNS("nil", x.attrKey, x.attrValue)
|
n.setAttributeNS(ns, x.attrKey, x.attrValue)
|
||||||
else:
|
else:
|
||||||
n.setAttribute(x.attrKey, x.attrValue)
|
n.setAttribute(x.attrKey, x.attrValue)
|
||||||
|
|
||||||
of xmlCData:
|
of xmlCData:
|
||||||
n.appendChild(doc.createCDATASection(x.charData()))
|
n.appendChild(doc.createCDATASection(x.charData()))
|
||||||
of xmlComment:
|
of xmlComment:
|
||||||
|
|
@ -76,15 +114,12 @@ proc parseElement(x: var TXmlParser, doc: var PDocument): PElement =
|
||||||
raise newException(EMismatchedTag,
|
raise newException(EMismatchedTag,
|
||||||
"Mismatched tag at line " & $x.getLine() & " column " & $x.getColumn)
|
"Mismatched tag at line " & $x.getLine() & " column " & $x.getColumn)
|
||||||
|
|
||||||
proc loadXML*(path: string): PDocument =
|
proc loadXMLStream*(stream: PStream): PDocument =
|
||||||
## Loads and parses XML from file specified by ``path``, and returns
|
## Loads and parses XML from a stream specified by ``stream``, and returns
|
||||||
## a ``PDocument``
|
## a ``PDocument``
|
||||||
|
|
||||||
var s = newFileStream(path, fmRead)
|
|
||||||
if s == nil: raise newException(EIO, "Unable to read file " & path)
|
|
||||||
|
|
||||||
var x: TXmlParser
|
var x: TXmlParser
|
||||||
open(x, s, path, {reportComments})
|
open(x, stream, nil, {reportComments})
|
||||||
|
|
||||||
var XmlDoc: PDocument
|
var XmlDoc: PDocument
|
||||||
var DOM: PDOMImplementation = getDOM()
|
var DOM: PDOMImplementation = getDOM()
|
||||||
|
|
@ -102,10 +137,32 @@ proc loadXML*(path: string): PDocument =
|
||||||
else:
|
else:
|
||||||
raise newException(EParserError, "Unexpected XML Parser event")
|
raise newException(EParserError, "Unexpected XML Parser event")
|
||||||
|
|
||||||
close(x)
|
|
||||||
return XmlDoc
|
return XmlDoc
|
||||||
|
|
||||||
|
proc loadXML*(xml: string): PDocument =
|
||||||
|
## Loads and parses XML from a string specified by ``xml``, and returns
|
||||||
|
## a ``PDocument``
|
||||||
|
var s = newStringStream(xml)
|
||||||
|
return loadXMLStream(s)
|
||||||
|
|
||||||
|
|
||||||
|
proc loadXMLFile*(path: string): PDocument =
|
||||||
|
## Loads and parses XML from a file specified by ``path``, and returns
|
||||||
|
## a ``PDocument``
|
||||||
|
|
||||||
|
var s = newFileStream(path, fmRead)
|
||||||
|
if s == nil: raise newException(EIO, "Unable to read file " & path)
|
||||||
|
return loadXMLStream(s)
|
||||||
|
|
||||||
|
|
||||||
when isMainModule:
|
when isMainModule:
|
||||||
var xml = loadXML(r"C:\Users\Dominik\Desktop\Code\Nimrod\xmldom\test.xml")
|
var xml = loadXMLFile(r"C:\Users\Dominik\Desktop\Code\Nimrod\xmldom\test.xml")
|
||||||
|
#echo(xml.getElementsByTagName("m:test2")[0].namespaceURI)
|
||||||
|
#echo(xml.getElementsByTagName("bla:test")[0].namespaceURI)
|
||||||
|
#echo(xml.getElementsByTagName("test")[0].namespaceURI)
|
||||||
|
for i in items(xml.getElementsByTagName("*")):
|
||||||
|
if i.namespaceURI != nil:
|
||||||
|
echo(i.nodeName, "=", i.namespaceURI)
|
||||||
|
|
||||||
|
|
||||||
echo($xml)
|
echo($xml)
|
||||||
|
|
@ -25,6 +25,8 @@ proc raiseInvalidXml(errors: seq[string]) =
|
||||||
proc addNode(father, son: PXmlNode) =
|
proc addNode(father, son: PXmlNode) =
|
||||||
if son != nil: add(father, son)
|
if son != nil: add(father, son)
|
||||||
|
|
||||||
|
proc parse(x: var TXmlParser, errors: var seq[string]): PXmlNode
|
||||||
|
|
||||||
proc untilElementEnd(x: var TXmlParser, result: PXmlNode,
|
proc untilElementEnd(x: var TXmlParser, result: PXmlNode,
|
||||||
errors: var seq[string]) =
|
errors: var seq[string]) =
|
||||||
while true:
|
while true:
|
||||||
|
|
@ -33,11 +35,11 @@ proc untilElementEnd(x: var TXmlParser, result: PXmlNode,
|
||||||
if x.elementName == result.tag:
|
if x.elementName == result.tag:
|
||||||
next(x)
|
next(x)
|
||||||
else:
|
else:
|
||||||
errors.add(errorMsg(x, "</" & result.tag & "$1> expected"))
|
errors.add(errorMsg(x, "</" & result.tag & "> expected"))
|
||||||
# do not skip it here!
|
# do not skip it here!
|
||||||
break
|
break
|
||||||
of xmlEof:
|
of xmlEof:
|
||||||
errors.add(errorMsg(x, "</" & result.tag & "$1> expected"))
|
errors.add(errorMsg(x, "</" & result.tag & "> expected"))
|
||||||
break
|
break
|
||||||
else:
|
else:
|
||||||
result.addNode(parse(x, errors))
|
result.addNode(parse(x, errors))
|
||||||
|
|
@ -91,7 +93,7 @@ proc parse(x: var TXmlParser, errors: var seq[string]): PXmlNode =
|
||||||
next(x)
|
next(x)
|
||||||
of xmlEntity:
|
of xmlEntity:
|
||||||
## &entity;
|
## &entity;
|
||||||
## XXX To implement!
|
errors.add(errorMsg(x, "unknown entity: " & x.entityName))
|
||||||
next(x)
|
next(x)
|
||||||
of xmlEof: nil
|
of xmlEof: nil
|
||||||
|
|
||||||
|
|
@ -110,6 +112,8 @@ proc parseXml*(s: PStream, filename: string,
|
||||||
of xmlComment, xmlWhitespace: nil # just skip it
|
of xmlComment, xmlWhitespace: nil # just skip it
|
||||||
of xmlError:
|
of xmlError:
|
||||||
errors.add(errorMsg(x))
|
errors.add(errorMsg(x))
|
||||||
|
of xmlSpecial:
|
||||||
|
errors.add(errorMsg(x, "<some_tag> expected"))
|
||||||
else:
|
else:
|
||||||
errors.add(errorMsg(x, "<some_tag> expected"))
|
errors.add(errorMsg(x, "<some_tag> expected"))
|
||||||
break
|
break
|
||||||
|
|
@ -122,17 +126,33 @@ proc parseXml*(s: PStream): PXmlNode =
|
||||||
result = parseXml(s, "unknown_html_doc", errors)
|
result = parseXml(s, "unknown_html_doc", errors)
|
||||||
if errors.len > 0: raiseInvalidXMl(errors)
|
if errors.len > 0: raiseInvalidXMl(errors)
|
||||||
|
|
||||||
proc loadXml*(path: string, reportErrors = false): PXmlNode =
|
proc loadXml*(path: string, errors: var seq[string]): PXmlNode =
|
||||||
## Loads and parses XML from file specified by ``path``, and returns
|
## Loads and parses XML from file specified by ``path``, and returns
|
||||||
## a ``PXmlNode``. If `reportErrors` is true, the parsing errors are
|
## a ``PXmlNode``. Every occured parsing error is added to the `errors`
|
||||||
## ``echo``ed, otherwise an exception is thrown.
|
## sequence.
|
||||||
var s = newFileStream(path, fmRead)
|
var s = newFileStream(path, fmRead)
|
||||||
if s == nil: raise newException(EIO, "Unable to read file: " & path)
|
if s == nil: raise newException(EIO, "Unable to read file: " & path)
|
||||||
|
result = parseXml(s, path, errors)
|
||||||
|
|
||||||
|
proc loadXml*(path: string): PXmlNode =
|
||||||
|
## Loads and parses XML from file specified by ``path``, and returns
|
||||||
|
## a ``PXmlNode``. All parsing errors are turned into an ``EInvalidXML``
|
||||||
|
## exception.
|
||||||
|
var errors: seq[string] = @[]
|
||||||
|
result = loadXml(path, errors)
|
||||||
|
if errors.len > 0: raiseInvalidXMl(errors)
|
||||||
|
|
||||||
|
when isMainModule:
|
||||||
|
import os
|
||||||
|
|
||||||
var errors: seq[string] = @[]
|
var errors: seq[string] = @[]
|
||||||
result = parseXml(s, path, errors)
|
var x = loadXml(paramStr(1), errors)
|
||||||
if reportErrors:
|
for e in items(errors): echo e
|
||||||
for msg in items(errors): echo(msg)
|
|
||||||
elif errors.len > 0:
|
var f: TFile
|
||||||
raiseInvalidXMl(errors)
|
if open(f, "xmltest.txt", fmWrite):
|
||||||
|
f.write($x)
|
||||||
|
f.close()
|
||||||
|
else:
|
||||||
|
quit("cannot write test.txt")
|
||||||
|
|
||||||
|
|
@ -153,8 +153,15 @@ proc addIndent(result: var string, indent: int) =
|
||||||
result.add("\n")
|
result.add("\n")
|
||||||
for i in 1..indent: result.add(' ')
|
for i in 1..indent: result.add(' ')
|
||||||
|
|
||||||
|
proc noWhitespace(n: PXmlNode): bool =
|
||||||
|
#for i in 1..n.len-1:
|
||||||
|
# if n[i].kind != n[0].kind: return true
|
||||||
|
for i in 0..n.len-1:
|
||||||
|
if n[i].kind in {xnText, xnEntity}: return true
|
||||||
|
|
||||||
proc add*(result: var string, n: PXmlNode, indent = 0, indWidth = 2) =
|
proc add*(result: var string, n: PXmlNode, indent = 0, indWidth = 2) =
|
||||||
## adds the textual representation of `n` to `result`.
|
## adds the textual representation of `n` to `result`.
|
||||||
|
if n == nil: return
|
||||||
case n.k
|
case n.k
|
||||||
of xnElement:
|
of xnElement:
|
||||||
result.add('<')
|
result.add('<')
|
||||||
|
|
@ -168,10 +175,19 @@ proc add*(result: var string, n: PXmlNode, indent = 0, indWidth = 2) =
|
||||||
result.add('"')
|
result.add('"')
|
||||||
if n.len > 0:
|
if n.len > 0:
|
||||||
result.add('>')
|
result.add('>')
|
||||||
|
if n.len > 1:
|
||||||
|
if noWhitespace(n):
|
||||||
|
# for mixed leaves, we cannot output whitespace for readability,
|
||||||
|
# because this would be wrong. For example: ``a<b>b</b>`` is
|
||||||
|
# different from ``a <b>b</b>``.
|
||||||
|
for i in 0..n.len-1: result.add(n[i], indent+indWidth, indWidth)
|
||||||
|
else:
|
||||||
for i in 0..n.len-1:
|
for i in 0..n.len-1:
|
||||||
result.addIndent(indent+indWidth)
|
result.addIndent(indent+indWidth)
|
||||||
result.add(n[i], indent+indWidth, indWidth)
|
result.add(n[i], indent+indWidth, indWidth)
|
||||||
result.addIndent(indent)
|
result.addIndent(indent)
|
||||||
|
else:
|
||||||
|
result.add(n[0], indent+indWidth, indWidth)
|
||||||
result.add("</")
|
result.add("</")
|
||||||
result.add(n.fTag)
|
result.add(n.fTag)
|
||||||
result.add(">")
|
result.add(">")
|
||||||
|
|
|
||||||
|
|
@ -50,7 +50,8 @@ type
|
||||||
tkBind, tkBlock, tkBreak, tkCase, tkCast,
|
tkBind, tkBlock, tkBreak, tkCase, tkCast,
|
||||||
tkConst, tkContinue, tkConverter, tkDiscard, tkDistinct, tkDiv, tkElif,
|
tkConst, tkContinue, tkConverter, tkDiscard, tkDistinct, tkDiv, tkElif,
|
||||||
tkElse, tkEnd, tkEnum, tkExcept, tkFinally, tkFor, tkFrom, tkGeneric, tkIf,
|
tkElse, tkEnd, tkEnum, tkExcept, tkFinally, tkFor, tkFrom, tkGeneric, tkIf,
|
||||||
tkImplies, tkImport, tkIn, tkInclude, tkIs, tkIsnot, tkIterator, tkLambda,
|
tkImplies, tkImport, tkIn, tkInclude, tkIs, tkIsnot, tkIterator,
|
||||||
|
tkLambda, tkLet,
|
||||||
tkMacro, tkMethod, tkMod, tkNil, tkNot, tkNotin, tkObject, tkOf, tkOr,
|
tkMacro, tkMethod, tkMod, tkNil, tkNot, tkNotin, tkObject, tkOf, tkOr,
|
||||||
tkOut, tkProc, tkPtr, tkRaise, tkRef, tkReturn, tkShl, tkShr, tkTemplate,
|
tkOut, tkProc, tkPtr, tkRaise, tkRef, tkReturn, tkShl, tkShr, tkTemplate,
|
||||||
tkTry, tkTuple, tkType, tkVar, tkWhen, tkWhile, tkWith, tkWithout, tkXor,
|
tkTry, tkTuple, tkType, tkVar, tkWhen, tkWhile, tkWith, tkWithout, tkXor,
|
||||||
|
|
@ -82,7 +83,8 @@ const
|
||||||
"bind", "block", "break", "case", "cast",
|
"bind", "block", "break", "case", "cast",
|
||||||
"const", "continue", "converter", "discard", "distinct", "div", "elif",
|
"const", "continue", "converter", "discard", "distinct", "div", "elif",
|
||||||
"else", "end", "enum", "except", "finally", "for", "from", "generic", "if",
|
"else", "end", "enum", "except", "finally", "for", "from", "generic", "if",
|
||||||
"implies", "import", "in", "include", "is", "isnot", "iterator", "lambda",
|
"implies", "import", "in", "include", "is", "isnot", "iterator",
|
||||||
|
"lambda", "let",
|
||||||
"macro", "method", "mod", "nil", "not", "notin", "object", "of", "or",
|
"macro", "method", "mod", "nil", "not", "notin", "object", "of", "or",
|
||||||
"out", "proc", "ptr", "raise", "ref", "return", "shl", "shr", "template",
|
"out", "proc", "ptr", "raise", "ref", "return", "shl", "shr", "template",
|
||||||
"try", "tuple", "type", "var", "when", "while", "with", "without", "xor",
|
"try", "tuple", "type", "var", "when", "while", "with", "without", "xor",
|
||||||
|
|
@ -770,6 +772,7 @@ proc rawGetTok(L: var TLexer, tok: var TToken) =
|
||||||
of '\"':
|
of '\"':
|
||||||
getString(L, tok, false)
|
getString(L, tok, false)
|
||||||
of '\'':
|
of '\'':
|
||||||
|
tok.tokType = tkCharLit
|
||||||
getCharacter(L, tok)
|
getCharacter(L, tok)
|
||||||
tok.tokType = tkCharLit
|
tok.tokType = tkCharLit
|
||||||
of lexbase.EndOfFile:
|
of lexbase.EndOfFile:
|
||||||
|
|
|
||||||
|
|
@ -26,7 +26,8 @@ type
|
||||||
wBind, wBlock, wBreak, wCase, wCast, wConst,
|
wBind, wBlock, wBreak, wCase, wCast, wConst,
|
||||||
wContinue, wConverter, wDiscard, wDistinct, wDiv, wElif, wElse, wEnd, wEnum,
|
wContinue, wConverter, wDiscard, wDistinct, wDiv, wElif, wElse, wEnd, wEnum,
|
||||||
wExcept, wFinally, wFor, wFrom, wGeneric, wIf, wImplies, wImport, wIn,
|
wExcept, wFinally, wFor, wFrom, wGeneric, wIf, wImplies, wImport, wIn,
|
||||||
wInclude, wIs, wIsnot, wIterator, wLambda, wMacro, wMethod, wMod, wNil,
|
wInclude, wIs, wIsnot, wIterator, wLambda, wLet,
|
||||||
|
wMacro, wMethod, wMod, wNil,
|
||||||
wNot, wNotin, wObject, wOf, wOr, wOut, wProc, wPtr, wRaise, wRef, wReturn,
|
wNot, wNotin, wObject, wOf, wOr, wOut, wProc, wPtr, wRaise, wRef, wReturn,
|
||||||
wShl, wShr, wTemplate, wTry, wTuple, wType, wVar, wWhen, wWhile, wWith,
|
wShl, wShr, wTemplate, wTry, wTuple, wType, wVar, wWhen, wWhile, wWith,
|
||||||
wWithout, wXor, wYield,
|
wWithout, wXor, wYield,
|
||||||
|
|
@ -67,7 +68,8 @@ const
|
||||||
"bind", "block", "break", "case", "cast",
|
"bind", "block", "break", "case", "cast",
|
||||||
"const", "continue", "converter", "discard", "distinct", "div", "elif",
|
"const", "continue", "converter", "discard", "distinct", "div", "elif",
|
||||||
"else", "end", "enum", "except", "finally", "for", "from", "generic", "if",
|
"else", "end", "enum", "except", "finally", "for", "from", "generic", "if",
|
||||||
"implies", "import", "in", "include", "is", "isnot", "iterator", "lambda",
|
"implies", "import", "in", "include", "is", "isnot", "iterator",
|
||||||
|
"lambda", "let",
|
||||||
"macro", "method", "mod", "nil", "not", "notin", "object", "of", "or",
|
"macro", "method", "mod", "nil", "not", "notin", "object", "of", "or",
|
||||||
"out", "proc", "ptr", "raise", "ref", "return", "shl", "shr", "template",
|
"out", "proc", "ptr", "raise", "ref", "return", "shl", "shr", "template",
|
||||||
"try", "tuple", "type", "var", "when", "while", "with", "without", "xor",
|
"try", "tuple", "type", "var", "when", "while", "with", "without", "xor",
|
||||||
|
|
|
||||||
|
|
@ -1,2 +1,25 @@
|
||||||
<!DOCTYPE some doctype >
|
<!DOCTYPE some doctype >
|
||||||
|
|
||||||
|
<html>
|
||||||
|
<head>
|
||||||
|
<title>Test title!</title>
|
||||||
|
</head>
|
||||||
|
|
||||||
|
<body>
|
||||||
|
Ein Text mit vielen Zeichenumbrüchen. <br />
|
||||||
|
<br><br>
|
||||||
|
|
||||||
|
<ul>
|
||||||
|
<li>first <span class = "34" >Item.</span>
|
||||||
|
<li>second Item.
|
||||||
|
<li>third item. Mit ä.
|
||||||
|
</ul>
|
||||||
|
|
||||||
|
Para 0.
|
||||||
|
|
||||||
|
<p>Para1. </p>
|
||||||
|
<p>Para 2.
|
||||||
|
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -290,7 +290,8 @@ proc writeFile(filename, content, newline: string) =
|
||||||
quit("Cannot open for writing: " & filename)
|
quit("Cannot open for writing: " & filename)
|
||||||
|
|
||||||
proc srcdist(c: var TConfigData) =
|
proc srcdist(c: var TConfigData) =
|
||||||
for x in walkFiles("lib/*.h"): CopyFile("build" / extractFilename(x), x)
|
for x in walkFiles("lib/*.h"):
|
||||||
|
CopyFile(dest="build" / extractFilename(x), source=x)
|
||||||
for osA in 1..c.oses.len:
|
for osA in 1..c.oses.len:
|
||||||
for cpuA in 1..c.cpus.len:
|
for cpuA in 1..c.cpus.len:
|
||||||
var dir = buildDir(osA, cpuA)
|
var dir = buildDir(osA, cpuA)
|
||||||
|
|
@ -307,7 +308,7 @@ proc srcdist(c: var TConfigData) =
|
||||||
readCFiles(c, osA, cpuA)
|
readCFiles(c, osA, cpuA)
|
||||||
for i in 0 .. c.cfiles[osA][cpuA].len-1:
|
for i in 0 .. c.cfiles[osA][cpuA].len-1:
|
||||||
var dest = dir / extractFilename(c.cfiles[osA][cpuA][i])
|
var dest = dir / extractFilename(c.cfiles[osA][cpuA][i])
|
||||||
CopyFile(dest, c.cfiles[osA][cpuA][i])
|
CopyFile(dest=dest, source=c.cfiles[osA][cpuA][i])
|
||||||
c.cfiles[osA][cpuA][i] = dest
|
c.cfiles[osA][cpuA][i] = dest
|
||||||
# second pass: remove duplicate files
|
# second pass: remove duplicate files
|
||||||
for osA in countdown(c.oses.len, 1):
|
for osA in countdown(c.oses.len, 1):
|
||||||
|
|
|
||||||
|
|
@ -179,7 +179,7 @@ proc buildPdfDoc(c: var TConfigData, destPath: string) =
|
||||||
Exec("pdflatex " & changeFileExt(d, "tex"))
|
Exec("pdflatex " & changeFileExt(d, "tex"))
|
||||||
# delete all the crappy temporary files:
|
# delete all the crappy temporary files:
|
||||||
var pdf = splitFile(d).name & ".pdf"
|
var pdf = splitFile(d).name & ".pdf"
|
||||||
moveFile(destPath / pdf, pdf)
|
moveFile(dest=destPath / pdf, source=pdf)
|
||||||
removeFile(changeFileExt(pdf, "aux"))
|
removeFile(changeFileExt(pdf, "aux"))
|
||||||
if existsFile(changeFileExt(pdf, "toc")):
|
if existsFile(changeFileExt(pdf, "toc")):
|
||||||
removeFile(changeFileExt(pdf, "toc"))
|
removeFile(changeFileExt(pdf, "toc"))
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,7 @@ proc walker(dir: string) =
|
||||||
for kind, path in walkDir(dir):
|
for kind, path in walkDir(dir):
|
||||||
case kind
|
case kind
|
||||||
of pcFile:
|
of pcFile:
|
||||||
moveFile(newName(path), path)
|
moveFile(dest=newName(path), source=path)
|
||||||
# test if installation still works:
|
# test if installation still works:
|
||||||
if execShellCmd(r"nimrod c --force_build tests\tlastmod") == 0:
|
if execShellCmd(r"nimrod c --force_build tests\tlastmod") == 0:
|
||||||
echo "Optional: ", path
|
echo "Optional: ", path
|
||||||
|
|
|
||||||
|
|
@ -21,6 +21,8 @@ Bugfixes
|
||||||
- Fixed a bug in ``os.setFilePermissions`` for Windows.
|
- Fixed a bug in ``os.setFilePermissions`` for Windows.
|
||||||
- An overloadable symbol can now have the same name as an imported module.
|
- An overloadable symbol can now have the same name as an imported module.
|
||||||
- Fixed a serious bug in ``strutils.cmpIgnoreCase``.
|
- Fixed a serious bug in ``strutils.cmpIgnoreCase``.
|
||||||
|
- Fixed ``unicode.toUTF8``.
|
||||||
|
- The compiler now rejects ``'\n'``.
|
||||||
|
|
||||||
|
|
||||||
Additions
|
Additions
|
||||||
|
|
@ -40,7 +42,7 @@ Additions
|
||||||
- Added ``xmldom`` module.
|
- Added ``xmldom`` module.
|
||||||
- Added ``xmldomparser`` module.
|
- Added ``xmldomparser`` module.
|
||||||
- Added ``xmltree`` module.
|
- Added ``xmltree`` module.
|
||||||
- Added ``xmltreeparser`` module.
|
- Added ``xmlparser`` module.
|
||||||
- Added ``htmlparser`` module.
|
- Added ``htmlparser`` module.
|
||||||
- Many wrappers now do not contain redundant name prefixes (like ``GTK_``,
|
- Many wrappers now do not contain redundant name prefixes (like ``GTK_``,
|
||||||
``lua``). The new wrappers are available in ``lib/newwrap``. Change
|
``lua``). The new wrappers are available in ``lib/newwrap``. Change
|
||||||
|
|
@ -63,6 +65,11 @@ Changes affecting backwards compatibility
|
||||||
the standard library's path.
|
the standard library's path.
|
||||||
- The compiler does not include a Pascal parser for bootstrapping purposes any
|
- The compiler does not include a Pascal parser for bootstrapping purposes any
|
||||||
more. Instead there is a ``pas2nim`` tool that contains the old functionality.
|
more. Instead there is a ``pas2nim`` tool that contains the old functionality.
|
||||||
|
- The procs ``os.copyFile`` and ``os.moveFile`` have been deprecated
|
||||||
|
temporarily, so that the compiler warns about their usage. Use them with
|
||||||
|
named arguments only, because the parameter order will change the next
|
||||||
|
version!
|
||||||
|
- ``atomic`` and ``let`` are now keywords.
|
||||||
|
|
||||||
|
|
||||||
2009-12-21 Version 0.8.6 released
|
2009-12-21 Version 0.8.6 released
|
||||||
|
|
|
||||||
|
|
@ -31,6 +31,7 @@ srcdoc: "pure/streams;pure/terminal;pure/cgi;impure/web;pure/unicode"
|
||||||
srcdoc: "impure/zipfiles;pure/xmlgen;pure/macros;pure/parseutils;pure/browsers"
|
srcdoc: "impure/zipfiles;pure/xmlgen;pure/macros;pure/parseutils;pure/browsers"
|
||||||
srcdoc: "impure/db_postgres;impure/db_mysql;pure/httpserver;pure/httpclient"
|
srcdoc: "impure/db_postgres;impure/db_mysql;pure/httpserver;pure/httpclient"
|
||||||
srcdoc: "pure/ropes;pure/unidecode/unidecode;pure/xmldom;pure/xmldomparser"
|
srcdoc: "pure/ropes;pure/unidecode/unidecode;pure/xmldom;pure/xmldomparser"
|
||||||
|
srcdoc: "pure/xmlparser;pure/htmlparser;pure/xmltree"
|
||||||
|
|
||||||
webdoc: "wrappers/libcurl;pure/md5;wrappers/mysql;wrappers/iup"
|
webdoc: "wrappers/libcurl;pure/md5;wrappers/mysql;wrappers/iup"
|
||||||
webdoc: "wrappers/sqlite3;wrappers/python;wrappers/tcl"
|
webdoc: "wrappers/sqlite3;wrappers/python;wrappers/tcl"
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue