more enhancements for the lib

This commit is contained in:
Andreas Rumpf 2010-02-08 22:07:45 +01:00
commit 44ed48ffa6
17 changed files with 65591 additions and 55 deletions

View file

@ -43,7 +43,7 @@
##
## echo(postContent("http://validator.w3.org/check", headers, body))
import sockets, strutils, parseurl, pegs, parseutils, strtabs
import sockets, strutils, parseurl, parseutils, strtabs
type
TResponse* = tuple[
@ -166,20 +166,32 @@ proc parseResponse(s: TSocket): TResponse =
# Parse the version
# Parses the first line of the headers
# ``HTTP/1.1`` 200 OK
var matches: array[0..1, string]
var L = d.matchLen(peg"\i 'HTTP/' {'1.1'/'1.0'} \s+ {(!\n .)*}\n",
matches, i)
if L < 0: httpError("invalid HTTP header")
result.version = matches[0]
result.status = matches[1]
var L = skipIgnoreCase(d, "HTTP/1.1", i)
if L > 0:
result.version = "1.1"
inc(i, L)
else:
L = skipIgnoreCase(d, "HTTP/1.0", i)
if L > 0:
result.version = "1.0"
inc(i, L)
else:
httpError("invalid HTTP header")
L = skipWhiteSpace(d, i)
if L <= 0: httpError("invalid HTTP header")
inc(i, L)
result.status = ""
while d[i] notin {'\C', '\L', '\0'}:
result.status.add(d[i])
inc(i)
if d[i] == '\C': inc(i)
if d[i] == '\L': inc(i)
else: httpError("invalid HTTP header, CR-LF expected")
# Parse the headers
# Everything after the first line leading up to the body
# htype: hvalue
result.headers = newStringTable(modeCaseInsensitive)
while true:
var key = ""

26
lib/pure/unidecode/gen.py Normal file
View file

@ -0,0 +1,26 @@
#! usr/bin/env python
# -*- coding: utf-8 -*-
# Generates the unidecode.dat module
# (c) 2010 Andreas Rumpf
from unidecode import unidecode
def main2():
data = []
for x in xrange(128, 0xffff + 1):
u = eval("u'\u%04x'" % x)
val = unidecode(u)
data.append(val)
f = open("unidecode.dat", "wb+")
for d in data:
f.write("%s\n" % d)
f.close()
main2()

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,65 @@
#
#
# Nimrod's Runtime Library
# (c) Copyright 2010 Andreas Rumpf
#
# See the file "copying.txt", included in this
# distribution, for details about the copyright.
#
## This module is based on Python's Unidecode module by Tomaz Solc,
## which in turn is based on the ``Text::Unidecode`` Perl module by
## Sean M. Burke
## (http://search.cpan.org/~sburke/Text-Unidecode-0.04/lib/Text/Unidecode.pm).
##
## It provides a single proc that does Unicode to ASCII transliterations:
## It finds the sequence of ASCII characters that is the closest approximation
## to the Unicode string.
##
## For example, the closest to string "Äußerst" in ASCII is "Ausserst". Some
## information is lost in this transformation, of course, since several Unicode
## strings can be transformed in the same ASCII representation. So this is a
## strictly one-way transformation. However a human reader will probably
## still be able to guess what original string was meant from the context.
##
## This module needs the data file "unidecode.dat" to work, so it has to be
## shipped with the application!
import unicode
proc loadTranslationTable(filename: string): seq[string] =
newSeq(result, 0xffff)
var i = 0
for line in lines(filename):
result[i] = line
inc(i)
var
translationTable: seq[string]
var
datafile* = "unidecode.dat" ## location can be overwritten for deployment
proc unidecode*(s: string): string =
## Finds the sequence of ASCII characters that is the closest approximation
## to the UTF-8 string `s`.
##
## Example:
##
## ..code-block:: nimrod
## unidecode("\x53\x17\x4E\xB0")
##
## Results in: "Bei Jing"
##
result = ""
for r in runes(s):
var c = int(r)
if c <=% 127: add(result, chr(c))
elif c <=% 0xffff:
if isNil(translationTable):
translationTable = loadTranslationTable(datafile)
add(result, translationTable[c-128])
when isMainModule:
echo unidecode("Äußerst")