Improve documentation for unidecode (#16986)
* Improve documentation for unidecode Minor changes to gen.py * Fix typo in gen.py
This commit is contained in:
parent
635c0b6cb9
commit
00551f972e
2 changed files with 28 additions and 26 deletions
|
|
@ -11,7 +11,7 @@ try:
|
||||||
except ImportError:
|
except ImportError:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
def main2():
|
def main():
|
||||||
f = open("unidecode.dat", "wb+")
|
f = open("unidecode.dat", "wb+")
|
||||||
for x in range(128, 0xffff + 1):
|
for x in range(128, 0xffff + 1):
|
||||||
u = eval("u'\\u%04x'" % x)
|
u = eval("u'\\u%04x'" % x)
|
||||||
|
|
@ -19,12 +19,12 @@ def main2():
|
||||||
val = unidecode(u)
|
val = unidecode(u)
|
||||||
|
|
||||||
# f.write("%x | " % x)
|
# f.write("%x | " % x)
|
||||||
if x==0x2028: # U+2028 = LINE SEPARATOR
|
if x == 0x2028: # U+2028 = LINE SEPARATOR
|
||||||
val = ""
|
val = ""
|
||||||
elif x==0x2029: # U+2028 = PARAGRAPH SEPARATOR
|
elif x == 0x2029: # U+2029 = PARAGRAPH SEPARATOR
|
||||||
val = ""
|
val = ""
|
||||||
f.write("%s\n" % val)
|
f.write("%s\n" % val)
|
||||||
|
|
||||||
f.close()
|
f.close()
|
||||||
|
|
||||||
main2()
|
main()
|
||||||
|
|
|
||||||
|
|
@ -7,30 +7,31 @@
|
||||||
# distribution, for details about the copyright.
|
# distribution, for details about the copyright.
|
||||||
#
|
#
|
||||||
|
|
||||||
## This module is based on Python's Unidecode module by Tomaz Solc,
|
## This module is based on Python's [Unidecode](https://pypi.org/project/Unidecode/)
|
||||||
## which in turn is based on the ``Text::Unidecode`` Perl module by
|
## module by Tomaz Solc, which in turn is based on the
|
||||||
## Sean M. Burke
|
## [Text::Unidecode](https://metacpan.org/pod/Text::Unidecode)
|
||||||
## (http://search.cpan.org/~sburke/Text-Unidecode-0.04/lib/Text/Unidecode.pm ).
|
## Perl module by Sean M. Burke.
|
||||||
##
|
##
|
||||||
## It provides a single proc that does Unicode to ASCII transliterations:
|
## It provides a `unidecode proc <#unidecode,string>`_ that does
|
||||||
## It finds the sequence of ASCII characters that is the closest approximation
|
## Unicode to ASCII transliterations: It finds the sequence of ASCII characters
|
||||||
## to the Unicode string.
|
## that is the closest approximation to the Unicode string.
|
||||||
##
|
##
|
||||||
## For example, the closest to string "Äußerst" in ASCII is "Ausserst". Some
|
## For example, the closest to string "Äußerst" in ASCII is "Ausserst". Some
|
||||||
## information is lost in this transformation, of course, since several Unicode
|
## information is lost in this transformation, of course, since several Unicode
|
||||||
## strings can be transformed in the same ASCII representation. So this is a
|
## strings can be transformed to the same ASCII representation. So this is a
|
||||||
## strictly one-way transformation. However a human reader will probably
|
## strictly one-way transformation. However, a human reader will probably
|
||||||
## still be able to guess what original string was meant from the context.
|
## still be able to guess from the context, what the original string was.
|
||||||
##
|
##
|
||||||
## This module needs the data file "unidecode.dat" to work: This file is
|
## This module needs the data file `unidecode.dat` to work: This file is
|
||||||
## embedded as a resource into your application by default. But you an also
|
## embedded as a resource into your application by default. You can also
|
||||||
## define the symbol ``--define:noUnidecodeTable`` during compile time and
|
## define the symbol `--define:noUnidecodeTable` during compile time and
|
||||||
## use the `loadUnidecodeTable` proc to initialize this module.
|
## use the `loadUnidecodeTable proc <#loadUnidecodeTable>`_ to initialize
|
||||||
|
## this module.
|
||||||
|
|
||||||
import unicode
|
import std/unicode
|
||||||
|
|
||||||
when not defined(noUnidecodeTable):
|
when not defined(noUnidecodeTable):
|
||||||
import strutils
|
import std/strutils
|
||||||
|
|
||||||
const translationTable = splitLines(slurp"unidecode/unidecode.dat")
|
const translationTable = splitLines(slurp"unidecode/unidecode.dat")
|
||||||
else:
|
else:
|
||||||
|
|
@ -38,10 +39,10 @@ else:
|
||||||
var translationTable: seq[string]
|
var translationTable: seq[string]
|
||||||
|
|
||||||
proc loadUnidecodeTable*(datafile = "unidecode.dat") =
|
proc loadUnidecodeTable*(datafile = "unidecode.dat") =
|
||||||
## loads the datafile that `unidecode` to work. This is only required if
|
## Loads the datafile that `unidecode <#unidecode,string>`_ needs to work.
|
||||||
## the module was compiled with the ``--define:noUnidecodeTable`` switch.
|
## This is only required if the module was compiled with the
|
||||||
## This needs to be called by the main thread before any thread can make a
|
## `--define:noUnidecodeTable` switch. This needs to be called by the
|
||||||
## call to `unidecode`.
|
## main thread before any thread can make a call to `unidecode`.
|
||||||
when defined(noUnidecodeTable):
|
when defined(noUnidecodeTable):
|
||||||
newSeq(translationTable, 0xffff)
|
newSeq(translationTable, 0xffff)
|
||||||
var i = 0
|
var i = 0
|
||||||
|
|
@ -53,10 +54,11 @@ proc unidecode*(s: string): string =
|
||||||
## Finds the sequence of ASCII characters that is the closest approximation
|
## Finds the sequence of ASCII characters that is the closest approximation
|
||||||
## to the UTF-8 string `s`.
|
## to the UTF-8 string `s`.
|
||||||
runnableExamples:
|
runnableExamples:
|
||||||
assert unidecode("北京") == "Bei Jing "
|
doAssert unidecode("北京") == "Bei Jing "
|
||||||
|
doAssert unidecode("Äußerst") == "Ausserst"
|
||||||
|
|
||||||
result = ""
|
result = ""
|
||||||
for r in runes(s):
|
for r in runes(s):
|
||||||
var c = int(r)
|
var c = int(r)
|
||||||
if c <=% 127: add(result, chr(c))
|
if c <=% 127: add(result, chr(c))
|
||||||
elif c <% translationTable.len: add(result, translationTable[c-128])
|
elif c <% translationTable.len: add(result, translationTable[c - 128])
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue