Improve documentation for unidecode (#16986)

* Improve documentation for unidecode

Minor changes to gen.py

* Fix typo in gen.py
This commit is contained in:
konsumlamm 2021-02-09 22:47:07 +01:00 • committed by GitHub
commit 00551f972e
No known key found for this signature in database
GPG key ID: 4AEE18F83AFDEB23
2 changed files with 28 additions and 26 deletions

View file

@ -11,7 +11,7 @@ try:
except ImportError: except ImportError:
pass pass
def main2(): def main():
f = open("unidecode.dat", "wb+") f = open("unidecode.dat", "wb+")
for x in range(128, 0xffff + 1): for x in range(128, 0xffff + 1):
u = eval("u'\\u%04x'" % x) u = eval("u'\\u%04x'" % x)
@ -19,12 +19,12 @@ def main2():
val = unidecode(u) val = unidecode(u)
# f.write("%x | " % x) # f.write("%x | " % x)
if x==0x2028: # U+2028 = LINE SEPARATOR if x == 0x2028: # U+2028 = LINE SEPARATOR
val = "" val = ""
elif x==0x2029: # U+2028 = PARAGRAPH SEPARATOR elif x == 0x2029: # U+2029 = PARAGRAPH SEPARATOR
val = "" val = ""
f.write("%s\n" % val) f.write("%s\n" % val)
f.close() f.close()
main2() main()

View file

@ -7,30 +7,31 @@
# distribution, for details about the copyright. # distribution, for details about the copyright.
# #
## This module is based on Python's Unidecode module by Tomaz Solc, ## This module is based on Python's [Unidecode](https://pypi.org/project/Unidecode/)
## which in turn is based on the ``Text::Unidecode`` Perl module by ## module by Tomaz Solc, which in turn is based on the
## Sean M. Burke ## [Text::Unidecode](https://metacpan.org/pod/Text::Unidecode)
## (http://search.cpan.org/~sburke/Text-Unidecode-0.04/lib/Text/Unidecode.pm ). ## Perl module by Sean M. Burke.
## ##
## It provides a single proc that does Unicode to ASCII transliterations: ## It provides a `unidecode proc <#unidecode,string>`_ that does
## It finds the sequence of ASCII characters that is the closest approximation ## Unicode to ASCII transliterations: It finds the sequence of ASCII characters
## to the Unicode string. ## that is the closest approximation to the Unicode string.
## ##
## For example, the closest to string "Äußerst" in ASCII is "Ausserst". Some ## For example, the closest to string "Äußerst" in ASCII is "Ausserst". Some
## information is lost in this transformation, of course, since several Unicode ## information is lost in this transformation, of course, since several Unicode
## strings can be transformed in the same ASCII representation. So this is a ## strings can be transformed to the same ASCII representation. So this is a
## strictly one-way transformation. However a human reader will probably ## strictly one-way transformation. However, a human reader will probably
## still be able to guess what original string was meant from the context. ## still be able to guess from the context, what the original string was.
## ##
## This module needs the data file "unidecode.dat" to work: This file is ## This module needs the data file `unidecode.dat` to work: This file is
## embedded as a resource into your application by default. But you an also ## embedded as a resource into your application by default. You can also
## define the symbol ``--define:noUnidecodeTable`` during compile time and ## define the symbol `--define:noUnidecodeTable` during compile time and
## use the `loadUnidecodeTable` proc to initialize this module. ## use the `loadUnidecodeTable proc <#loadUnidecodeTable>`_ to initialize
## this module.
import unicode import std/unicode
when not defined(noUnidecodeTable): when not defined(noUnidecodeTable):
import strutils import std/strutils
const translationTable = splitLines(slurp"unidecode/unidecode.dat") const translationTable = splitLines(slurp"unidecode/unidecode.dat")
else: else:
@ -38,10 +39,10 @@ else:
var translationTable: seq[string] var translationTable: seq[string]
proc loadUnidecodeTable*(datafile = "unidecode.dat") = proc loadUnidecodeTable*(datafile = "unidecode.dat") =
## loads the datafile that `unidecode` to work. This is only required if ## Loads the datafile that `unidecode <#unidecode,string>`_ needs to work.
## the module was compiled with the ``--define:noUnidecodeTable`` switch. ## This is only required if the module was compiled with the
## This needs to be called by the main thread before any thread can make a ## `--define:noUnidecodeTable` switch. This needs to be called by the
## call to `unidecode`. ## main thread before any thread can make a call to `unidecode`.
when defined(noUnidecodeTable): when defined(noUnidecodeTable):
newSeq(translationTable, 0xffff) newSeq(translationTable, 0xffff)
var i = 0 var i = 0
@ -53,10 +54,11 @@ proc unidecode*(s: string): string =
## Finds the sequence of ASCII characters that is the closest approximation ## Finds the sequence of ASCII characters that is the closest approximation
## to the UTF-8 string `s`. ## to the UTF-8 string `s`.
runnableExamples: runnableExamples:
assert unidecode("北京") == "Bei Jing " doAssert unidecode("北京") == "Bei Jing "
doAssert unidecode("Äußerst") == "Ausserst"
result = "" result = ""
for r in runes(s): for r in runes(s):
var c = int(r) var c = int(r)
if c <=% 127: add(result, chr(c)) if c <=% 127: add(result, chr(c))
elif c <% translationTable.len: add(result, translationTable[c-128]) elif c <% translationTable.len: add(result, translationTable[c - 128])