This commit is contained in:
parent
80389b8053
commit
8f79bc5f3d
10 changed files with 291 additions and 140 deletions
|
|
@ -37,6 +37,18 @@
|
|||
## .. code:: Nim
|
||||
## for l in ["C", "c++", "jAvA", "Nim", "c#"]: echo getSourceLanguage(l)
|
||||
##
|
||||
## There is also a `Cmd` pseudo-language supported, which is a simple generic
|
||||
## shell/cmdline tokenizer (UNIX shell/Powershell/Windows Command):
|
||||
## no escaping, no programming language constructs besides variable definition
|
||||
## at the beginning of line. It supports these operators:
|
||||
##
|
||||
## .. code:: Cmd
|
||||
## & && | || ( ) '' "" ; # for comments
|
||||
##
|
||||
## Instead of escaping always use quotes like here
|
||||
## `nimgrep --ext:'nim|nims' file.name`:cmd: shows how to input ``|``.
|
||||
## Any argument that contains ``.`` or ``/`` or ``\`` will be treated
|
||||
## as a file or directory.
|
||||
|
||||
import
|
||||
strutils
|
||||
|
|
@ -45,7 +57,7 @@ from algorithm import binarySearch
|
|||
type
|
||||
SourceLanguage* = enum
|
||||
langNone, langNim, langCpp, langCsharp, langC, langJava,
|
||||
langYaml, langPython
|
||||
langYaml, langPython, langCmd
|
||||
TokenClass* = enum
|
||||
gtEof, gtNone, gtWhitespace, gtDecNumber, gtBinNumber, gtHexNumber,
|
||||
gtOctNumber, gtFloatNumber, gtIdentifier, gtKeyword, gtStringLit,
|
||||
|
|
@ -53,7 +65,7 @@ type
|
|||
gtOperator, gtPunctuation, gtComment, gtLongComment, gtRegularExpression,
|
||||
gtTagStart, gtTagEnd, gtKey, gtValue, gtRawData, gtAssembler,
|
||||
gtPreprocessor, gtDirective, gtCommand, gtRule, gtHyperlink, gtLabel,
|
||||
gtReference, gtOther
|
||||
gtReference, gtProgram, gtOption, gtOther
|
||||
GeneralTokenizer* = object of RootObj
|
||||
kind*: TokenClass
|
||||
start*, length*: int
|
||||
|
|
@ -64,14 +76,17 @@ type
|
|||
|
||||
const
|
||||
sourceLanguageToStr*: array[SourceLanguage, string] = ["none",
|
||||
"Nim", "C++", "C#", "C", "Java", "Yaml", "Python"]
|
||||
"Nim", "C++", "C#", "C", "Java", "Yaml", "Python", "Cmd"]
|
||||
tokenClassToStr*: array[TokenClass, string] = ["Eof", "None", "Whitespace",
|
||||
"DecNumber", "BinNumber", "HexNumber", "OctNumber", "FloatNumber",
|
||||
"Identifier", "Keyword", "StringLit", "LongStringLit", "CharLit",
|
||||
"EscapeSequence", "Operator", "Punctuation", "Comment", "LongComment",
|
||||
"RegularExpression", "TagStart", "TagEnd", "Key", "Value", "RawData",
|
||||
"Assembler", "Preprocessor", "Directive", "Command", "Rule", "Hyperlink",
|
||||
"Label", "Reference", "Other"]
|
||||
"Label", "Reference",
|
||||
# start from lower-case if there is a corresponding RST role (see rst.nim)
|
||||
"program", "option",
|
||||
"Other"]
|
||||
|
||||
# The following list comes from doc/keywords.txt, make sure it is
|
||||
# synchronized with this array by running the module itself as a test case.
|
||||
|
|
@ -898,6 +913,65 @@ proc pythonNextToken(g: var GeneralTokenizer) =
|
|||
"with", "yield"]
|
||||
nimNextToken(g, keywords)
|
||||
|
||||
proc cmdNextToken(g: var GeneralTokenizer) =
|
||||
var pos = g.pos
|
||||
g.start = g.pos
|
||||
if g.state == low(TokenClass):
|
||||
g.state = gtProgram
|
||||
case g.buf[pos]
|
||||
of ' ', '\t'..'\r':
|
||||
g.kind = gtWhitespace
|
||||
while g.buf[pos] in {' ', '\t'..'\r'}:
|
||||
if g.buf[pos] == '\n':
|
||||
g.state = gtProgram
|
||||
inc(pos)
|
||||
of '\'', '"':
|
||||
g.kind = gtOption
|
||||
let q = g.buf[pos]
|
||||
inc(pos)
|
||||
while g.buf[pos] notin {q, '\0'}:
|
||||
inc(pos)
|
||||
if g.buf[pos] == q: inc(pos)
|
||||
of '#':
|
||||
g.kind = gtComment
|
||||
while g.buf[pos] notin {'\n', '\0'}:
|
||||
inc(pos)
|
||||
of '&', '|':
|
||||
g.kind = gtOperator
|
||||
inc(pos)
|
||||
if g.buf[pos] == g.buf[pos-1]: inc(pos)
|
||||
g.state = gtProgram
|
||||
of '(':
|
||||
g.kind = gtOperator
|
||||
g.state = gtProgram
|
||||
inc(pos)
|
||||
of ')':
|
||||
g.kind = gtOperator
|
||||
inc(pos)
|
||||
of ';':
|
||||
g.state = gtProgram
|
||||
g.kind = gtOperator
|
||||
inc(pos)
|
||||
of '\0': g.kind = gtEof
|
||||
else:
|
||||
if g.state == gtProgram:
|
||||
g.kind = gtProgram
|
||||
g.state = gtOption
|
||||
else:
|
||||
g.kind = gtOption
|
||||
while g.buf[pos] notin {' ', '\t'..'\r', '&', '|', '(', ')', '\'', '"', '\0'}:
|
||||
if g.buf[pos] == ';' and g.buf[pos+1] == ' ':
|
||||
# (check space because ';' can be used inside arguments in Win bat)
|
||||
break
|
||||
if g.kind == gtOption and g.buf[pos] in {'/', '\\', '.'}:
|
||||
g.kind = gtIdentifier # for file/dir name
|
||||
elif g.kind == gtProgram and g.buf[pos] == '=':
|
||||
g.kind = gtIdentifier # for env variable setting at beginning of line
|
||||
g.state = gtProgram
|
||||
inc(pos)
|
||||
g.length = pos - g.pos
|
||||
g.pos = pos
|
||||
|
||||
proc getNextToken*(g: var GeneralTokenizer, lang: SourceLanguage) =
|
||||
g.lang = lang
|
||||
case lang
|
||||
|
|
@ -909,6 +983,7 @@ proc getNextToken*(g: var GeneralTokenizer, lang: SourceLanguage) =
|
|||
of langJava: javaNextToken(g)
|
||||
of langYaml: yamlNextToken(g)
|
||||
of langPython: pythonNextToken(g)
|
||||
of langCmd: cmdNextToken(g)
|
||||
|
||||
when isMainModule:
|
||||
var keywords: seq[string]
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue