RST: implement internal targets (#16614)

This commit is contained in:
Andrey Makarov 2021-01-11 21:51:04 +03:00 • committed by GitHub
commit fd5c8ef208
No known key found for this signature in database
GPG key ID: 4AEE18F83AFDEB23
6 changed files with 379 additions and 101 deletions

View file

@ -7,10 +7,19 @@
# distribution, for details about the copyright.
#
## This module implements a `reStructuredText`:idx: (RST) parser. A large
## subset is implemented. Some features of the `markdown`:idx: syntax are
## also supported. Nim can output the result to HTML (command ``rst2html``)
## or Latex (command ``rst2tex``).
## ==================================
## rst: Nim-flavored reStructuredText
## ==================================
##
## This module implements a `reStructuredText`:idx: (RST) parser.
## A large subset is implemented with some limitations_ and
## `Nim-specific features`_.
## A few `extra features`_ of the `Markdown`:idx: syntax are
## also supported.
##
## Nim can output the result to HTML (commands ``nim doc`` for
## ``*.nim`` files and ``nim rst2html`` for ``*.rst`` files) or
## Latex (command ``nim rst2tex`` for ``*.rst``).
##
## If you are new to RST please consider reading the following:
##
@ -18,6 +27,9 @@
## 2) an `RST reference`_: a comprehensive cheatsheet for RST
## 3) a more formal 50-page `RST specification`_.
##
## Features
## --------
##
## Supported standard RST features:
##
## * body elements
@ -43,27 +55,33 @@
## + comments
## * inline markup
## + *emphasis*, **strong emphasis**,
## ``inline literals``, hyperlink references, substitution references,
## standalone hyperlinks
## ``inline literals``, hyperlink references (including embedded URI),
## substitution references, standalone hyperlinks,
## internal links (inline and outline)
## + \`interpreted text\` with roles ``:literal:``, ``:strong:``,
## ``emphasis``, ``:sub:``/``:subscript:``, ``:sup:``/``:supscript:``
## (see `RST roles list`_ for description).
## + inline internal targets
##
## Additional features:
## .. _`Nim-specific features`:
##
## Additional Nim-specific features:
##
## * directives: ``code-block``, ``title``, ``index``
## * ***triple emphasis*** (bold and italic) using \*\*\*
## * ``:idx:`` role for \`interpreted text\` to include the link to this
## text into an index (example: `Nim index`_).
##
## .. _`extra features`:
##
## Optional additional features, turned on by ``options: RstParseOption`` in
## `rstParse proc <#rstParse,string,string,int,int,bool,RstParseOptions,FindFileHandler,MsgHandler>`_:
##
## * emoji / smiley symbols
## * markdown tables
## * markdown code blocks
## * markdown links
## * markdown headlines
## * Markdown tables
## * Markdown code blocks
## * Markdown links
## * Markdown headlines
## * using ``1`` as auto-enumerator in enumerated lists like RST ``#``
## (auto-enumerator ``1`` can not be used with ``#`` in the same list)
##
@ -73,7 +91,8 @@
## .. warning:: Using Nim-specific features can cause other RST implementations
## to fail on your document.
##
## Limitations:
## Limitations
## -----------
##
## * no Unicode support in character width calculations
## * body elements
@ -89,10 +108,21 @@
## - no ``role`` directives and no custom interpreted text roles
## - some standard roles are not supported (check `RST roles list`_)
## - no footnotes & citations support
## - no inline internal targets
## * inline markup
## - no simple-inline-markup
## - no embedded URI and aliases
## - no embedded aliases
##
## Usage
## -----
##
## See `Nim DocGen Tools Guide <docgen.html>`_ for the details about
## ``nim doc``, ``nim rst2html`` and ``nim rst2tex`` commands.
##
## See `packages/docutils/rstgen module <rstgen.html>`_ to know how to
## generate HTML or Latex strings to embed them into your documents.
##
## .. Tip:: Import ``packages/docutils/rst`` to use this module
## programmatically.
##
## .. _quick introduction: https://docutils.sourceforge.io/docs/user/rst/quickstart.html
## .. _RST reference: https://docutils.sourceforge.io/docs/user/rst/quickref.html
@ -100,13 +130,6 @@
## .. _RST directives list: https://docutils.sourceforge.io/docs/ref/rst/directives.html
## .. _RST roles list: https://docutils.sourceforge.io/docs/ref/rst/roles.html
## .. _Nim index: https://nim-lang.org/docs/theindex.html
##
## See `Nim DocGen Tools Guide <docgen.html>`_ for the details about
## ``nim doc``, ``nim rst2html`` and ``nim rst2tex`` commands.
##
## .. note:: Import ``packages/docutils/rst`` to use this module.
##
## See also `packages/docutils/rstgen module <rstgen.html>`_.
import
os, strutils, rstast
@ -118,7 +141,7 @@ type
roSupportSmilies, ## make the RST parser support smilies like ``:)``
roSupportRawDirective, ## support the ``raw`` directive (don't support
## it for sandboxing)
roSupportMarkdown ## support additional features of markdown
roSupportMarkdown ## support additional features of Markdown
RstParseOptions* = set[RstParseOption]
@ -131,7 +154,7 @@ type
meCannotOpenFile = "cannot open '$1'",
meExpected = "'$1' expected",
meGridTableNotImplemented = "grid table is not implemented",
meMarkdownIllformedTable = "illformed delimiter row of a markdown table",
meMarkdownIllformedTable = "illformed delimiter row of a Markdown table",
meNewSectionExpected = "new section expected",
meGeneralParseError = "general parse error",
meInvalidDirective = "invalid directive: '$1'",
@ -379,12 +402,16 @@ type
Substitution = object
key*: string
value*: PRstNode
AnchorSubst = tuple
mainAnchor: string
aliases: seq[string]
SharedState = object
options: RstParseOptions # parsing options
uLevel, oLevel: int # counters for the section levels
subs: seq[Substitution] # substitutions
refs: seq[Substitution] # references
anchors: seq[AnchorSubst] # internal target substitutions
underlineToLevel: LevelMap # Saves for each possible title adornment
# character its level in the
# current document.
@ -405,6 +432,7 @@ type
filename*: string
line*, col*: int
hasToc*: bool
curAnchor*: string # variable to track latest anchor in s.anchors
EParseError* = object of ValueError
@ -577,6 +605,38 @@ proc findRef(p: var RstParser, key: string): PRstNode =
if key == p.s.refs[i].key:
return p.s.refs[i].value
proc addAnchor(p: var RstParser, refn: string, reset: bool) =
## add anchor `refn` to anchor aliases and update last anchor ``curAnchor``
if p.curAnchor == "":
p.s.anchors.add (refn, @[refn])
else:
p.s.anchors[^1].mainAnchor = refn
p.s.anchors[^1].aliases.add refn
if reset:
p.curAnchor = ""
else:
p.curAnchor = refn
proc findMainAnchor(p: RstParser, refn: string): string =
for subst in p.s.anchors:
if subst.mainAnchor == refn: # no need to rename
result = subst.mainAnchor
break
var toLeave = false
for anchor in subst.aliases:
if anchor == refn: # this anchor will be named as mainAnchor
result = subst.mainAnchor
toLeave = true
if toLeave:
break
proc newRstNodeA(p: var RstParser, kind: RstNodeKind): PRstNode =
## create node and consume the current anchor
result = newRstNode(kind)
if p.curAnchor != "":
result.anchor = p.curAnchor
p.curAnchor = ""
proc newLeaf(p: var RstParser): PRstNode =
result = newRstNode(rnLeaf, currentTok(p).symbol)
@ -629,7 +689,10 @@ proc isInlineMarkupEnd(p: RstParser, markup: string): bool =
proc isInlineMarkupStart(p: RstParser, markup: string): bool =
# rst rules: https://docutils.sourceforge.io/docs/ref/rst/restructuredtext.html#inline-markup-recognition-rules
var d: char
result = currentTok(p).symbol == markup
if markup != "_`":
result = currentTok(p).symbol == markup
else: # _` is a 2 token case
result = currentTok(p).symbol == "_" and nextTok(p).symbol == "`"
if not result: return
# Rule 6:
result = p.idx == 0 or prevTok(p).kind in {tkIndent, tkWhite} or
@ -873,7 +936,7 @@ proc parseMarkdownCodeblock(p: var RstParser): PRstNode =
inc p.idx
var lb = newRstNode(rnLiteralBlock)
lb.add(n)
result = newRstNode(rnCodeBlock)
result = newRstNodeA(p, rnCodeBlock)
result.add(args)
result.add(PRstNode(nil))
result.add(lb)
@ -918,6 +981,13 @@ proc parseInline(p: var RstParser, father: PRstNode) =
var n = newRstNode(rnEmphasis)
parseUntil(p, n, "*", true)
father.add(n)
elif isInlineMarkupStart(p, "_`"):
var n = newRstNode(rnInlineTarget)
inc p.idx
parseUntil(p, n, "`", false)
let refn = rstnodeToRefname(n)
p.s.anchors.add (refn, @[refn])
father.add(n)
elif roSupportMarkdown in p.s.options and currentTok(p).symbol == "```":
inc p.idx
father.add(parseMarkdownCodeblock(p))
@ -1049,7 +1119,7 @@ proc parseFields(p: var RstParser): PRstNode =
if currentTok(p).kind == tkIndent and nextTok(p).symbol == ":" or
atStart:
var col = if atStart: currentTok(p).col else: currentTok(p).ival
result = newRstNode(rnFieldList)
result = newRstNodeA(p, rnFieldList)
if not atStart: inc p.idx
while true:
result.add(parseField(p))
@ -1090,7 +1160,7 @@ proc getArgument(n: PRstNode): string =
proc parseDotDot(p: var RstParser): PRstNode {.gcsafe.}
proc parseLiteralBlock(p: var RstParser): PRstNode =
result = newRstNode(rnLiteralBlock)
result = newRstNodeA(p, rnLiteralBlock)
var n = newRstNode(rnLeaf, "")
if currentTok(p).kind == tkIndent:
var indent = currentTok(p).ival
@ -1248,7 +1318,7 @@ proc parseLineBlock(p: var RstParser): PRstNode =
result = nil
if nextTok(p).kind in {tkWhite, tkIndent}:
var col = currentTok(p).col
result = newRstNode(rnLineBlock)
result = newRstNodeA(p, rnLineBlock)
while true:
var item = newRstNode(rnLineBlockItem)
if nextTok(p).kind == tkWhite:
@ -1314,6 +1384,7 @@ proc parseHeadline(p: var RstParser): PRstNode =
var c = nextTok(p).symbol[0]
inc p.idx, 2
result.level = getLevel(p.s.underlineToLevel, p.s.uLevel, c)
addAnchor(p, rstnodeToRefname(result), reset=true)
type
IntSeq = seq[int]
@ -1349,7 +1420,7 @@ proc parseSimpleTable(p: var RstParser): PRstNode =
c: char
q: RstParser
a, b: PRstNode
result = newRstNode(rnTable)
result = newRstNodeA(p, rnTable)
cols = @[]
row = @[]
a = nil
@ -1428,7 +1499,7 @@ proc parseMarkdownTable(p: var RstParser): PRstNode =
colNum: int
a, b: PRstNode
q: RstParser
result = newRstNode(rnMarkdownTable)
result = newRstNodeA(p, rnMarkdownTable)
proc parseRow(p: var RstParser, cellKind: RstNodeKind, result: PRstNode) =
row = readTableRow(p)
@ -1452,7 +1523,7 @@ proc parseMarkdownTable(p: var RstParser): PRstNode =
parseRow(p, rnTableDataCell, result)
proc parseTransition(p: var RstParser): PRstNode =
result = newRstNode(rnTransition)
result = newRstNodeA(p, rnTransition)
inc p.idx
if currentTok(p).kind == tkIndent: inc p.idx
if currentTok(p).kind == tkIndent: inc p.idx
@ -1475,13 +1546,14 @@ proc parseOverline(p: var RstParser): PRstNode =
if currentTok(p).kind == tkAdornment:
inc p.idx # XXX: check?
if currentTok(p).kind == tkIndent: inc p.idx
addAnchor(p, rstnodeToRefname(result), reset=true)
proc parseBulletList(p: var RstParser): PRstNode =
result = nil
if nextTok(p).kind == tkWhite:
var bullet = currentTok(p).symbol
var col = currentTok(p).col
result = newRstNode(rnBulletList)
result = newRstNodeA(p, rnBulletList)
pushInd(p, p.tok[p.idx + 2].col)
inc p.idx, 2
while true:
@ -1497,7 +1569,7 @@ proc parseBulletList(p: var RstParser): PRstNode =
popInd(p)
proc parseOptionList(p: var RstParser): PRstNode =
result = newRstNode(rnOptionList)
result = newRstNodeA(p, rnOptionList)
while true:
if isOptionList(p):
var a = newRstNode(rnOptionGroup)
@ -1530,7 +1602,7 @@ proc parseDefinitionList(p: var RstParser): PRstNode =
if j >= 1 and p.tok[j].kind == tkIndent and
p.tok[j].ival > currInd(p) and p.tok[j - 1].symbol != "::":
var col = currentTok(p).col
result = newRstNode(rnDefList)
result = newRstNodeA(p, rnDefList)
while true:
j = p.idx
var a = newRstNode(rnDefName)
@ -1568,7 +1640,7 @@ proc parseEnumList(p: var RstParser): PRstNode =
wildToken: array[0..5, int] = [4, 3, 3, 4, 3, 3] # number of tokens
wildIndex: array[0..5, int] = [1, 0, 0, 1, 0, 0]
# position of enumeration sequence (number/letter) in enumerator
result = newRstNode(rnEnumList)
result = newRstNodeA(p, rnEnumList)
let col = currentTok(p).col
var w = 0
while w < wildcards.len:
@ -1623,6 +1695,7 @@ proc sonKind(father: PRstNode, i: int): RstNodeKind =
if i < father.len: result = father.sons[i].kind
proc parseSection(p: var RstParser, result: PRstNode) =
## parse top-level RST elements: sections, transitions and body elements.
while true:
var leave = false
assert(p.idx >= 0)
@ -1631,7 +1704,7 @@ proc parseSection(p: var RstParser, result: PRstNode) =
inc p.idx
elif currentTok(p).ival > currInd(p):
pushInd(p, currentTok(p).ival)
var a = newRstNode(rnBlockQuote)
var a = newRstNodeA(p, rnBlockQuote)
parseSection(p, a)
result.add(a)
popInd(p)
@ -1667,7 +1740,7 @@ proc parseSection(p: var RstParser, result: PRstNode) =
#InternalError("rst.parseSection()")
discard
if a == nil and k != rnDirective:
a = newRstNode(rnParagraph)
a = newRstNodeA(p, rnParagraph)
parseParagraph(p, a)
result.addIfNotNil(a)
if sonKind(result, 0) == rnParagraph and sonKind(result, 1) != rnParagraph:
@ -1703,7 +1776,7 @@ proc parseDirective(p: var RstParser, flags: DirFlags): PRstNode =
##
## Both rnDirArg and rnFieldList children nodes might be nil, so you need to
## check them before accessing.
result = newRstNode(rnDirective)
result = newRstNodeA(p, rnDirective)
var args: PRstNode = nil
var options: PRstNode = nil
if hasArg in flags:
@ -1981,7 +2054,10 @@ proc parseDotDot(p: var RstParser): PRstNode =
var a = getReferenceName(p, ":")
if currentTok(p).kind == tkWhite: inc p.idx
var b = untilEol(p)
setRef(p, rstnodeToRefname(a), b)
if len(b) == 0 and b.text == "": # set internal anchor
addAnchor(p, rstnodeToRefname(a), reset=false)
else: # external hyperlink
setRef(p, rstnodeToRefname(a), b)
elif match(p, p.idx, " |"):
# substitution definitions:
inc p.idx, 2
@ -2009,6 +2085,7 @@ proc parseDotDot(p: var RstParser): PRstNode =
result = parseComment(p)
proc resolveSubs(p: var RstParser, n: PRstNode): PRstNode =
## resolve substitutions and anchor aliases
result = n
if n == nil: return
case n.kind
@ -2022,12 +2099,20 @@ proc resolveSubs(p: var RstParser, n: PRstNode): PRstNode =
if e != "": result = newRstNode(rnLeaf, e)
else: rstMessage(p, mwUnknownSubstitution, key)
of rnRef:
var y = findRef(p, rstnodeToRefname(n))
let refn = rstnodeToRefname(n)
var y = findRef(p, refn)
if y != nil:
result = newRstNode(rnHyperlink)
n.kind = rnInner
result.add(n)
result.add(y)
else:
let s = findMainAnchor(p, refn)
if s != "":
result = newRstNode(rnInternalRef)
n.kind = rnInner
result.add(n)
result.add(newRstNode(rnLeaf, s))
of rnLeaf:
discard
of rnContents:
@ -2045,5 +2130,6 @@ proc rstParse*(text, filename: string,
p.filename = filename
p.line = line
p.col = column + getTokens(text, roSkipPounds in options, p.tok)
result = resolveSubs(p, parseDoc(p))
let unresolved = parseDoc(p)
result = resolveSubs(p, unresolved)
hasToc = p.hasToc