RST: implement internal targets (#16614)
This commit is contained in:
parent
335f849c36
commit
fd5c8ef208
6 changed files with 379 additions and 101 deletions
|
|
@ -7,10 +7,19 @@
|
|||
# distribution, for details about the copyright.
|
||||
#
|
||||
|
||||
## This module implements a `reStructuredText`:idx: (RST) parser. A large
|
||||
## subset is implemented. Some features of the `markdown`:idx: syntax are
|
||||
## also supported. Nim can output the result to HTML (command ``rst2html``)
|
||||
## or Latex (command ``rst2tex``).
|
||||
## ==================================
|
||||
## rst: Nim-flavored reStructuredText
|
||||
## ==================================
|
||||
##
|
||||
## This module implements a `reStructuredText`:idx: (RST) parser.
|
||||
## A large subset is implemented with some limitations_ and
|
||||
## `Nim-specific features`_.
|
||||
## A few `extra features`_ of the `Markdown`:idx: syntax are
|
||||
## also supported.
|
||||
##
|
||||
## Nim can output the result to HTML (commands ``nim doc`` for
|
||||
## ``*.nim`` files and ``nim rst2html`` for ``*.rst`` files) or
|
||||
## Latex (command ``nim rst2tex`` for ``*.rst``).
|
||||
##
|
||||
## If you are new to RST please consider reading the following:
|
||||
##
|
||||
|
|
@ -18,6 +27,9 @@
|
|||
## 2) an `RST reference`_: a comprehensive cheatsheet for RST
|
||||
## 3) a more formal 50-page `RST specification`_.
|
||||
##
|
||||
## Features
|
||||
## --------
|
||||
##
|
||||
## Supported standard RST features:
|
||||
##
|
||||
## * body elements
|
||||
|
|
@ -43,27 +55,33 @@
|
|||
## + comments
|
||||
## * inline markup
|
||||
## + *emphasis*, **strong emphasis**,
|
||||
## ``inline literals``, hyperlink references, substitution references,
|
||||
## standalone hyperlinks
|
||||
## ``inline literals``, hyperlink references (including embedded URI),
|
||||
## substitution references, standalone hyperlinks,
|
||||
## internal links (inline and outline)
|
||||
## + \`interpreted text\` with roles ``:literal:``, ``:strong:``,
|
||||
## ``emphasis``, ``:sub:``/``:subscript:``, ``:sup:``/``:supscript:``
|
||||
## (see `RST roles list`_ for description).
|
||||
## + inline internal targets
|
||||
##
|
||||
## Additional features:
|
||||
## .. _`Nim-specific features`:
|
||||
##
|
||||
## Additional Nim-specific features:
|
||||
##
|
||||
## * directives: ``code-block``, ``title``, ``index``
|
||||
## * ***triple emphasis*** (bold and italic) using \*\*\*
|
||||
## * ``:idx:`` role for \`interpreted text\` to include the link to this
|
||||
## text into an index (example: `Nim index`_).
|
||||
##
|
||||
## .. _`extra features`:
|
||||
##
|
||||
## Optional additional features, turned on by ``options: RstParseOption`` in
|
||||
## `rstParse proc <#rstParse,string,string,int,int,bool,RstParseOptions,FindFileHandler,MsgHandler>`_:
|
||||
##
|
||||
## * emoji / smiley symbols
|
||||
## * markdown tables
|
||||
## * markdown code blocks
|
||||
## * markdown links
|
||||
## * markdown headlines
|
||||
## * Markdown tables
|
||||
## * Markdown code blocks
|
||||
## * Markdown links
|
||||
## * Markdown headlines
|
||||
## * using ``1`` as auto-enumerator in enumerated lists like RST ``#``
|
||||
## (auto-enumerator ``1`` can not be used with ``#`` in the same list)
|
||||
##
|
||||
|
|
@ -73,7 +91,8 @@
|
|||
## .. warning:: Using Nim-specific features can cause other RST implementations
|
||||
## to fail on your document.
|
||||
##
|
||||
## Limitations:
|
||||
## Limitations
|
||||
## -----------
|
||||
##
|
||||
## * no Unicode support in character width calculations
|
||||
## * body elements
|
||||
|
|
@ -89,10 +108,21 @@
|
|||
## - no ``role`` directives and no custom interpreted text roles
|
||||
## - some standard roles are not supported (check `RST roles list`_)
|
||||
## - no footnotes & citations support
|
||||
## - no inline internal targets
|
||||
## * inline markup
|
||||
## - no simple-inline-markup
|
||||
## - no embedded URI and aliases
|
||||
## - no embedded aliases
|
||||
##
|
||||
## Usage
|
||||
## -----
|
||||
##
|
||||
## See `Nim DocGen Tools Guide <docgen.html>`_ for the details about
|
||||
## ``nim doc``, ``nim rst2html`` and ``nim rst2tex`` commands.
|
||||
##
|
||||
## See `packages/docutils/rstgen module <rstgen.html>`_ to know how to
|
||||
## generate HTML or Latex strings to embed them into your documents.
|
||||
##
|
||||
## .. Tip:: Import ``packages/docutils/rst`` to use this module
|
||||
## programmatically.
|
||||
##
|
||||
## .. _quick introduction: https://docutils.sourceforge.io/docs/user/rst/quickstart.html
|
||||
## .. _RST reference: https://docutils.sourceforge.io/docs/user/rst/quickref.html
|
||||
|
|
@ -100,13 +130,6 @@
|
|||
## .. _RST directives list: https://docutils.sourceforge.io/docs/ref/rst/directives.html
|
||||
## .. _RST roles list: https://docutils.sourceforge.io/docs/ref/rst/roles.html
|
||||
## .. _Nim index: https://nim-lang.org/docs/theindex.html
|
||||
##
|
||||
## See `Nim DocGen Tools Guide <docgen.html>`_ for the details about
|
||||
## ``nim doc``, ``nim rst2html`` and ``nim rst2tex`` commands.
|
||||
##
|
||||
## .. note:: Import ``packages/docutils/rst`` to use this module.
|
||||
##
|
||||
## See also `packages/docutils/rstgen module <rstgen.html>`_.
|
||||
|
||||
import
|
||||
os, strutils, rstast
|
||||
|
|
@ -118,7 +141,7 @@ type
|
|||
roSupportSmilies, ## make the RST parser support smilies like ``:)``
|
||||
roSupportRawDirective, ## support the ``raw`` directive (don't support
|
||||
## it for sandboxing)
|
||||
roSupportMarkdown ## support additional features of markdown
|
||||
roSupportMarkdown ## support additional features of Markdown
|
||||
|
||||
RstParseOptions* = set[RstParseOption]
|
||||
|
||||
|
|
@ -131,7 +154,7 @@ type
|
|||
meCannotOpenFile = "cannot open '$1'",
|
||||
meExpected = "'$1' expected",
|
||||
meGridTableNotImplemented = "grid table is not implemented",
|
||||
meMarkdownIllformedTable = "illformed delimiter row of a markdown table",
|
||||
meMarkdownIllformedTable = "illformed delimiter row of a Markdown table",
|
||||
meNewSectionExpected = "new section expected",
|
||||
meGeneralParseError = "general parse error",
|
||||
meInvalidDirective = "invalid directive: '$1'",
|
||||
|
|
@ -379,12 +402,16 @@ type
|
|||
Substitution = object
|
||||
key*: string
|
||||
value*: PRstNode
|
||||
AnchorSubst = tuple
|
||||
mainAnchor: string
|
||||
aliases: seq[string]
|
||||
|
||||
SharedState = object
|
||||
options: RstParseOptions # parsing options
|
||||
uLevel, oLevel: int # counters for the section levels
|
||||
subs: seq[Substitution] # substitutions
|
||||
refs: seq[Substitution] # references
|
||||
anchors: seq[AnchorSubst] # internal target substitutions
|
||||
underlineToLevel: LevelMap # Saves for each possible title adornment
|
||||
# character its level in the
|
||||
# current document.
|
||||
|
|
@ -405,6 +432,7 @@ type
|
|||
filename*: string
|
||||
line*, col*: int
|
||||
hasToc*: bool
|
||||
curAnchor*: string # variable to track latest anchor in s.anchors
|
||||
|
||||
EParseError* = object of ValueError
|
||||
|
||||
|
|
@ -577,6 +605,38 @@ proc findRef(p: var RstParser, key: string): PRstNode =
|
|||
if key == p.s.refs[i].key:
|
||||
return p.s.refs[i].value
|
||||
|
||||
proc addAnchor(p: var RstParser, refn: string, reset: bool) =
|
||||
## add anchor `refn` to anchor aliases and update last anchor ``curAnchor``
|
||||
if p.curAnchor == "":
|
||||
p.s.anchors.add (refn, @[refn])
|
||||
else:
|
||||
p.s.anchors[^1].mainAnchor = refn
|
||||
p.s.anchors[^1].aliases.add refn
|
||||
if reset:
|
||||
p.curAnchor = ""
|
||||
else:
|
||||
p.curAnchor = refn
|
||||
|
||||
proc findMainAnchor(p: RstParser, refn: string): string =
|
||||
for subst in p.s.anchors:
|
||||
if subst.mainAnchor == refn: # no need to rename
|
||||
result = subst.mainAnchor
|
||||
break
|
||||
var toLeave = false
|
||||
for anchor in subst.aliases:
|
||||
if anchor == refn: # this anchor will be named as mainAnchor
|
||||
result = subst.mainAnchor
|
||||
toLeave = true
|
||||
if toLeave:
|
||||
break
|
||||
|
||||
proc newRstNodeA(p: var RstParser, kind: RstNodeKind): PRstNode =
|
||||
## create node and consume the current anchor
|
||||
result = newRstNode(kind)
|
||||
if p.curAnchor != "":
|
||||
result.anchor = p.curAnchor
|
||||
p.curAnchor = ""
|
||||
|
||||
proc newLeaf(p: var RstParser): PRstNode =
|
||||
result = newRstNode(rnLeaf, currentTok(p).symbol)
|
||||
|
||||
|
|
@ -629,7 +689,10 @@ proc isInlineMarkupEnd(p: RstParser, markup: string): bool =
|
|||
proc isInlineMarkupStart(p: RstParser, markup: string): bool =
|
||||
# rst rules: https://docutils.sourceforge.io/docs/ref/rst/restructuredtext.html#inline-markup-recognition-rules
|
||||
var d: char
|
||||
result = currentTok(p).symbol == markup
|
||||
if markup != "_`":
|
||||
result = currentTok(p).symbol == markup
|
||||
else: # _` is a 2 token case
|
||||
result = currentTok(p).symbol == "_" and nextTok(p).symbol == "`"
|
||||
if not result: return
|
||||
# Rule 6:
|
||||
result = p.idx == 0 or prevTok(p).kind in {tkIndent, tkWhite} or
|
||||
|
|
@ -873,7 +936,7 @@ proc parseMarkdownCodeblock(p: var RstParser): PRstNode =
|
|||
inc p.idx
|
||||
var lb = newRstNode(rnLiteralBlock)
|
||||
lb.add(n)
|
||||
result = newRstNode(rnCodeBlock)
|
||||
result = newRstNodeA(p, rnCodeBlock)
|
||||
result.add(args)
|
||||
result.add(PRstNode(nil))
|
||||
result.add(lb)
|
||||
|
|
@ -918,6 +981,13 @@ proc parseInline(p: var RstParser, father: PRstNode) =
|
|||
var n = newRstNode(rnEmphasis)
|
||||
parseUntil(p, n, "*", true)
|
||||
father.add(n)
|
||||
elif isInlineMarkupStart(p, "_`"):
|
||||
var n = newRstNode(rnInlineTarget)
|
||||
inc p.idx
|
||||
parseUntil(p, n, "`", false)
|
||||
let refn = rstnodeToRefname(n)
|
||||
p.s.anchors.add (refn, @[refn])
|
||||
father.add(n)
|
||||
elif roSupportMarkdown in p.s.options and currentTok(p).symbol == "```":
|
||||
inc p.idx
|
||||
father.add(parseMarkdownCodeblock(p))
|
||||
|
|
@ -1049,7 +1119,7 @@ proc parseFields(p: var RstParser): PRstNode =
|
|||
if currentTok(p).kind == tkIndent and nextTok(p).symbol == ":" or
|
||||
atStart:
|
||||
var col = if atStart: currentTok(p).col else: currentTok(p).ival
|
||||
result = newRstNode(rnFieldList)
|
||||
result = newRstNodeA(p, rnFieldList)
|
||||
if not atStart: inc p.idx
|
||||
while true:
|
||||
result.add(parseField(p))
|
||||
|
|
@ -1090,7 +1160,7 @@ proc getArgument(n: PRstNode): string =
|
|||
|
||||
proc parseDotDot(p: var RstParser): PRstNode {.gcsafe.}
|
||||
proc parseLiteralBlock(p: var RstParser): PRstNode =
|
||||
result = newRstNode(rnLiteralBlock)
|
||||
result = newRstNodeA(p, rnLiteralBlock)
|
||||
var n = newRstNode(rnLeaf, "")
|
||||
if currentTok(p).kind == tkIndent:
|
||||
var indent = currentTok(p).ival
|
||||
|
|
@ -1248,7 +1318,7 @@ proc parseLineBlock(p: var RstParser): PRstNode =
|
|||
result = nil
|
||||
if nextTok(p).kind in {tkWhite, tkIndent}:
|
||||
var col = currentTok(p).col
|
||||
result = newRstNode(rnLineBlock)
|
||||
result = newRstNodeA(p, rnLineBlock)
|
||||
while true:
|
||||
var item = newRstNode(rnLineBlockItem)
|
||||
if nextTok(p).kind == tkWhite:
|
||||
|
|
@ -1314,6 +1384,7 @@ proc parseHeadline(p: var RstParser): PRstNode =
|
|||
var c = nextTok(p).symbol[0]
|
||||
inc p.idx, 2
|
||||
result.level = getLevel(p.s.underlineToLevel, p.s.uLevel, c)
|
||||
addAnchor(p, rstnodeToRefname(result), reset=true)
|
||||
|
||||
type
|
||||
IntSeq = seq[int]
|
||||
|
|
@ -1349,7 +1420,7 @@ proc parseSimpleTable(p: var RstParser): PRstNode =
|
|||
c: char
|
||||
q: RstParser
|
||||
a, b: PRstNode
|
||||
result = newRstNode(rnTable)
|
||||
result = newRstNodeA(p, rnTable)
|
||||
cols = @[]
|
||||
row = @[]
|
||||
a = nil
|
||||
|
|
@ -1428,7 +1499,7 @@ proc parseMarkdownTable(p: var RstParser): PRstNode =
|
|||
colNum: int
|
||||
a, b: PRstNode
|
||||
q: RstParser
|
||||
result = newRstNode(rnMarkdownTable)
|
||||
result = newRstNodeA(p, rnMarkdownTable)
|
||||
|
||||
proc parseRow(p: var RstParser, cellKind: RstNodeKind, result: PRstNode) =
|
||||
row = readTableRow(p)
|
||||
|
|
@ -1452,7 +1523,7 @@ proc parseMarkdownTable(p: var RstParser): PRstNode =
|
|||
parseRow(p, rnTableDataCell, result)
|
||||
|
||||
proc parseTransition(p: var RstParser): PRstNode =
|
||||
result = newRstNode(rnTransition)
|
||||
result = newRstNodeA(p, rnTransition)
|
||||
inc p.idx
|
||||
if currentTok(p).kind == tkIndent: inc p.idx
|
||||
if currentTok(p).kind == tkIndent: inc p.idx
|
||||
|
|
@ -1475,13 +1546,14 @@ proc parseOverline(p: var RstParser): PRstNode =
|
|||
if currentTok(p).kind == tkAdornment:
|
||||
inc p.idx # XXX: check?
|
||||
if currentTok(p).kind == tkIndent: inc p.idx
|
||||
addAnchor(p, rstnodeToRefname(result), reset=true)
|
||||
|
||||
proc parseBulletList(p: var RstParser): PRstNode =
|
||||
result = nil
|
||||
if nextTok(p).kind == tkWhite:
|
||||
var bullet = currentTok(p).symbol
|
||||
var col = currentTok(p).col
|
||||
result = newRstNode(rnBulletList)
|
||||
result = newRstNodeA(p, rnBulletList)
|
||||
pushInd(p, p.tok[p.idx + 2].col)
|
||||
inc p.idx, 2
|
||||
while true:
|
||||
|
|
@ -1497,7 +1569,7 @@ proc parseBulletList(p: var RstParser): PRstNode =
|
|||
popInd(p)
|
||||
|
||||
proc parseOptionList(p: var RstParser): PRstNode =
|
||||
result = newRstNode(rnOptionList)
|
||||
result = newRstNodeA(p, rnOptionList)
|
||||
while true:
|
||||
if isOptionList(p):
|
||||
var a = newRstNode(rnOptionGroup)
|
||||
|
|
@ -1530,7 +1602,7 @@ proc parseDefinitionList(p: var RstParser): PRstNode =
|
|||
if j >= 1 and p.tok[j].kind == tkIndent and
|
||||
p.tok[j].ival > currInd(p) and p.tok[j - 1].symbol != "::":
|
||||
var col = currentTok(p).col
|
||||
result = newRstNode(rnDefList)
|
||||
result = newRstNodeA(p, rnDefList)
|
||||
while true:
|
||||
j = p.idx
|
||||
var a = newRstNode(rnDefName)
|
||||
|
|
@ -1568,7 +1640,7 @@ proc parseEnumList(p: var RstParser): PRstNode =
|
|||
wildToken: array[0..5, int] = [4, 3, 3, 4, 3, 3] # number of tokens
|
||||
wildIndex: array[0..5, int] = [1, 0, 0, 1, 0, 0]
|
||||
# position of enumeration sequence (number/letter) in enumerator
|
||||
result = newRstNode(rnEnumList)
|
||||
result = newRstNodeA(p, rnEnumList)
|
||||
let col = currentTok(p).col
|
||||
var w = 0
|
||||
while w < wildcards.len:
|
||||
|
|
@ -1623,6 +1695,7 @@ proc sonKind(father: PRstNode, i: int): RstNodeKind =
|
|||
if i < father.len: result = father.sons[i].kind
|
||||
|
||||
proc parseSection(p: var RstParser, result: PRstNode) =
|
||||
## parse top-level RST elements: sections, transitions and body elements.
|
||||
while true:
|
||||
var leave = false
|
||||
assert(p.idx >= 0)
|
||||
|
|
@ -1631,7 +1704,7 @@ proc parseSection(p: var RstParser, result: PRstNode) =
|
|||
inc p.idx
|
||||
elif currentTok(p).ival > currInd(p):
|
||||
pushInd(p, currentTok(p).ival)
|
||||
var a = newRstNode(rnBlockQuote)
|
||||
var a = newRstNodeA(p, rnBlockQuote)
|
||||
parseSection(p, a)
|
||||
result.add(a)
|
||||
popInd(p)
|
||||
|
|
@ -1667,7 +1740,7 @@ proc parseSection(p: var RstParser, result: PRstNode) =
|
|||
#InternalError("rst.parseSection()")
|
||||
discard
|
||||
if a == nil and k != rnDirective:
|
||||
a = newRstNode(rnParagraph)
|
||||
a = newRstNodeA(p, rnParagraph)
|
||||
parseParagraph(p, a)
|
||||
result.addIfNotNil(a)
|
||||
if sonKind(result, 0) == rnParagraph and sonKind(result, 1) != rnParagraph:
|
||||
|
|
@ -1703,7 +1776,7 @@ proc parseDirective(p: var RstParser, flags: DirFlags): PRstNode =
|
|||
##
|
||||
## Both rnDirArg and rnFieldList children nodes might be nil, so you need to
|
||||
## check them before accessing.
|
||||
result = newRstNode(rnDirective)
|
||||
result = newRstNodeA(p, rnDirective)
|
||||
var args: PRstNode = nil
|
||||
var options: PRstNode = nil
|
||||
if hasArg in flags:
|
||||
|
|
@ -1981,7 +2054,10 @@ proc parseDotDot(p: var RstParser): PRstNode =
|
|||
var a = getReferenceName(p, ":")
|
||||
if currentTok(p).kind == tkWhite: inc p.idx
|
||||
var b = untilEol(p)
|
||||
setRef(p, rstnodeToRefname(a), b)
|
||||
if len(b) == 0 and b.text == "": # set internal anchor
|
||||
addAnchor(p, rstnodeToRefname(a), reset=false)
|
||||
else: # external hyperlink
|
||||
setRef(p, rstnodeToRefname(a), b)
|
||||
elif match(p, p.idx, " |"):
|
||||
# substitution definitions:
|
||||
inc p.idx, 2
|
||||
|
|
@ -2009,6 +2085,7 @@ proc parseDotDot(p: var RstParser): PRstNode =
|
|||
result = parseComment(p)
|
||||
|
||||
proc resolveSubs(p: var RstParser, n: PRstNode): PRstNode =
|
||||
## resolve substitutions and anchor aliases
|
||||
result = n
|
||||
if n == nil: return
|
||||
case n.kind
|
||||
|
|
@ -2022,12 +2099,20 @@ proc resolveSubs(p: var RstParser, n: PRstNode): PRstNode =
|
|||
if e != "": result = newRstNode(rnLeaf, e)
|
||||
else: rstMessage(p, mwUnknownSubstitution, key)
|
||||
of rnRef:
|
||||
var y = findRef(p, rstnodeToRefname(n))
|
||||
let refn = rstnodeToRefname(n)
|
||||
var y = findRef(p, refn)
|
||||
if y != nil:
|
||||
result = newRstNode(rnHyperlink)
|
||||
n.kind = rnInner
|
||||
result.add(n)
|
||||
result.add(y)
|
||||
else:
|
||||
let s = findMainAnchor(p, refn)
|
||||
if s != "":
|
||||
result = newRstNode(rnInternalRef)
|
||||
n.kind = rnInner
|
||||
result.add(n)
|
||||
result.add(newRstNode(rnLeaf, s))
|
||||
of rnLeaf:
|
||||
discard
|
||||
of rnContents:
|
||||
|
|
@ -2045,5 +2130,6 @@ proc rstParse*(text, filename: string,
|
|||
p.filename = filename
|
||||
p.line = line
|
||||
p.col = column + getTokens(text, roSkipPounds in options, p.tok)
|
||||
result = resolveSubs(p, parseDoc(p))
|
||||
let unresolved = parseDoc(p)
|
||||
result = resolveSubs(p, unresolved)
|
||||
hasToc = p.hasToc
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue