Implement Markdown definition lists (+ migration) (#20333)

Implements definition lists Markdown extension adopted in a few
implementations including:
* [Pandoc](
  https://pandoc.org/MANUAL.html#definition-lists)
* [kramdown](
  https://kramdown.gettalong.org/quickref.html#definition-lists)
* [PHP extra Markdown](
  https://michelf.ca/projects/php-markdown/extra/#def-list)

Also affected files have been migrated.
RST definition lists are turned off for Markdown: this solves the
problem of broken formatting mentioned in
https://github.com/nim-lang/Nim/pull/20292.
This commit is contained in:
Andrey Makarov 2022-09-11 20:52:43 +03:00 • committed by GitHub
commit 088487f652
No known key found for this signature in database
GPG key ID: 4AEE18F83AFDEB23
15 changed files with 266 additions and 97 deletions

View file

@ -75,16 +75,16 @@ type
## comment".`
##
## `pattern: string`
## the string that was used to create the pattern. For details on how
## : the string that was used to create the pattern. For details on how
## to write a pattern, please see `the official PCRE pattern
## documentation.
## <https://www.pcre.org/original/doc/html/pcrepattern.html>`_
##
## `captureCount: int`
## the number of captures that the pattern has.
## : the number of captures that the pattern has.
##
## `captureNameId: Table[string, int]`
## a table from the capture names to their numeric id.
## : a table from the capture names to their numeric id.
##
##
## Options
@ -151,36 +151,36 @@ type
## execution. On failure, it is none, on success, it is some.
##
## `pattern: Regex`
## the pattern that is being matched
## : the pattern that is being matched
##
## `str: string`
## the string that was matched against
## : the string that was matched against
##
## `captures[]: string`
## the string value of whatever was captured at that id. If the value
## : the string value of whatever was captured at that id. If the value
## is invalid, then behavior is undefined. If the id is `-1`, then
## the whole match is returned. If the given capture was not matched,
## `nil` is returned. See examples for `match`.
##
## `captureBounds[]: HSlice[int, int]`
## gets the bounds of the given capture according to the same rules as
## : gets the bounds of the given capture according to the same rules as
## the above. If the capture is not filled, then `None` is returned.
## The bounds are both inclusive. See examples for `match`.
##
## `match: string`
## the full text of the match.
## : the full text of the match.
##
## `matchBounds: HSlice[int, int]`
## the bounds of the match, as in `captureBounds[]`
## : the bounds of the match, as in `captureBounds[]`
##
## `(captureBounds|captures).toTable`
## returns a table with each named capture as a key.
## : returns a table with each named capture as a key.
##
## `(captureBounds|captures).toSeq`
## returns all the captures by their number.
## : returns all the captures by their number.
##
## `$: string`
## same as `match`
## : same as `match`
pattern*: Regex ## The regex doing the matching.
## Not nil.
str*: string ## The string that was matched against.
@ -583,11 +583,11 @@ proc find*(str: string, pattern: Regex, start = 0, endpos = int.high): Option[Re
## positions.
##
## `start`
## The start point at which to start matching. `|abc` is `0`;
## : The start point at which to start matching. `|abc` is `0`;
## `a|bc` is `1`
##
## `endpos`
## The maximum index for a match; `int.high` means the end of the
## : The maximum index for a match; `int.high` means the end of the
## string, otherwise it’s an inclusive upper bound.
return str.matchImpl(pattern, start, endpos, 0)

View file

@ -2064,7 +2064,22 @@ proc getWrappableIndent(p: RstParser): int =
elif nextIndent >= currentTok(p).col: # may be a definition list [case.2]
result = currentTok(p).col
else:
result = nextIndent # [case.3]
result = nextIndent # allow parsing next lines [case.3]
proc getMdBlockIndent(p: RstParser): int =
## Markdown version of `getWrappableIndent`.
if currentTok(p).kind == tkIndent:
result = currentTok(p).ival
else:
var nextIndent = p.tok[tokenAfterNewline(p)-1].ival
# TODO: Markdown-compliant definition should allow nextIndent == currInd(p):
if nextIndent <= currInd(p): # parse only this line
result = currentTok(p).col
else:
result = nextIndent # allow parsing next lines [case.3]
template isRst(p: RstParser): bool = roPreferMarkdown notin p.s.options
template isMd(p: RstParser): bool = roPreferMarkdown in p.s.options
proc parseField(p: var RstParser): PRstNode =
## Returns a parsed rnField node.
@ -2309,6 +2324,39 @@ proc isDefList(p: RstParser): bool =
p.tok[j].kind in {tkWord, tkOther, tkPunct} and
p.tok[j - 2].symbol != "::"
proc `$`(t: Token): string = # for debugging only
result = "(" & $t.kind & " line=" & $t.line & " col=" & $t.col
if t.kind == tkIndent: result = result & " ival=" & $t.ival & ")"
else: result = result & " symbol=" & t.symbol & ")"
proc skipNewlines(p: RstParser, j: int): int =
result = j
while p.tok[result].kind != tkEof and p.tok[result].kind == tkIndent:
inc result # skip blank lines
proc skipNewlines(p: var RstParser) =
p.idx = skipNewlines(p, p.idx)
const maxMdRelInd = 3 ## In Markdown: maximum indentation that does not yet
## make the indented block a code
proc isMdRelInd(outerInd, nestedInd: int): bool =
result = outerInd <= nestedInd and nestedInd <= outerInd + maxMdRelInd
proc isMdDefBody(p: RstParser, j: int, termCol: int): bool =
let defCol = p.tok[j].col
result = p.tok[j].symbol == ":" and
isMdRelInd(termCol, defCol) and
p.tok[j+1].kind == tkWhite and
p.tok[j+2].kind in {tkWord, tkOther, tkPunct}
proc isMdDefListItem(p: RstParser, idx: int): bool =
var j = tokenAfterNewline(p, idx)
j = skipNewlines(p, j)
let termCol = p.tok[j].col
result = isMdRelInd(currInd(p), termCol) and
isMdDefBody(p, j, termCol)
proc isOptionList(p: RstParser): bool =
result = match(p, p.idx, "-w") or match(p, p.idx, "--w") or
match(p, p.idx, "/w") or match(p, p.idx, "//w")
@ -2381,8 +2429,10 @@ proc whichSection(p: RstParser): RstNodeKind =
result = rnEnumList
elif isOptionList(p):
result = rnOptionList
elif isDefList(p):
elif isRst(p) and isDefList(p):
result = rnDefList
elif isMd(p) and isMdDefListItem(p, p.idx):
result = rnMdDefList
else:
result = rnParagraph
of tkWord, tkOther, tkWhite:
@ -2391,7 +2441,9 @@ proc whichSection(p: RstParser): RstNodeKind =
if isAdornmentHeadline(p, tokIdx): result = rnHeadline
else: result = rnParagraph
elif match(p, p.idx, "e) ") or match(p, p.idx, "e. "): result = rnEnumList
elif isDefList(p): result = rnDefList
elif isRst(p) and isDefList(p): result = rnDefList
elif isMd(p) and isMdDefListItem(p, p.idx):
result = rnMdDefList
else: result = rnParagraph
else: result = rnLeaf
@ -2921,6 +2973,36 @@ proc parseOptionList(p: var RstParser): PRstNode =
if currentTok(p).kind != tkEof: dec p.idx # back to tkIndent
break
proc parseMdDefinitionList(p: var RstParser): PRstNode =
## Parses (Pandoc/kramdown/PHPextra) Mardkown definition lists.
result = newRstNodeA(p, rnMdDefList)
let termCol = currentTok(p).col
while true:
var item = newRstNode(rnDefItem)
var term = newRstNode(rnDefName)
parseLine(p, term)
skipNewlines(p)
inc p.idx, 2 # skip ":" and space
item.add(term)
while true:
var def = newRstNode(rnDefBody)
let indent = getMdBlockIndent(p)
pushInd(p, indent)
parseSection(p, def)
popInd(p)
item.add(def)
let j = skipNewlines(p, p.idx)
if isMdDefBody(p, j, termCol): # parse next definition body
p.idx = j + 2 # skip ":" and space
else:
break
result.add(item)
let j = skipNewlines(p, p.idx)
if p.tok[j].col == termCol and isMdDefListItem(p, j):
p.idx = j # parse next item
else:
break
proc parseDefinitionList(p: var RstParser): PRstNode =
result = nil
var j = tokenAfterNewline(p) - 1
@ -3094,6 +3176,7 @@ proc parseSection(p: var RstParser, result: PRstNode) =
of rnLeaf: rstMessage(p, meNewSectionExpected, "(syntax error)")
of rnParagraph: discard
of rnDefList: a = parseDefinitionList(p)
of rnMdDefList: a = parseMdDefinitionList(p)
of rnFieldList:
if p.idx > 0: dec p.idx
a = parseFields(p)
@ -3120,9 +3203,6 @@ proc parseSectionWrapper(p: var RstParser): PRstNode =
while result.kind == rnInner and result.len == 1:
result = result.sons[0]
proc `$`(t: Token): string =
result = $t.kind & ' ' & t.symbol
proc parseDoc(p: var RstParser): PRstNode =
result = parseSectionWrapper(p)
if currentTok(p).kind != tkEof:

View file

@ -27,7 +27,7 @@ type
rnBulletItem, # a bullet item
rnEnumList, # an enumerated list
rnEnumItem, # an enumerated item
rnDefList, # a definition list
rnDefList, rnMdDefList, # a definition list (RST/Markdown)
rnDefItem, # an item of a definition list consisting of ...
rnDefName, # ... a name part ...
rnDefBody, # ... and a body part ...

View file

@ -1212,7 +1212,7 @@ proc renderRstToOut(d: PDoc, n: PRstNode, result: var string) =
of rnBulletItem, rnEnumItem:
renderAux(d, n, "<li$2>$1</li>\n", "\\item $2$1\n", result)
of rnEnumList: renderEnumList(d, n, result)
of rnDefList:
of rnDefList, rnMdDefList:
renderAux(d, n, "<dl$2 class=\"docutils\">$1</dl>\n",
"\\begin{description}\n$2\n$1\\end{description}\n", result)
of rnDefItem: renderAux(d, n, result)

View file

@ -825,10 +825,10 @@ proc rotateLeft*[T](arg: var openArray[T]; slice: HSlice[int, int];
## If an invalid range (`HSlice`) is passed, it raises `IndexDefect`.
##
## `slice`
## The indices of the element range that should be rotated.
## : The indices of the element range that should be rotated.
##
## `dist`
## The distance in amount of elements that the data should be rotated.
## : The distance in amount of elements that the data should be rotated.
## Can be negative, can be any number.
##
## **See also:**
@ -876,10 +876,10 @@ proc rotatedLeft*[T](arg: openArray[T]; slice: HSlice[int, int],
## If an invalid range (`HSlice`) is passed, it raises `IndexDefect`.
##
## `slice`
## The indices of the element range that should be rotated.
## : The indices of the element range that should be rotated.
##
## `dist`
## The distance in amount of elements that the data should be rotated.
## : The distance in amount of elements that the data should be rotated.
## Can be negative, can be any number.
##
## **See also:**

View file

@ -180,15 +180,15 @@ The square brackets `[]` indicate an optional element.
The optional `align` flag can be one of the following:
`<`
Forces the field to be left-aligned within the available
: Forces the field to be left-aligned within the available
space. (This is the default for strings.)
`>`
Forces the field to be right-aligned within the available space.
: Forces the field to be right-aligned within the available space.
(This is the default for numbers.)
`^`
Forces the field to be centered within the available space.
: Forces the field to be centered within the available space.
Note that unless a minimum field width is defined, the field width
will always be the same size as the data to fill it, so that the alignment

View file

@ -78,19 +78,23 @@ proc floorDivPow2(x: int32; n: int32): int32 {.inline.} =
return x shr n
## Returns floor(log_10(2^e))
## ```c
## static inline int32_t FloorLog10Pow2(int32_t e)
## {
## SF_ASSERT(e >= -1500);
## SF_ASSERT(e <= 1500);
## return FloorDivPow2(e * 1262611, 22);
## }
## ```
## Returns floor(log_10(3/4 2^e))
## ```c
## static inline int32_t FloorLog10ThreeQuartersPow2(int32_t e)
## {
## SF_ASSERT(e >= -1500);
## SF_ASSERT(e <= 1500);
## return FloorDivPow2(e * 1262611 - 524031, 22);
## }
## ```
## Returns floor(log_2(10^e))
proc floorLog2Pow10(e: int32): int32 {.inline.} =