Merge branch 'newparser' of github.com:Araq/Nimrod into newparser

This commit is contained in:
Araq 2013-04-22 19:30:03 +02:00
commit 8dc9ad7ce3
12 changed files with 948 additions and 927 deletions

View file

@ -237,7 +237,7 @@ proc genItem(d: PDoc, n, nameNode: PNode, k: TSymKind) =
of tkSymbol: of tkSymbol:
dispA(result, "<span class=\"Identifier\">$1</span>", dispA(result, "<span class=\"Identifier\">$1</span>",
"\\spanIdentifier{$1}", [toRope(esc(d.target, literal))]) "\\spanIdentifier{$1}", [toRope(esc(d.target, literal))])
of tkInd, tkSad, tkDed, tkSpaces, tkInvalid: of tkSpaces, tkInvalid:
app(result, literal) app(result, literal)
of tkParLe, tkParRi, tkBracketLe, tkBracketRi, tkCurlyLe, tkCurlyRi, of tkParLe, tkParRi, tkBracketLe, tkBracketRi, tkCurlyLe, tkCurlyRi,
tkBracketDotLe, tkBracketDotRi, tkCurlyDotLe, tkCurlyDotRi, tkParDotLe, tkBracketDotLe, tkBracketDotRi, tkCurlyDotLe, tkCurlyDotRi, tkParDotLe,

View file

@ -1,7 +1,7 @@
# #
# #
# The Nimrod Compiler # The Nimrod Compiler
# (c) Copyright 2012 Andreas Rumpf # (c) Copyright 2013 Andreas Rumpf
# #
# See the file "copying.txt", included in this # See the file "copying.txt", included in this
# distribution, for details about the copyright. # distribution, for details about the copyright.
@ -58,8 +58,7 @@ type
tkParDotLe, tkParDotRi, # (. and .) tkParDotLe, tkParDotRi, # (. and .)
tkComma, tkSemiColon, tkComma, tkSemiColon,
tkColon, tkColonColon, tkEquals, tkDot, tkDotDot, tkColon, tkColonColon, tkEquals, tkDot, tkDotDot,
tkOpr, tkComment, tkAccent, tkInd, tkSad, tkOpr, tkComment, tkAccent,
tkDed, # pseudo token types used by the source renderers:
tkSpaces, tkInfixOpr, tkPrefixOpr, tkPostfixOpr, tkSpaces, tkInfixOpr, tkPrefixOpr, tkPostfixOpr,
TTokTypes* = set[TTokType] TTokTypes* = set[TTokType]
@ -91,8 +90,8 @@ const
")", "[", "]", "{", "}", "[.", ".]", "{.", ".}", "(.", ".)", ")", "[", "]", "{", "}", "[.", ".]", "{.", ".}", "(.", ".)",
",", ";", ",", ";",
":", "::", "=", ".", "..", ":", "::", "=", ".", "..",
"tkOpr", "tkComment", "`", "[new indentation]", "tkOpr", "tkComment", "`",
"[same indentation]", "[dedentation]", "tkSpaces", "tkInfixOpr", "tkSpaces", "tkInfixOpr",
"tkPrefixOpr", "tkPostfixOpr"] "tkPrefixOpr", "tkPostfixOpr"]
type type
@ -102,7 +101,8 @@ type
base2, base8, base16 base2, base8, base16
TToken* = object # a Nimrod token TToken* = object # a Nimrod token
tokType*: TTokType # the type of the token tokType*: TTokType # the type of the token
indent*: int # the indentation; only valid if tokType = tkIndent indent*: int # the indentation; != -1 if the token has been
# preceeded with indentation
ident*: PIdent # the parsed identifier ident*: PIdent # the parsed identifier
iNumber*: BiggestInt # the parsed integer literal iNumber*: BiggestInt # the parsed integer literal
fNumber*: BiggestFloat # the parsed floating point literal fNumber*: BiggestFloat # the parsed floating point literal
@ -113,8 +113,6 @@ type
TLexer* = object of TBaseLexer TLexer* = object of TBaseLexer
fileIdx*: int32 fileIdx*: int32
indentStack*: seq[int] # the indentation stack
dedent*: int # counter for DED token generation
indentAhead*: int # if > 0 an indendation has already been read indentAhead*: int # if > 0 an indendation has already been read
# this is needed because scanning comments # this is needed because scanning comments
# needs so much look-ahead # needs so much look-ahead
@ -122,9 +120,6 @@ type
var gLinesCompiled*: int # all lines that have been compiled var gLinesCompiled*: int # all lines that have been compiled
proc pushInd*(L: var TLexer, indent: int)
proc popInd*(L: var TLexer)
proc isKeyword*(kind: TTokType): bool proc isKeyword*(kind: TTokType): bool
proc openLexer*(lex: var TLexer, fileidx: int32, inputstream: PLLStream) proc openLexer*(lex: var TLexer, fileidx: int32, inputstream: PLLStream)
proc rawGetTok*(L: var TLexer, tok: var TToken) proc rawGetTok*(L: var TLexer, tok: var TToken)
@ -154,29 +149,12 @@ proc isNimrodIdentifier*(s: string): bool =
inc(i) inc(i)
result = true result = true
proc pushInd(L: var TLexer, indent: int) =
var length = len(L.indentStack)
setlen(L.indentStack, length + 1)
if (indent > L.indentStack[length - 1]):
L.indentstack[length] = indent
else:
InternalError("pushInd")
proc popInd(L: var TLexer) =
var length = len(L.indentStack)
setlen(L.indentStack, length - 1)
proc findIdent(L: TLexer, indent: int): bool =
for i in countdown(len(L.indentStack) - 1, 0):
if L.indentStack[i] == indent:
return true
proc tokToStr*(tok: TToken): string = proc tokToStr*(tok: TToken): string =
case tok.tokType case tok.tokType
of tkIntLit..tkInt64Lit: result = $tok.iNumber of tkIntLit..tkInt64Lit: result = $tok.iNumber
of tkFloatLit..tkFloat64Lit: result = $tok.fNumber of tkFloatLit..tkFloat64Lit: result = $tok.fNumber
of tkInvalid, tkStrLit..tkCharLit, tkComment: result = tok.literal of tkInvalid, tkStrLit..tkCharLit, tkComment: result = tok.literal
of tkParLe..tkColon, tkEof, tkInd, tkSad, tkDed, tkAccent: of tkParLe..tkColon, tkEof, tkAccent:
result = tokTypeToStr[tok.tokType] result = tokTypeToStr[tok.tokType]
else: else:
if tok.ident != nil: if tok.ident != nil:
@ -216,7 +194,6 @@ proc fillToken(L: var TToken) =
proc openLexer(lex: var TLexer, fileIdx: int32, inputstream: PLLStream) = proc openLexer(lex: var TLexer, fileIdx: int32, inputstream: PLLStream) =
openBaseLexer(lex, inputstream) openBaseLexer(lex, inputstream)
lex.indentStack = @[0]
lex.fileIdx = fileIdx lex.fileIdx = fileIdx
lex.indentAhead = - 1 lex.indentAhead = - 1
inc(lex.Linenumber, inputstream.lineOffset) inc(lex.Linenumber, inputstream.lineOffset)
@ -434,9 +411,10 @@ proc GetNumber(L: var TLexer): TToken =
result.tokType = tkInt64Lit result.tokType = tkInt64Lit
elif result.tokType != tkInt64Lit: elif result.tokType != tkInt64Lit:
lexMessage(L, errInvalidNumber, result.literal) lexMessage(L, errInvalidNumber, result.literal)
except EInvalidValue: lexMessage(L, errInvalidNumber, result.literal) except EInvalidValue:
except EOverflow: lexMessage(L, errNumberOutOfRange, result.literal) lexMessage(L, errInvalidNumber, result.literal)
except EOutOfRange: lexMessage(L, errNumberOutOfRange, result.literal) except EOverflow, EOutOfRange:
lexMessage(L, errNumberOutOfRange, result.literal)
L.bufpos = endpos L.bufpos = endpos
proc handleHexChar(L: var TLexer, xi: var int) = proc handleHexChar(L: var TLexer, xi: var int) =
@ -651,24 +629,6 @@ proc getOperator(L: var TLexer, tok: var TToken) =
Inc(pos) Inc(pos)
endOperator(L, tok, pos, h) endOperator(L, tok, pos, h)
proc handleIndentation(L: var TLexer, tok: var TToken, indent: int) =
tok.indent = indent
var i = high(L.indentStack)
if indent > L.indentStack[i]:
tok.tokType = tkInd
elif indent == L.indentStack[i]:
tok.tokType = tkSad
else:
# check we have the indentation somewhere in the stack:
while (i >= 0) and (indent != L.indentStack[i]):
dec(i)
inc(L.dedent)
dec(L.dedent)
tok.tokType = tkDed
if i < 0:
tok.tokType = tkSad # for the parser it is better as SAD
lexMessage(L, errInvalidIndentation)
proc scanComment(L: var TLexer, tok: var TToken) = proc scanComment(L: var TLexer, tok: var TToken) =
var pos = L.bufpos var pos = L.bufpos
var buf = L.buf var buf = L.buf
@ -705,7 +665,6 @@ proc scanComment(L: var TLexer, tok: var TToken) =
else: else:
if buf[pos] > ' ': if buf[pos] > ' ':
L.indentAhead = indent L.indentAhead = indent
inc(L.dedent)
break break
L.bufpos = pos L.bufpos = pos
@ -718,7 +677,7 @@ proc skip(L: var TLexer, tok: var TToken) =
Inc(pos) Inc(pos)
of Tabulator: of Tabulator:
lexMessagePos(L, errTabulatorsAreNotAllowed, pos) lexMessagePos(L, errTabulatorsAreNotAllowed, pos)
inc(pos) # BUGFIX inc(pos)
of CR, LF: of CR, LF:
pos = HandleCRLF(L, pos) pos = HandleCRLF(L, pos)
buf = L.buf buf = L.buf
@ -726,8 +685,8 @@ proc skip(L: var TLexer, tok: var TToken) =
while buf[pos] == ' ': while buf[pos] == ' ':
Inc(pos) Inc(pos)
Inc(indent) Inc(indent)
if (buf[pos] > ' '): if buf[pos] > ' ':
handleIndentation(L, tok, indent) tok.indent = indent
break break
else: else:
break # EndOfFile also leaves the loop break # EndOfFile also leaves the loop
@ -735,22 +694,15 @@ proc skip(L: var TLexer, tok: var TToken) =
proc rawGetTok(L: var TLexer, tok: var TToken) = proc rawGetTok(L: var TLexer, tok: var TToken) =
fillToken(tok) fillToken(tok)
if L.dedent > 0: if L.indentAhead >= 0:
dec(L.dedent) tok.indent = L.indentAhead
if L.indentAhead >= 0: L.indentAhead = -1
handleIndentation(L, tok, L.indentAhead) else:
L.indentAhead = - 1 tok.indent = -1
else:
tok.tokType = tkDed
return
skip(L, tok) skip(L, tok)
# got an documentation comment or tkIndent, return that:
if tok.toktype != tkInvalid: return
var c = L.buf[L.bufpos] var c = L.buf[L.bufpos]
if c in SymStartChars - {'r', 'R', 'l'}: if c in SymStartChars - {'r', 'R', 'l'}:
getSymbol(L, tok) getSymbol(L, tok)
elif c in {'0'..'9'}:
tok = getNumber(L)
else: else:
case c case c
of '#': of '#':
@ -769,7 +721,7 @@ proc rawGetTok(L: var TLexer, tok: var TToken) =
of 'l': of 'l':
# if we parsed exactly one character and its a small L (l), this # if we parsed exactly one character and its a small L (l), this
# is treated as a warning because it may be confused with the number 1 # is treated as a warning because it may be confused with the number 1
if not (L.buf[L.bufpos + 1] in (SymChars + {'_'})): if L.buf[L.bufpos+1] notin (SymChars + {'_'}):
lexMessage(L, warnSmallLshouldNotBeUsed) lexMessage(L, warnSmallLshouldNotBeUsed)
getSymbol(L, tok) getSymbol(L, tok)
of 'r', 'R': of 'r', 'R':
@ -780,7 +732,7 @@ proc rawGetTok(L: var TLexer, tok: var TToken) =
getSymbol(L, tok) getSymbol(L, tok)
of '(': of '(':
Inc(L.bufpos) Inc(L.bufpos)
if (L.buf[L.bufPos] == '.') and (L.buf[L.bufPos + 1] != '.'): if L.buf[L.bufPos] == '.' and L.buf[L.bufPos+1] != '.':
tok.toktype = tkParDotLe tok.toktype = tkParDotLe
Inc(L.bufpos) Inc(L.bufpos)
else: else:
@ -790,7 +742,7 @@ proc rawGetTok(L: var TLexer, tok: var TToken) =
Inc(L.bufpos) Inc(L.bufpos)
of '[': of '[':
Inc(L.bufpos) Inc(L.bufpos)
if (L.buf[L.bufPos] == '.') and (L.buf[L.bufPos + 1] != '.'): if L.buf[L.bufPos] == '.' and L.buf[L.bufPos+1] != '.':
tok.toktype = tkBracketDotLe tok.toktype = tkBracketDotLe
Inc(L.bufpos) Inc(L.bufpos)
else: else:
@ -799,20 +751,20 @@ proc rawGetTok(L: var TLexer, tok: var TToken) =
tok.toktype = tkBracketRi tok.toktype = tkBracketRi
Inc(L.bufpos) Inc(L.bufpos)
of '.': of '.':
if L.buf[L.bufPos + 1] == ']': if L.buf[L.bufPos+1] == ']':
tok.tokType = tkBracketDotRi tok.tokType = tkBracketDotRi
Inc(L.bufpos, 2) Inc(L.bufpos, 2)
elif L.buf[L.bufPos + 1] == '}': elif L.buf[L.bufPos+1] == '}':
tok.tokType = tkCurlyDotRi tok.tokType = tkCurlyDotRi
Inc(L.bufpos, 2) Inc(L.bufpos, 2)
elif L.buf[L.bufPos + 1] == ')': elif L.buf[L.bufPos+1] == ')':
tok.tokType = tkParDotRi tok.tokType = tkParDotRi
Inc(L.bufpos, 2) Inc(L.bufpos, 2)
else: else:
getOperator(L, tok) getOperator(L, tok)
of '{': of '{':
Inc(L.bufpos) Inc(L.bufpos)
if (L.buf[L.bufPos] == '.') and (L.buf[L.bufPos+1] != '.'): if L.buf[L.bufPos] == '.' and L.buf[L.bufPos+1] != '.':
tok.toktype = tkCurlyDotLe tok.toktype = tkCurlyDotLe
Inc(L.bufpos) Inc(L.bufpos)
else: else:
@ -838,13 +790,16 @@ proc rawGetTok(L: var TLexer, tok: var TToken) =
tok.tokType = tkCharLit tok.tokType = tkCharLit
getCharacter(L, tok) getCharacter(L, tok)
tok.tokType = tkCharLit tok.tokType = tkCharLit
of '0'..'9':
tok = getNumber(L)
else: else:
if c in OpChars: if c in OpChars:
getOperator(L, tok) getOperator(L, tok)
elif c == lexbase.EndOfFile: elif c == lexbase.EndOfFile:
tok.toktype = tkEof tok.toktype = tkEof
tok.indent = 0
else: else:
tok.literal = c & "" tok.literal = $c
tok.tokType = tkInvalid tok.tokType = tkInvalid
lexMessage(L, errInvalidToken, c & " (\\" & $(ord(c)) & ')') lexMessage(L, errInvalidToken, c & " (\\" & $(ord(c)) & ')')
Inc(L.bufpos) Inc(L.bufpos)

View file

@ -711,7 +711,7 @@ var
proc writeSurroundingSrc(info: TLineInfo) = proc writeSurroundingSrc(info: TLineInfo) =
const indent = " " const indent = " "
MsgWriteln(indent & info.sourceLine.data) MsgWriteln(indent & info.sourceLine.ropeToStr)
MsgWriteln(indent & repeatChar(info.col, ' ') & '^') MsgWriteln(indent & repeatChar(info.col, ' ') & '^')
proc liMessage(info: TLineInfo, msg: TMsgKind, arg: string, proc liMessage(info: TLineInfo, msg: TMsgKind, arg: string,
@ -786,8 +786,9 @@ proc sourceLine*(i: TLineInfo): PRope =
for line in lines(i.toFullPath): for line in lines(i.toFullPath):
addSourceLine i.fileIndex, line.string addSourceLine i.fileIndex, line.string
InternalAssert i.fileIndex < fileInfos.len and InternalAssert i.fileIndex < fileInfos.len
i.line <= fileInfos[i.fileIndex].lines.len # can happen if the error points to EOF:
if i.line > fileInfos[i.fileIndex].lines.len: return nil
result = fileInfos[i.fileIndex].lines[i.line-1] result = fileInfos[i.fileIndex].lines[i.line-1]

View file

@ -19,7 +19,7 @@ import
proc ppGetTok(L: var TLexer, tok: var TToken) = proc ppGetTok(L: var TLexer, tok: var TToken) =
# simple filter # simple filter
rawGetTok(L, tok) rawGetTok(L, tok)
while tok.tokType in {tkInd, tkSad, tkDed, tkComment}: rawGetTok(L, tok) while tok.tokType in {tkComment}: rawGetTok(L, tok)
proc parseExpr(L: var TLexer, tok: var TToken): bool proc parseExpr(L: var TLexer, tok: var TToken): bool
proc parseAtom(L: var TLexer, tok: var TToken): bool = proc parseAtom(L: var TLexer, tok: var TToken): bool =

File diff suppressed because it is too large Load diff

View file

@ -81,13 +81,13 @@ proc addTok(g: var TSrcGen, kind: TTokType, s: string) =
proc addPendingNL(g: var TSrcGen) = proc addPendingNL(g: var TSrcGen) =
if g.pendingNL >= 0: if g.pendingNL >= 0:
addTok(g, tkInd, "\n" & repeatChar(g.pendingNL)) addTok(g, tkSpaces, "\n" & repeatChar(g.pendingNL))
g.lineLen = g.pendingNL g.lineLen = g.pendingNL
g.pendingNL = - 1 g.pendingNL = - 1
proc putNL(g: var TSrcGen, indent: int) = proc putNL(g: var TSrcGen, indent: int) =
if g.pendingNL >= 0: addPendingNL(g) if g.pendingNL >= 0: addPendingNL(g)
else: addTok(g, tkInd, "\n") else: addTok(g, tkSpaces, "\n")
g.pendingNL = indent g.pendingNL = indent
g.lineLen = indent g.lineLen = indent

View file

@ -58,6 +58,37 @@ proc fitNode(c: PContext, formal: PType, arg: PNode): PNode =
result = copyNode(arg) result = copyNode(arg)
result.typ = formal result.typ = formal
proc commonType*(x, y: PType): PType =
# new type relation that is used for array constructors,
# if expressions, etc.:
if x == nil: return y
var a = skipTypes(x, {tyGenericInst})
var b = skipTypes(y, {tyGenericInst})
result = x
if a.kind in {tyExpr, tyNil}: return y
elif b.kind in {tyExpr, tyNil}: return x
elif b.kind in {tyArray, tyArrayConstr, tySet, tySequence} and
a.kind == b.kind:
# check for seq[empty] vs. seq[int]
let idx = ord(b.kind in {tyArray, tyArrayConstr})
if a.sons[idx].kind == tyEmpty: return y
#elif b.sons[idx].kind == tyEmpty: return x
else:
var k = tyNone
if a.kind in {tyRef, tyPtr}:
k = a.kind
if b.kind != a.kind: return x
a = a.sons[0]
b = b.sons[0]
if a.kind == tyObject and b.kind == tyObject:
result = commonSuperclass(a, b)
# this will trigger an error later:
if result.isNil: return x
if k != tyNone:
let r = result
result = NewType(k, r.owner)
result.addSonSkipIntLit(r)
proc isTopLevel(c: PContext): bool {.inline.} = proc isTopLevel(c: PContext): bool {.inline.} =
result = c.tab.tos <= 2 result = c.tab.tos <= 2

View file

@ -904,6 +904,26 @@ proc inheritanceDiff*(a, b: PType): int =
inc(result) inc(result)
result = high(int) result = high(int)
proc commonSuperclass*(a, b: PType): PType =
# quick check: are they the same?
if sameObjectTypes(a, b): return a
# simple algorithm: we store all ancestors of 'a' in a ID-set and walk 'b'
# up until the ID is found:
assert a.kind == tyObject
assert b.kind == tyObject
var x = a
var ancestors = initIntSet()
while x != nil:
x = skipTypes(x, skipPtrs)
ancestors.incl(x.id)
x = x.sons[0]
var y = b
while y != nil:
y = skipTypes(y, skipPtrs)
if ancestors.contains(y.id): return y
y = y.sons[0]
proc typeAllowedAux(marker: var TIntSet, typ: PType, kind: TSymKind): bool proc typeAllowedAux(marker: var TIntSet, typ: PType, kind: TSymKind): bool
proc typeAllowedNode(marker: var TIntSet, n: PNode, kind: TSymKind): bool = proc typeAllowedNode(marker: var TIntSet, n: PNode, kind: TSymKind): bool =
result = true result = true

View file

@ -1,204 +1,181 @@
module ::= ([COMMENT] [SAD] stmt)* module = stmt ^* (';' / IND{=})
comma = ',' COMMENT?
semicolon = ';' COMMENT?
colon = ':' COMMENT?
colcom = ':' COMMENT?
comma ::= ',' [COMMENT] [IND] operator = OP0 | OP1 | OP2 | OP3 | OP4 | OP5 | OP6 | OP7 | OP8 | OP9
semicolon ::= ';' [COMMENT] [IND] | 'or' | 'xor' | 'and'
| 'is' | 'isnot' | 'in' | 'notin' | 'of'
| 'div' | 'mod' | 'shl' | 'shr' | 'not' | 'addr' | 'static' | '..'
operator ::= OP0 | OP1 | OP2 | OP3 | OP4 | OP5 | OP6 | OP7 | OP8 | OP9 prefixOperator = operator
| 'or' | 'xor' | 'and'
| 'is' | 'isnot' | 'in' | 'notin' | 'of'
| 'div' | 'mod' | 'shl' | 'shr' | 'not' | 'addr' | 'static' | '..'
prefixOperator ::= operator optInd = COMMENT?
optPar = (IND{>} | IND{=})?
optInd ::= [COMMENT] [IND]
optPar ::= [IND] | [SAD]
lowestExpr ::= assignExpr (OP0 optInd assignExpr)*
assignExpr ::= orExpr (OP1 optInd orExpr)*
orExpr ::= andExpr (OP2 optInd andExpr)*
andExpr ::= cmpExpr (OP3 optInd cmpExpr)*
cmpExpr ::= sliceExpr (OP4 optInd sliceExpr)*
sliceExpr ::= ampExpr (OP5 optInd ampExpr)*
ampExpr ::= plusExpr (OP6 optInd plusExpr)*
plusExpr ::= mulExpr (OP7 optInd mulExpr)*
mulExpr ::= dollarExpr (OP8 optInd dollarExpr)*
dollarExpr ::= primary (OP9 optInd primary)*
indexExpr ::= expr
castExpr ::= 'cast' '[' optInd typeDesc optPar ']' '(' optInd expr optPar ')'
symbol ::= '`' (KEYWORD | IDENT | operator | '(' ')' | '[' ']' | '{' '}'
| '=' | literal)+ '`'
| IDENT
primaryPrefix ::= (prefixOperator | 'bind') optInd
primarySuffix ::= '.' optInd symbol [generalizedLit]
| '(' optInd namedExprList optPar ')'
| '[' optInd [indexExpr (comma indexExpr)* [comma]] optPar ']'
| '{' optInd [indexExpr (comma indexExpr)* [comma]] optPar '}'
primary ::= primaryPrefix* (symbol [generalizedLit] |
constructor | castExpr)
primarySuffix*
simpleExpr = assignExpr (OP0 optInd assignExpr)*
assignExpr = orExpr (OP1 optInd orExpr)*
orExpr = andExpr (OP2 optInd andExpr)*
andExpr = cmpExpr (OP3 optInd cmpExpr)*
cmpExpr = sliceExpr (OP4 optInd sliceExpr)*
sliceExpr = ampExpr (OP5 optInd ampExpr)*
ampExpr = plusExpr (OP6 optInd plusExpr)*
plusExpr = mulExpr (OP7 optInd mulExpr)*
mulExpr = dollarExpr (OP8 optInd dollarExpr)*
dollarExpr = primary (OP9 optInd primary)*
symbol = '`' (KEYW|IDENT|operator|'(' ')'|'[' ']'|'{' '}'|'='|literal)+ '`'
| IDENT
indexExpr = expr
indexExprList = indexExpr ^+ comma
exprColonEqExpr = expr (':'|'=' expr)?
exprList = expr ^+ comma
dotExpr = expr '.' optInd ('type' | 'addr' | symbol)
qualifiedIdent = symbol ('.' optInd ('type' | 'addr' | symbol))?
exprColonEqExprList = exprColonEqExpr (comma exprColonEqExpr)* (comma)?
setOrTableConstr = '{' ((exprColonEqExpr comma)* | ':' ) '}'
castExpr = 'cast' '[' optInd typeDesc optPar ']' '(' optInd expr optPar ')'
generalizedLit ::= GENERALIZED_STR_LIT | GENERALIZED_TRIPLESTR_LIT generalizedLit ::= GENERALIZED_STR_LIT | GENERALIZED_TRIPLESTR_LIT
identOrLiteral = generalizedLit | symbol
literal ::= INT_LIT | INT8_LIT | INT16_LIT | INT32_LIT | INT64_LIT | INT_LIT | INT8_LIT | INT16_LIT | INT32_LIT | INT64_LIT
| UINT_LIT | UINT8_LIT | UINT16_LIT | UINT32_LIT | UINT64_LIT | UINT_LIT | UINT8_LIT | UINT16_LIT | UINT32_LIT | UINT64_LIT
| FLOAT_LIT | FLOAT32_LIT | FLOAT64_LIT | FLOAT_LIT | FLOAT32_LIT | FLOAT64_LIT
| STR_LIT | RSTR_LIT | TRIPLESTR_LIT | STR_LIT | RSTR_LIT | TRIPLESTR_LIT
| CHAR_LIT | CHAR_LIT
| NIL | NIL
| tupleConstr | arrayConstr | setOrTableConstr
constructor ::= literal | castExpr
| '[' optInd colonExprList optPar ']' tupleConstr = '(' optInd (exprColonEqExpr comma?)* optPar ')'
| '{' optInd ':' | colonExprList optPar '}' arrayConstr = '[' optInd (exprColonEqExpr comma?)* optPar ']'
| '(' optInd colonExprList optPar ')' primarySuffix = '(' (exprColonEqExpr comma?)* ')' doBlocks?
| doBlocks
colonExpr ::= expr [':' expr] | '.' optInd ('type' | 'addr' | symbol) generalizedLit?
colonExprList ::= [colonExpr (comma colonExpr)* [comma]] | '[' optInd indexExprList optPar ']'
| '{' optInd indexExprList optPar '}'
namedExpr ::= expr ['=' expr] condExpr = expr colcom expr optInd
namedExprList ::= [namedExpr (comma namedExpr)* [comma]] ('elif' expr colcom expr optInd)*
'else' colcom expr
exprOrType ::= lowestExpr ifExpr = 'if' condExpr
| 'if' expr ':' expr ('elif' expr ':' expr)* 'else' ':' expr whenExpr = 'when' condExpr
| 'var' exprOrType pragma = '{.' optInd (exprColonExpr comma?)* optPar ('.}' | '}')
| 'ref' exprOrType identVis = symbol opr? # postfix position
| 'ptr' exprOrType identWithPragma = identVis pragma?
| 'type' exprOrType declColonEquals = identWithPragma (comma identWithPragma)* comma?
| 'tuple' tupleDesc (':' optInd typeDesc)? ('=' optInd expr)?
identColonEquals = ident (comma ident)* comma?
expr ::= exprOrType (':' optInd typeDesc)? ('=' optInd expr)?)
| 'proc' paramList [pragma] ['=' stmt] inlTupleDecl = 'tuple'
| 'iterator' paramList [pragma] ['=' stmt] [' optInd (identColonEquals (comma/semicolon)?)* optPar ']'
extTupleDecl = 'tuple'
exprList ::= [expr (comma expr)* [comma]] COMMENT? (IND{>} identColonEquals (IND{=} identColonEquals)*)?
paramList = '(' identColonEquals ^* (comma/semicolon) ')'
paramListArrow = paramList? ('->' optInd typeDesc)?
qualifiedIdent ::= symbol ['.' symbol] paramListColon = paramList? (':' optInd typeDesc)?
doBlock = 'do' paramListArrow pragmas? colcom stmt
typeDesc ::= (exprOrType doBlocks = doBlock ^* IND{=}
| 'proc' paramList [pragma] procExpr = 'proc' paramListColon pragmas? ('=' COMMENT? stmt)?
| 'iterator' paramList [pragma] ) expr = (ifExpr
['not' expr] # for now only 'not nil' suffix is supported | whenExpr
| caseExpr)
macroStmt ::= ':' [stmt] ('of' [exprList] ':' stmt / simpleExpr
|'elif' expr ':' stmt typeKeyw = 'var' | 'ref' | 'ptr' | 'shared' | 'type' | 'tuple'
|'except' exceptList ':' stmt )* | 'proc' | 'iterator' | 'distinct' | 'object' | 'enum'
['else' ':' stmt] primary = typeKeyw typeDescK
/ prefixOperator* identOrLiteral primarySuffix*
pragmaBlock ::= pragma [':' stmt] / 'addr' primary
/ 'static' primary
simpleStmt ::= returnStmt / 'bind' primary
| yieldStmt typeDesc = simpleExpr
| discardStmt typeDefAux = simpleExpr
| raiseStmt exprStmt = simpleExpr
| breakStmt (( '=' optInd expr )
| continueStmt / ( expr ^+ comma
| pragmaBlock doBlocks
| importStmt / ':' stmt? ( IND{=} 'of' exprList ':' stmt
| fromStmt | IND{=} 'elif' expr ':' stmt
| includeStmt | IND{=} 'except' exprList ':' stmt
| exprStmt | IND{=} 'else' ':' stmt )*
complexStmt ::= ifStmt | whileStmt | caseStmt | tryStmt | forStmt ))?
| blockStmt | staticStmt | asmStmt importStmt = 'import' optInd expr
| procDecl | iteratorDecl | macroDecl | templateDecl | methodDecl ((comma expr)*
| constSection | letSection | varSection / 'except' optInd (expr ^+ comma))
| typeSection | whenStmt | bindStmt includeStmt = 'include' optInd expr ^+ comma
fromStmt = 'from' expr 'import' optInd expr (comma expr)*
indPush ::= IND # and push indentation onto the stack returnStmt = 'return' optInd expr?
indPop ::= # pop indentation from the stack raiseStmt = 'raise' optInd expr?
yieldStmt = 'yield' optInd expr?
stmt ::= simpleStmt [SAD] discardStmt = 'discard' optInd expr?
| indPush (complexStmt | simpleStmt) breakStmt = 'break' optInd expr?
([SAD] (complexStmt | simpleStmt))* continueStmt = 'break' optInd expr?
DED indPop condStmt = expr colcom stmt COMMENT?
(IND{=} 'elif' expr colcom stmt)*
exprStmt ::= lowestExpr ['=' expr | [expr (comma expr)*] [macroStmt]] (IND{=} 'else' colcom stmt)?
returnStmt ::= 'return' [expr] ifStmt = 'if' condStmt
yieldStmt ::= 'yield' expr whenStmt = 'when' condStmt
discardStmt ::= 'discard' expr whileStmt = 'while' expr colcom stmt
raiseStmt ::= 'raise' [expr] ofBranch = 'of' exprList colcom stmt
breakStmt ::= 'break' [symbol] ofBranches = ofBranch (IND{=} ofBranch)*
continueStmt ::= 'continue' (IND{=} 'elif' expr colcom stmt)*
ifStmt ::= 'if' expr ':' stmt ('elif' expr ':' stmt)* ['else' ':' stmt] (IND{=} 'else' colcom stmt)?
whenStmt ::= 'when' expr ':' stmt ('elif' expr ':' stmt)* ['else' ':' stmt] caseStmt = 'case' expr ':'? COMMENT?
caseStmt ::= 'case' expr [':'] ('of' exprList ':' stmt)* (IND{>} ofBranches DED
('elif' expr ':' stmt)* | IND{=} ofBranches)
['else' ':' stmt] tryStmt = 'try' colcom stmt &(IND{=}? 'except'|'finally')
whileStmt ::= 'while' expr ':' stmt (IND{=}? 'except' exprList colcom stmt)*
forStmt ::= 'for' symbol (comma symbol)* 'in' expr ':' stmt (IND{=}? 'finally' colcom stmt)?
exceptList ::= [qualifiedIdent (comma qualifiedIdent)*] exceptBlock = 'except' colcom stmt
forStmt = 'for' symbol (comma symbol)* 'in' expr colcom stmt
tryStmt ::= 'try' ':' stmt blockStmt = 'block' symbol? colcom stmt
('except' exceptList ':' stmt)* staticStmt = 'static' colcom stmt
['finally' ':' stmt] asmStmt = 'asm' pragma? (STR_LIT | RSTR_LIT | TRIPLE_STR_LIT)
asmStmt ::= 'asm' [pragma] (STR_LIT | RSTR_LIT | TRIPLESTR_LIT) genericParam = symbol (comma symbol)* (colon expr)? ('=' optInd expr)?
blockStmt ::= 'block' [symbol] ':' stmt genericParamList = '[' optInd
staticStmt ::= 'static' ':' stmt genericParam ^* (comma/semicolon) optPar ']'
filename ::= symbol | STR_LIT | RSTR_LIT | TRIPLESTR_LIT pattern = '{' stmt '}'
importStmt ::= 'import' filename (comma filename)* indAndComment = (IND{>} COMMENT)? | COMMENT?
includeStmt ::= 'include' filename (comma filename)* routine = optInd identVis pattern? genericParamList?
bindStmt ::= 'bind' qualifiedIdent (comma qualifiedIdent)* paramListColon pragma? ('=' COMMENT? stmt)? indAndComment
fromStmt ::= 'from' filename 'import' symbol (comma symbol)* commentStmt = COMMENT
section(p) = COMMENT? p / (IND{>} (p / COMMENT)^+IND{=} DED)
pragma ::= '{.' optInd (colonExpr [comma])* optPar ('.}' | '}') constant = identWithPragma (colon typedesc)? '=' optInd expr indAndComment
enum = 'enum' optInd (symbol optInd ('=' optInd expr COMMENT?)? comma?)+
param ::= symbol (comma symbol)* (':' typeDesc ['=' expr] | '=' expr) objectWhen = 'when' expr colcom objectPart COMMENT?
paramList ::= ['(' [param (comma|semicolon param)*] optPar ')'] [':' typeDesc] ('elif' expr colcom objectPart COMMENT?)*
('else' colcom objectPart COMMENT?)?
genericConstraint ::= 'object' | 'tuple' | 'enum' | 'proc' | 'ref' | 'ptr' objectBranch = 'of' exprList colcom objectPart
| 'var' | 'distinct' | 'iterator' | primary objectBranches = objectBranch (IND{=} objectBranch)*
genericConstraints ::= genericConstraint ( '|' optInd genericConstraint )* (IND{=} 'elif' expr colcom objectPart)*
(IND{=} 'else' colcom objectPart)?
genericParam ::= symbol [':' genericConstraints] ['=' expr] objectCase = 'case' identWithPragma ':' typeDesc ':'? COMMENT?
genericParams ::= '[' genericParam (comma|semicolon genericParam)* optPar ']' (IND{>} objectBranches DED
| IND{=} objectBranches)
objectPart = IND{>} objectPart^+IND{=} DED
routineDecl := symbol ['*'] [genericParams] paramList [pragma] ['=' stmt] / objectWhen / objectCase / 'nil' / declColonEquals
procDecl ::= 'proc' routineDecl object = 'object' pragma? ('of' typeDesc)? COMMENT? objectPart
macroDecl ::= 'macro' routineDecl distinct = 'distinct' optInd typeDesc
iteratorDecl ::= 'iterator' routineDecl typeDef = identWithPragma genericParamList? '=' optInd typeDefAux
templateDecl ::= 'template' routineDecl indAndComment?
methodDecl ::= 'method' routineDecl varTuple = '(' optInd identWithPragma ^+ comma optPar ')' '=' optInd expr
variable = (varTuple / identColonEquals) indAndComment
colonAndEquals ::= [':' typeDesc] '=' expr bindStmt = 'bind' optInd qualifiedIdent ^+ comma
mixinStmt = 'mixin' optInd qualifiedIdent ^+ comma
constDecl ::= symbol ['*'] [pragma] colonAndEquals [COMMENT | IND COMMENT] pragmaStmt = pragma (':' COMMENT? stmt)?
| COMMENT simpleStmt = ((returnStmt | raiseStmt | yieldStmt | discardStmt | breakStmt
constSection ::= 'const' indPush constDecl (SAD constDecl)* DED indPop | continueStmt | pragmaStmt | importStmt | exportStmt | fromStmt
letSection ::= 'let' indPush constDecl (SAD constDecl)* DED indPop | includeStmt | commentStmt) / exprStmt) COMMENT?
complexOrSimpleStmt = (ifStmt | whenStmt | whileStmt
typeDef ::= typeDesc | objectDef | enumDef | 'distinct' typeDesc | tryStmt | finallyStmt | exceptStmt | forStmt
| blockStmt | staticStmt | asmStmt
objectField ::= symbol ['*'] [pragma] | 'proc' routine
objectIdentPart ::= objectField (comma objectField)* ':' typeDesc | 'method' routine
[COMMENT|IND COMMENT] | 'iterator' routine
| 'macro' routine
objectWhen ::= 'when' expr ':' [COMMENT] objectPart | 'template' routine
('elif' expr ':' [COMMENT] objectPart)* | 'converter' routine
['else' ':' [COMMENT] objectPart] | 'type' section(typeDef)
objectCase ::= 'case' expr ':' typeDesc [COMMENT] | 'const' section(constant)
('of' exprList ':' [COMMENT] objectPart)* | ('let' | 'var') section(variable)
['else' ':' [COMMENT] objectPart] | bindStmt | mixinStmt)
/ simpleStmt
objectPart ::= objectWhen | objectCase | objectIdentPart | 'nil' stmt = (IND{>} complexOrSimpleStmt^+(IND{=} / ';') DED)
| indPush objectPart (SAD objectPart)* DED indPop / simpleStmt
tupleDesc ::= '[' optInd [param (comma|semicolon param)*] optPar ']'
objectDef ::= 'object' [pragma] ['of' typeDesc] objectPart
enumField ::= symbol ['=' expr]
enumDef ::= 'enum' (enumField [comma] [COMMENT | IND COMMENT])+
typeDecl ::= COMMENT
| symbol ['*'] [genericParams] ['=' typeDef] [COMMENT | IND COMMENT]
typeSection ::= 'type' indPush typeDecl (SAD typeDecl)* DED indPop
colonOrEquals ::= ':' typeDesc ['=' expr] | '=' expr
varField ::= symbol ['*'] [pragma]
varPart ::= symbol (comma symbol)* colonOrEquals [COMMENT | IND COMMENT]
varSection ::= 'var' (varPart
| indPush (COMMENT|varPart)
(SAD (COMMENT|varPart))* DED indPop)

View file

@ -23,14 +23,25 @@ This document describes the lexis, the syntax, and the semantics of Nimrod.
The language constructs are explained using an extended BNF, in The language constructs are explained using an extended BNF, in
which ``(a)*`` means 0 or more ``a``'s, ``a+`` means 1 or more ``a``'s, and which ``(a)*`` means 0 or more ``a``'s, ``a+`` means 1 or more ``a``'s, and
``(a)?`` means an optional *a*; an alternative spelling for optional parts is ``(a)?`` means an optional *a*. Parentheses may be used to group elements.
``[a]``. The ``|`` symbol is used to mark alternatives
and has the lowest precedence. Parentheses may be used to group elements. The ``|``, ``/`` symbols are used to mark alternatives and have the lowest
precedence. ``/`` is the ordered choice that requires the parser to try the
alternatives in the given order. ``/`` is often used to ensure the grammar
is not ambiguous.
Non-terminals start with a lowercase letter, abstract terminal symbols are in Non-terminals start with a lowercase letter, abstract terminal symbols are in
UPPERCASE. Verbatim terminal symbols (including keywords) are quoted UPPERCASE. Verbatim terminal symbols (including keywords) are quoted
with ``'``. An example:: with ``'``. An example::
ifStmt ::= 'if' expr ':' stmts ('elif' expr ':' stmts)* ['else' stmts] ifStmt = 'if' expr ':' stmts ('elif' expr ':' stmts)* ('else' stmts)?
The binary ``^*`` operator is used as a shorthand for 0 or more occurances
separated by its second argument; likewise ``^+`` means 1 or more
occurances: ``a ^+ b`` is short for ``a (b a)*``
and ``a ^* b`` is short for ``(a (b a)*)?``. Example::
arrayConstructor = '[' expr ^* ',' ']'
Other parts of Nimrod - like scoping rules or runtime semantics are only Other parts of Nimrod - like scoping rules or runtime semantics are only
described in an informal manner for now. described in an informal manner for now.
@ -50,7 +61,7 @@ An `identifier`:idx: is a symbol declared as a name for a variable, type,
procedure, etc. The region of the program over which a declaration applies is procedure, etc. The region of the program over which a declaration applies is
called the `scope`:idx: of the declaration. Scopes can be nested. The meaning called the `scope`:idx: of the declaration. Scopes can be nested. The meaning
of an identifier is determined by the smallest enclosing scope in which the of an identifier is determined by the smallest enclosing scope in which the
identifier is declared. identifier is declared unless overloading resolution rules suggest otherwise.
An expression specifies a computation that produces a value or location. An expression specifies a computation that produces a value or location.
Expressions that produce locations are called `l-values`:idx:. An l-value Expressions that produce locations are called `l-values`:idx:. An l-value
@ -93,28 +104,31 @@ Nimrod's standard grammar describes an `indentation sensitive`:idx: language.
This means that all the control structures are recognized by indentation. This means that all the control structures are recognized by indentation.
Indentation consists only of spaces; tabulators are not allowed. Indentation consists only of spaces; tabulators are not allowed.
The terminals ``IND`` (indentation), ``DED`` (dedentation) and ``SAD`` The indentation handling is implemented as follows: The lexer annotates the
(same indentation) are generated by the scanner, denoting an indentation. following token with the preceeding number of spaces; indentation is not
a separate token. This trick allows parsing of Nimrod with only 1 token of
lookahead.
These terminals are only generated for lines that are not empty. The parser uses a stack of indentation levels: the stack consists of integers
counting the spaces. The indentation information is queried at strategic
places in the parser but ignored otherwise: The pseudo terminal ``IND{>}``
denotes an indentation that consists of more spaces than the entry at the top
of the stack; IND{=} an indentation that has the same number of spaces. ``DED``
is another pseudo terminal that describes the *action* of popping a value
from the stack, ``IND{>}`` then implies to push onto the stack.
The parser and the scanner communicate over a stack which indentation terminal With this notation we can now easily define the core of the grammar: A block of
should be generated: the stack consists of integers counting the spaces. The statements (simplified example)::
stack is initialized with a zero on its top. The scanner reads from the stack:
If the current indentation token consists of more spaces than the entry at the ifStmt = 'if' expr ':' stmt
top of the stack, a ``IND`` token is generated, else if it consists of the same (IND{=} 'elif' expr ':' stmt)*
number of spaces, a ``SAD`` token is generated. If it consists of fewer spaces, (IND{=} 'else' ':' stmt)?
a ``DED`` token is generated for any item on the stack that is greater than the
current. These items are later popped from the stack by the parser. At the end simpleStmt = ifStmt / ...
of the file, a ``DED`` token is generated for each number remaining on the
stack that is larger than zero. stmt = IND{>} stmt ^+ IND{=} DED # list of statements
/ simpleStmt # or a simple statement
Because the grammar contains some optional ``IND`` tokens, the scanner cannot
push new indentation levels. This has to be done by the parser. The symbol
``indPush`` indicates that an ``IND`` token is expected; the current number of
leading spaces is pushed onto the stack by the parser. The symbol ``indPop``
denotes that the parser pops an item from the indentation stack. No token is
consumed by ``indPop``.
Comments Comments
@ -416,8 +430,8 @@ and not the two tokens `{.`:tok:, `.}`:tok:.
Syntax Syntax
====== ======
This section lists Nimrod's standard syntax in ENBF. How the parser receives This section lists Nimrod's standard syntax. How the parser handles
indentation tokens is already described in the `Lexical Analysis`_ section. the indentation is already described in the `Lexical Analysis`_ section.
Nimrod allows user-definable operators. Nimrod allows user-definable operators.
Binary operators have 10 different levels of precedence. Binary operators have 10 different levels of precedence.
@ -1040,7 +1054,7 @@ an ``object`` type or a ``ref object`` type:
.. code-block:: nimrod .. code-block:: nimrod
var student = TStudent(name: "Anton", age: 5, id: 3) var student = TStudent(name: "Anton", age: 5, id: 3)
For a ``ref object`` type ``new`` is invoked implicitly. For a ``ref object`` type ``system.new`` is invoked implicitly.
Object variants Object variants
@ -1701,44 +1715,20 @@ Statements and expressions
========================== ==========================
Nimrod uses the common statement/expression paradigm: `Statements`:idx: do not Nimrod uses the common statement/expression paradigm: `Statements`:idx: do not
produce a value in contrast to expressions. Call expressions are statements. produce a value in contrast to expressions. However, some expressions are
If the called procedure returns a value, it is not a valid statement statements.
as statements do not produce values. To evaluate an expression for
side-effects and throw its value away, one can use the ``discard`` statement.
Statements are separated into `simple statements`:idx: and Statements are separated into `simple statements`:idx: and
`complex statements`:idx:. `complex statements`:idx:.
Simple statements are statements that cannot contain other statements like Simple statements are statements that cannot contain other statements like
assignments, calls or the ``return`` statement; complex statements can assignments, calls or the ``return`` statement; complex statements can
contain other statements. To avoid the `dangling else problem`:idx:, complex contain other statements. To avoid the `dangling else problem`:idx:, complex
statements always have to be intended:: statements always have to be intended. The details can be found in the grammar.
simpleStmt ::= returnStmt
| yieldStmt
| discardStmt
| raiseStmt
| breakStmt
| continueStmt
| pragma
| importStmt
| fromStmt
| includeStmt
| exprStmt
complexStmt ::= ifStmt | whileStmt | caseStmt | tryStmt | forStmt
| blockStmt | asmStmt
| procDecl | iteratorDecl | macroDecl | templateDecl
| constSection | letSection
| typeSection | whenStmt | varSection
Discard statement Discard statement
----------------- -----------------
Syntax::
discardStmt ::= 'discard' expr
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -1766,16 +1756,6 @@ been declared with the `discardable`:idx: pragma:
Var statement Var statement
------------- -------------
Syntax::
colonOrEquals ::= ':' typeDesc ['=' expr] | '=' expr
varField ::= symbol ['*'] [pragma]
varPart ::= symbol (comma symbol)* [comma] colonOrEquals [COMMENT | IND COMMENT]
varSection ::= 'var' (varPart
| indPush (COMMENT|varPart)
(SAD (COMMENT|varPart))* DED indPop)
`Var`:idx: statements declare new local and global variables and `Var`:idx: statements declare new local and global variables and
initialize them. A comma separated list of variables can be used to specify initialize them. A comma separated list of variables can be used to specify
variables of the same type: variables of the same type:
@ -1839,14 +1819,6 @@ For let variables the same pragmas are available as for ordinary variables.
Const section Const section
------------- -------------
Syntax::
colonAndEquals ::= [':' typeDesc] '=' expr
constDecl ::= symbol ['*'] [pragma] colonAndEquals [COMMENT | IND COMMENT]
| COMMENT
constSection ::= 'const' indPush constDecl (SAD constDecl)* DED indPop
`Constants`:idx: are symbols which are bound to a value. The constant's value `Constants`:idx: are symbols which are bound to a value. The constant's value
cannot change. The compiler must be able to evaluate the expression in a cannot change. The compiler must be able to evaluate the expression in a
constant declaration at compile time. constant declaration at compile time.
@ -1877,10 +1849,6 @@ they contain such a type.
Static statement/expression Static statement/expression
--------------------------- ---------------------------
Syntax::
staticExpr ::= 'static' '(' optInd expr optPar ')'
staticStmt ::= 'static' ':' stmt
A `static`:idx: statement/expression can be used to enforce compile A `static`:idx: statement/expression can be used to enforce compile
time evaluation explicitly. Enforced compile time evaluation can even evaluate time evaluation explicitly. Enforced compile time evaluation can even evaluate
code that has side effects: code that has side effects:
@ -1902,10 +1870,6 @@ support the FFI at compile time.
If statement If statement
------------ ------------
Syntax::
ifStmt ::= 'if' expr ':' stmt ('elif' expr ':' stmt)* ['else' ':' stmt]
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -1932,12 +1896,6 @@ part, execution continues with the statement after the ``if`` statement.
Case statement Case statement
-------------- --------------
Syntax::
caseStmt ::= 'case' expr [':'] ('of' sliceExprList ':' stmt)*
('elif' expr ':' stmt)*
['else' ':' stmt]
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -1998,10 +1956,6 @@ a list of its elements:
When statement When statement
-------------- --------------
Syntax::
whenStmt ::= 'when' expr ':' stmt ('elif' expr ':' stmt)* ['else' ':' stmt]
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -2032,10 +1986,6 @@ within ``object`` definitions.
Return statement Return statement
---------------- ----------------
Syntax::
returnStmt ::= 'return' [expr]
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -2063,10 +2013,6 @@ variables, ``result`` is initialized to (binary) zero:
Yield statement Yield statement
--------------- ---------------
Syntax::
yieldStmt ::= 'yield' expr
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -2083,10 +2029,6 @@ for further information.
Block statement Block statement
--------------- ---------------
Syntax::
blockStmt ::= 'block' [symbol] ':' stmt
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -2108,10 +2050,6 @@ block to specify which block is to leave.
Break statement Break statement
--------------- ---------------
Syntax::
breakStmt ::= 'break' [symbol]
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -2125,10 +2063,6 @@ absent, the innermost block is left.
While statement While statement
--------------- ---------------
Syntax::
whileStmt ::= 'while' expr ':' stmt
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -2147,10 +2081,6 @@ so that they can be left with a ``break`` statement.
Continue statement Continue statement
------------------ ------------------
Syntax::
continueStmt ::= 'continue'
A `continue`:idx: statement leads to the immediate next iteration of the A `continue`:idx: statement leads to the immediate next iteration of the
surrounding loop construct. It is only allowed within a loop. A continue surrounding loop construct. It is only allowed within a loop. A continue
statement is syntactic sugar for a nested block: statement is syntactic sugar for a nested block:
@ -2173,9 +2103,6 @@ Is equivalent to:
Assembler statement Assembler statement
------------------- -------------------
Syntax::
asmStmt ::= 'asm' [pragma] (STR_LIT | RSTR_LIT | TRIPLESTR_LIT)
The direct embedding of `assembler`:idx: code into Nimrod code is supported The direct embedding of `assembler`:idx: code into Nimrod code is supported
by the unsafe ``asm`` statement. Identifiers in the assembler code that refer to by the unsafe ``asm`` statement. Identifiers in the assembler code that refer to
@ -2203,8 +2130,7 @@ Example:
var y = if x > 8: 9 else: 10 var y = if x > 8: 9 else: 10
An if expression always results in a value, so the ``else`` part is An if expression always results in a value, so the ``else`` part is
required. ``Elif`` parts are also allowed (but unlikely to be good required. ``Elif`` parts are also allowed.
style).
When expression When expression
--------------- ---------------
@ -2311,18 +2237,8 @@ procedure declaration defines an identifier and associates it with a block
of code. of code.
A procedure may call itself recursively. A parameter may be given a default A procedure may call itself recursively. A parameter may be given a default
value that is used if the caller does not provide a value for this parameter. value that is used if the caller does not provide a value for this parameter.
The syntax is::
param ::= symbol (comma symbol)* (':' typeDesc ['=' expr] | '=' expr) If the proc declaration has no body, it is a `forward`:idx: declaration. If
paramList ::= ['(' [param (comma param)*] [SAD] ')'] [':' typeDesc]
genericParam ::= symbol [':' typeDesc] ['=' expr]
genericParams ::= '[' genericParam (comma genericParam)* [SAD] ']'
procDecl ::= 'proc' symbol ['*'] [genericParams] paramList [pragma]
['=' stmt]
If the ``= stmt`` part is missing, it is a `forward`:idx: declaration. If
the proc returns a value, the procedure body can access an implicitly declared the proc returns a value, the procedure body can access an implicitly declared
variable named `result`:idx: that represents the return value. Procs can be variable named `result`:idx: that represents the return value. Procs can be
overloaded. The overloading resolution algorithm tries to find the proc that is overloaded. The overloading resolution algorithm tries to find the proc that is
@ -2417,24 +2333,14 @@ Do notation
As a special more convenient notation, proc expressions involved in procedure As a special more convenient notation, proc expressions involved in procedure
calls can use the ``do`` keyword: calls can use the ``do`` keyword:
Syntax::
primarySuffix ::= 'do' ['(' namedExprList ')'] ['->' typeDesc] ':'
As a start, let's repeat the example from the previous section:
.. code-block:: nimrod
cities.sort do (x,y: string) -> int:
cmp(x.len, y.len)
``do`` is written after the parentheses enclosing the regular proc params.
The proc expression represented by the do block is appended to them.
Again, let's see the equivalent of the previous example:
.. code-block:: nimrod .. code-block:: nimrod
sort(cities) do (x,y: string) -> int: sort(cities) do (x,y: string) -> int:
cmp(x.len, y.len) cmp(x.len, y.len)
Finally, more than one ``do`` block can appear in a single call: ``do`` is written after the parentheses enclosing the regular proc params.
The proc expression represented by the do block is appended to them.
More than one ``do`` block can appear in a single call:
.. code-block:: nimrod .. code-block:: nimrod
proc performWithUndo(task: proc(), undo: proc()) = ... proc performWithUndo(task: proc(), undo: proc()) = ...
@ -2635,30 +2541,16 @@ evaluation or dead code elimination do not work with methods.
Iterators and the for statement Iterators and the for statement
=============================== ===============================
Syntax::
forStmt ::= 'for' symbol (comma symbol)* [comma] 'in' expr ':' stmt
param ::= symbol (comma symbol)* [comma] ':' typeDesc
paramList ::= ['(' [param (comma param)* [comma]] ')'] [':' typeDesc]
genericParam ::= symbol [':' typeDesc]
genericParams ::= '[' genericParam (comma genericParam)* [comma] ']'
iteratorDecl ::= 'iterator' symbol ['*'] [genericParams] paramList [pragma]
['=' stmt]
The `for`:idx: statement is an abstract mechanism to iterate over the elements The `for`:idx: statement is an abstract mechanism to iterate over the elements
of a container. It relies on an `iterator`:idx: to do so. Like ``while`` of a container. It relies on an `iterator`:idx: to do so. Like ``while``
statements, ``for`` statements open an `implicit block`:idx:, so that they statements, ``for`` statements open an `implicit block`:idx:, so that they
can be left with a ``break`` statement. can be left with a ``break`` statement.
The ``for`` loop declares The ``for`` loop declares iteration variables - their scope reaches until the
iteration variables (``x`` in the example) - their scope reaches until the
end of the loop body. The iteration variables' types are inferred by the end of the loop body. The iteration variables' types are inferred by the
return type of the iterator. return type of the iterator.
An iterator is similar to a procedure, except that it is always called in the An iterator is similar to a procedure, except that it can be called in the
context of a ``for`` loop. Iterators provide a way to specify the iteration over context of a ``for`` loop. Iterators provide a way to specify the iteration over
an abstract type. A key role in the execution of a ``for`` loop plays the an abstract type. A key role in the execution of a ``for`` loop plays the
``yield`` statement in the called iterator. Whenever a ``yield`` statement is ``yield`` statement in the called iterator. Whenever a ``yield`` statement is
@ -2686,9 +2578,10 @@ The compiler generates code as if the programmer would have written this:
echo(ch) echo(ch)
inc(i) inc(i)
If the iterator yields a tuple, there have to be as many iteration variables If the iterator yields a tuple, there can be as many iteration variables
as there are components in the tuple. The i'th iteration variable's type is as there are components in the tuple. The i'th iteration variable's type is
the type of the i'th component. the type of the i'th component. In other words, implicit tuple unpacking in a
for loop context is supported.
Implict items/pairs invocations Implict items/pairs invocations
@ -2792,23 +2685,10 @@ iterator that has already finished its work.
Type sections Type sections
============= =============
Syntax::
typeDef ::= typeDesc | objectDef | enumDef
genericParam ::= symbol [':' typeDesc]
genericParams ::= '[' genericParam (comma genericParam)* [comma] ']'
typeDecl ::= COMMENT
| symbol ['*'] [genericParams] ['=' typeDef] [COMMENT|IND COMMENT]
typeSection ::= 'type' indPush typeDecl (SAD typeDecl)* DED indPop
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
type # example demonstrates mutually recursive types type # example demonstrating mutually recursive types
PNode = ref TNode # a traced pointer to a TNode PNode = ref TNode # a traced pointer to a TNode
TNode = object TNode = object
le, ri: PNode # left and right subtrees le, ri: PNode # left and right subtrees
@ -2822,7 +2702,8 @@ Example:
A `type`:idx: section begins with the ``type`` keyword. It contains multiple A `type`:idx: section begins with the ``type`` keyword. It contains multiple
type definitions. A type definition binds a type to a name. Type definitions type definitions. A type definition binds a type to a name. Type definitions
can be recursive or even mutually recursive. Mutually recursive types are only can be recursive or even mutually recursive. Mutually recursive types are only
possible within a single ``type`` section. possible within a single ``type`` section. Nominal types like ``objects``
or ``enums`` can only be defined in a ``type`` section.
Exception handling Exception handling
@ -2831,14 +2712,6 @@ Exception handling
Try statement Try statement
------------- -------------
Syntax::
qualifiedIdent ::= symbol ['.' symbol]
exceptList ::= [qualifiedIdent (comma qualifiedIdent)* [comma]]
tryStmt ::= 'try' ':' stmt
('except' exceptList ':' stmt)*
['finally' ':' stmt]
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -2863,15 +2736,14 @@ Example:
close(f) close(f)
The statements after the `try`:idx: are executed in sequential order unless The statements after the `try`:idx: are executed in sequential order unless
an exception ``e`` is raised. If the exception type of ``e`` matches any an exception ``e`` is raised. If the exception type of ``e`` matches any
of the list ``exceptlist`` the corresponding statements are executed. listed in an ``except`` clause the corresponding statements are executed.
The statements following the ``except`` clauses are called The statements following the ``except`` clauses are called
`exception handlers`:idx:. `exception handlers`:idx:.
The empty `except`:idx: clause is executed if there is an exception that is The empty `except`:idx: clause is executed if there is an exception that is
in no list. It is similar to an ``else`` clause in ``if`` statements. not listed otherwise. It is similar to an ``else`` clause in ``if`` statements.
If there is a `finally`:idx: clause, it is always executed after the If there is a `finally`:idx: clause, it is always executed after the
exception handlers. exception handlers.
@ -2916,10 +2788,6 @@ statements. Example:
Raise statement Raise statement
--------------- ---------------
Syntax::
raiseStmt ::= 'raise' [expr]
Example: Example:
.. code-block:: nimrod .. code-block:: nimrod
@ -2948,17 +2816,21 @@ This allows for a Lisp-like `condition system`:idx:\:
.. code-block:: nimrod .. code-block:: nimrod
var myFile = open("broken.txt", fmWrite) var myFile = open("broken.txt", fmWrite)
try: try:
onRaise(proc (e: ref E_Base): bool = onRaise do (e: ref E_Base)-> bool:
if e of EIO: if e of EIO:
stdout.writeln "ok, writing to stdout instead" stdout.writeln "ok, writing to stdout instead"
else: else:
# do raise other exceptions: # do raise other exceptions:
result = true result = true
)
myFile.writeln "writing to broken file" myFile.writeln "writing to broken file"
finally: finally:
myFile.close() myFile.close()
``OnRaise`` can only *filter* raised exceptions, it cannot transform one
exception into another. (Nor should ``onRaise`` raise an exception though
this is currently not enforced.) This restriction keeps the exception tracking
analysis sound.
Effect system Effect system
============= =============
@ -3447,10 +3319,6 @@ Symbol binding within templates happens after template instantiation:
Bind statement Bind statement
-------------- --------------
Syntax::
bindStmt ::= 'bind' IDENT (comma IDENT)*
Exporting a template is a often a leaky abstraction as it can depend on Exporting a template is a often a leaky abstraction as it can depend on
symbols that are not visible from a client module. However, to compensate for symbols that are not visible from a client module. However, to compensate for
this case, a `bind`:idx: statement can be used: It declares all identifiers this case, a `bind`:idx: statement can be used: It declares all identifiers
@ -3715,18 +3583,11 @@ Statement Macros
---------------- ----------------
Statement macros are defined just as expression macros. However, they are Statement macros are defined just as expression macros. However, they are
invoked by an expression following a colon:: invoked by an expression following a colon.
exprStmt ::= lowestExpr ['=' expr | [expr (comma expr)* [comma]] [macroStmt]]
macroStmt ::= ':' [stmt] ('of' [sliceExprList] ':' stmt
| 'elif' expr ':' stmt
| 'except' exceptList ':' stmt )*
['else' ':' stmt]
The following example outlines a macro that generates a lexical analyzer from The following example outlines a macro that generates a lexical analyzer from
regular expressions: regular expressions:
.. code-block:: nimrod .. code-block:: nimrod
import macros import macros
@ -3799,7 +3660,7 @@ instantiation type using the param name:
var tree = new(TBinaryTree[int]) var tree = new(TBinaryTree[int])
When used with macros and .compileTime. procs on the other hand, the compiler When used with macros and .compileTime. procs on the other hand, the compiler
don't need to instantiate the code multiple times, because types then can be does not need to instantiate the code multiple times, because types then can be
manipulated using the unified internal symbol representation. In such context manipulated using the unified internal symbol representation. In such context
typedesc acts as any other type. One can create variables, store typedesc typedesc acts as any other type. One can create variables, store typedesc
values inside containers and so on. For example, here is how one can create values inside containers and so on. For example, here is how one can create
@ -4358,13 +4219,6 @@ the compiler encounters any static error.
Pragmas Pragmas
======= =======
Syntax::
colonExpr ::= expr [':' expr]
colonExprList ::= [colonExpr (comma colonExpr)* [comma]]
pragma ::= '{.' optInd (colonExpr [comma])* [SAD] ('.}' | '}')
Pragmas are Nimrod's method to give the compiler additional information / Pragmas are Nimrod's method to give the compiler additional information /
commands without introducing a massive number of new keywords. Pragmas are commands without introducing a massive number of new keywords. Pragmas are
processed on the fly during semantic checking. Pragmas are enclosed in the processed on the fly during semantic checking. Pragmas are enclosed in the
@ -4411,7 +4265,7 @@ calls to any base class destructors in both user-defined and generated
destructors. destructors.
A destructor is attached to the type it destructs; expressions of this type A destructor is attached to the type it destructs; expressions of this type
can then only be used in *destructible contexts*: can then only be used in *destructible contexts* and as parameters:
.. code-block:: nimrod .. code-block:: nimrod
type type
@ -4425,9 +4279,15 @@ can then only be used in *destructible contexts*:
proc open: TMyObj = proc open: TMyObj =
result = TMyObj(x: 1, y: 2, p: alloc(3)) result = TMyObj(x: 1, y: 2, p: alloc(3))
proc work(o: TMyObj) =
echo o.x
# No destructor invoked here for 'o' as 'o' is a parameter.
proc main() = proc main() =
# destructor automatically invoked at the end of the scope: # destructor automatically invoked at the end of the scope:
var x = open() var x = open()
# valid: pass 'x' to some other proc:
work(x)
# Error: usage of a type with a destructor in a non destructible context # Error: usage of a type with a destructor in a non destructible context
echo open() echo open()
@ -4849,8 +4709,8 @@ a dynamic library (``.dll`` files for Windows, ``lib*.so`` files for UNIX).
The non-optional argument has to be the name of the dynamic library: The non-optional argument has to be the name of the dynamic library:
.. code-block:: Nimrod .. code-block:: Nimrod
proc gtk_image_new(): PGtkWidget {. proc gtk_image_new(): PGtkWidget
cdecl, dynlib: "libgtk-x11-2.0.so", importc.} {.cdecl, dynlib: "libgtk-x11-2.0.so", importc.}
In general, importing a dynamic library does not require any special linker In general, importing a dynamic library does not require any special linker
options or linking with import libraries. This also implies that no *devel* options or linking with import libraries. This also implies that no *devel*
@ -4894,6 +4754,10 @@ strings, because they are precompiled.
**Note**: Passing variables to the ``dynlib`` pragma will fail at runtime **Note**: Passing variables to the ``dynlib`` pragma will fail at runtime
because of order of initialization problems. because of order of initialization problems.
**Note**: A ``dynlib`` import can be overriden with
the ``--dynlibOverride:name`` command line option. The Compiler User Guide
contains further information.
Dynlib pragma for export Dynlib pragma for export
------------------------ ------------------------
@ -4971,7 +4835,7 @@ Nimrod supports the `actor model`:idx: of concurrency natively:
type type
TMsgKind = enum TMsgKind = enum
mLine, mEof mLine, mEof
TMsg = object {.pure, final.} TMsg = object
case k: TMsgKind case k: TMsgKind
of mEof: nil of mEof: nil
of mLine: data: string of mLine: data: string

View file

@ -7,7 +7,17 @@ version 0.9.2
- acyclic vs prunable; introduce GC hints - acyclic vs prunable; introduce GC hints
- CGEN: ``restrict`` pragma + backend support; computed goto support - CGEN: ``restrict`` pragma + backend support; computed goto support
- document NimMain and check whether it works for threading - document NimMain and check whether it works for threading
- parser/grammar:
* check that of branches can only receive even simpler expressions, don't
allow 'of (var x = 23; nkIdent)'
* allow (var x = 12; for i in ... ; x) construct
* try except as an expression
- make use of commonType relation in expressions
- further expr/stmt unification:
- nkIfStmt vs nkIfExpr
- start with JS backend and support exprs everywhere
- then enhance C backend
- OR: do the temp stuff in transf
Bugs Bugs
==== ====
@ -29,14 +39,13 @@ version 0.9.4
============= =============
- macros as type pragmas - macros as type pragmas
- ``try`` as an expression
- provide tool/API to track leaks/object counts - provide tool/API to track leaks/object counts
- hybrid GC - hybrid GC
- use big blocks in the allocator - use big blocks in the allocator
- implement full 'not nil' checking - implement full 'not nil' checking
- make 'bind' default for templates and introduce 'mixin'; - make 'bind' default for templates and introduce 'mixin';
special rule for ``[]=`` special rule for ``[]=``
- implicit deref for parameter matching; overloading based on 'var T' - implicit deref for parameter matching
- ``=`` should be overloadable; requires specialization for ``=``; general - ``=`` should be overloadable; requires specialization for ``=``; general
lift mechanism in the compiler is already implemented for 'fields' lift mechanism in the compiler is already implemented for 'fields'
- lazy overloading resolution: - lazy overloading resolution:
@ -54,9 +63,14 @@ version 0.9.X
- improve the compiler as a service - improve the compiler as a service
- better support for macros that rewrite procs - better support for macros that rewrite procs
- macros need access to types and symbols (partially implemented) - macros need access to types and symbols (partially implemented)
- rethink the syntax/grammar: - perhaps: change comment handling in the AST
* parser is not strict enough with newlines - enforce 'simpleExpr' more often --> doesn't work; tkProc is
* change comment handling in the AST part of primary!
- the typeDesc/expr unification is weird and only necessary because of
the ambiguous a[T] construct: It would be easy to support a[expr] for
generics but require a[.typeDesc] if that's required; this would also
allow [.ref T.](x) for a more general type conversion construct; for
templates that would work too: T([.ref int])
Concurrency Concurrency
@ -96,7 +110,8 @@ Not essential for 1.0.0
- mocking support with ``tyProxy`` that does: fallback for ``.`` operator - mocking support with ``tyProxy`` that does: fallback for ``.`` operator
- overloading of ``.``? Special case ``.=``? - overloading of ``.``? Special case ``.=``?
- allow implicit forward declarations of procs via a pragma (so that the - allow implicit forward declarations of procs via a pragma (so that the
wrappers can deactivate it) wrappers can deactivate it): better solution: introduce the notion of a
'proc section' that is similar to a type section.
- implement the "snoopResult" pragma; no, make a strutils with string append - implement the "snoopResult" pragma; no, make a strutils with string append
semantics instead ... semantics instead ...
- implement "closure tuple consists of a single 'ref'" optimization - implement "closure tuple consists of a single 'ref'" optimization