new parser works

This commit is contained in:
Araq 2013-04-20 18:46:08 +02:00
commit 03764f0aba
4 changed files with 66 additions and 39 deletions

View file

@ -797,6 +797,7 @@ proc rawGetTok(L: var TLexer, tok: var TToken) =
getOperator(L, tok) getOperator(L, tok)
elif c == lexbase.EndOfFile: elif c == lexbase.EndOfFile:
tok.toktype = tkEof tok.toktype = tkEof
tok.indent = 0
else: else:
tok.literal = $c tok.literal = $c
tok.tokType = tkInvalid tok.tokType = tkInvalid

View file

@ -786,8 +786,9 @@ proc sourceLine*(i: TLineInfo): PRope =
for line in lines(i.toFullPath): for line in lines(i.toFullPath):
addSourceLine i.fileIndex, line.string addSourceLine i.fileIndex, line.string
InternalAssert i.fileIndex < fileInfos.len and InternalAssert i.fileIndex < fileInfos.len
i.line <= fileInfos[i.fileIndex].lines.len # can happen if the error points to EOF:
if i.line > fileInfos[i.fileIndex].lines.len: return nil
result = fileInfos[i.fileIndex].lines[i.line-1] result = fileInfos[i.fileIndex].lines[i.line-1]

View file

@ -99,6 +99,7 @@ template withInd(p: expr, body: stmt) {.immediate.} =
template realInd(p): bool = p.tok.indent > p.currInd template realInd(p): bool = p.tok.indent > p.currInd
template sameInd(p): bool = p.tok.indent == p.currInd template sameInd(p): bool = p.tok.indent == p.currInd
template sameOrNoInd(p): bool = p.tok.indent == p.currInd or p.tok.indent < 0
proc rawSkipComment(p: var TParser, node: PNode) = proc rawSkipComment(p: var TParser, node: PNode) =
if p.tok.tokType == tkComment: if p.tok.tokType == tkComment:
@ -313,7 +314,7 @@ proc indexExprList(p: var TParser, first: PNode, k: TNodeKind,
addSon(result, a) addSon(result, a)
if p.tok.tokType != tkComma: break if p.tok.tokType != tkComma: break
getTok(p) getTok(p)
optInd(p, a) skipComment(p, a)
optPar(p) optPar(p)
eat(p, endToken) eat(p, endToken)
@ -380,7 +381,7 @@ proc exprColonEqExprListAux(p: var TParser, endTok: TTokType, result: PNode) =
addSon(result, a) addSon(result, a)
if p.tok.tokType != tkComma: break if p.tok.tokType != tkComma: break
getTok(p) getTok(p)
optInd(p, a) skipComment(p, a)
optPar(p) optPar(p)
eat(p, endTok) eat(p, endTok)
@ -405,7 +406,7 @@ proc setOrTableConstr(p: var TParser): PNode =
addSon(result, a) addSon(result, a)
if p.tok.tokType != tkComma: break if p.tok.tokType != tkComma: break
getTok(p) getTok(p)
optInd(p, a) skipComment(p, a)
optPar(p) optPar(p)
eat(p, tkCurlyRi) # skip '}' eat(p, tkCurlyRi) # skip '}'
@ -560,7 +561,7 @@ proc primarySuffix(p: var TParser, r: PNode): PNode =
#| | '[' optInd indexExprList optPar ']' #| | '[' optInd indexExprList optPar ']'
#| | '{' optInd indexExprList optPar '}' #| | '{' optInd indexExprList optPar '}'
result = r result = r
while true: while p.tok.indent < 0:
case p.tok.tokType case p.tok.tokType
of tkParLe: of tkParLe:
var a = result var a = result
@ -595,12 +596,13 @@ proc lowestExprAux(p: var TParser, limit: int, mode: TPrimaryMode): PNode =
# expand while operators have priorities higher than 'limit' # expand while operators have priorities higher than 'limit'
var opPrec = getPrecedence(p.tok) var opPrec = getPrecedence(p.tok)
let modeB = if mode == pmTypeDef: pmTypeDesc else: mode let modeB = if mode == pmTypeDef: pmTypeDesc else: mode
while opPrec >= limit: # the operator itself must not start on a new line:
while opPrec >= limit and p.tok.indent < 0:
var leftAssoc = ord(IsLeftAssociative(p.tok)) var leftAssoc = ord(IsLeftAssociative(p.tok))
var a = newNodeP(nkInfix, p) var a = newNodeP(nkInfix, p)
var opNode = newIdentNodeP(p.tok.ident, p) # skip operator: var opNode = newIdentNodeP(p.tok.ident, p) # skip operator:
getTok(p) getTok(p)
optInd(p, opNode) optInd(p, opNode)
# read sub-expression with higher priority: # read sub-expression with higher priority:
var b = lowestExprAux(p, opPrec + leftAssoc, modeB) var b = lowestExprAux(p, opPrec + leftAssoc, modeB)
addSon(a, opNode) addSon(a, opNode)
@ -613,9 +615,9 @@ proc lowestExpr(p: var TParser, mode = pmNormal): PNode =
result = lowestExprAux(p, -1, mode) result = lowestExprAux(p, -1, mode)
proc parseIfExpr(p: var TParser, kind: TNodeKind): PNode = proc parseIfExpr(p: var TParser, kind: TNodeKind): PNode =
#| condExpr = expr ':' optInd expr optInd #| condExpr = expr colcom expr optInd
#| ('elif' expr ':' optInd expr optInd)* #| ('elif' expr colcom expr optInd)*
#| 'else' ':' optInd expr #| 'else' colcom expr
#| ifExpr = 'if' condExpr #| ifExpr = 'if' condExpr
#| whenExpr = 'when' condExpr #| whenExpr = 'when' condExpr
result = newNodeP(kind, p) result = newNodeP(kind, p)
@ -623,16 +625,14 @@ proc parseIfExpr(p: var TParser, kind: TNodeKind): PNode =
getTok(p) # skip `if`, `elif` getTok(p) # skip `if`, `elif`
var branch = newNodeP(nkElifExpr, p) var branch = newNodeP(nkElifExpr, p)
addSon(branch, parseExpr(p)) addSon(branch, parseExpr(p))
eat(p, tkColon) colcom(p, branch)
optInd(p, branch)
addSon(branch, parseExpr(p)) addSon(branch, parseExpr(p))
optInd(p, branch) optInd(p, branch)
addSon(result, branch) addSon(result, branch)
if p.tok.tokType != tkElif: break if p.tok.tokType != tkElif: break
var branch = newNodeP(nkElseExpr, p) var branch = newNodeP(nkElseExpr, p)
eat(p, tkElse) eat(p, tkElse)
eat(p, tkColon) colcom(p, branch)
optInd(p, branch)
addSon(branch, parseExpr(p)) addSon(branch, parseExpr(p))
addSon(result, branch) addSon(result, branch)
@ -646,7 +646,7 @@ proc parsePragma(p: var TParser): PNode =
addSon(result, a) addSon(result, a)
if p.tok.tokType == tkComma: if p.tok.tokType == tkComma:
getTok(p) getTok(p)
optInd(p, a) skipComment(p, a)
optPar(p) optPar(p)
if p.tok.tokType in {tkCurlyDotRi, tkCurlyRi}: getTok(p) if p.tok.tokType in {tkCurlyDotRi, tkCurlyRi}: getTok(p)
else: parMessage(p, errTokenExpected, ".}") else: parMessage(p, errTokenExpected, ".}")
@ -726,7 +726,7 @@ proc parseTuple(p: var TParser, indentAllowed = false): PNode =
addSon(result, a) addSon(result, a)
if p.tok.tokType notin {tkComma, tkSemicolon}: break if p.tok.tokType notin {tkComma, tkSemicolon}: break
getTok(p) getTok(p)
optInd(p, a) skipComment(p, a)
optPar(p) optPar(p)
eat(p, tkBracketRi) eat(p, tkBracketRi)
elif indentAllowed: elif indentAllowed:
@ -768,7 +768,7 @@ proc parseParamList(p: var TParser, retColon = true): PNode =
addSon(result, a) addSon(result, a)
if p.tok.tokType notin {tkComma, tkSemicolon}: break if p.tok.tokType notin {tkComma, tkSemicolon}: break
getTok(p) getTok(p)
optInd(p, a) skipComment(p, a)
optPar(p) optPar(p)
eat(p, tkParRi) eat(p, tkParRi)
let hasRet = if retColon: p.tok.tokType == tkColon let hasRet = if retColon: p.tok.tokType == tkColon
@ -941,6 +941,13 @@ proc parseTypeDefAux(p: var TParser): PNode =
#| typeDefAux = lowestExpr #| typeDefAux = lowestExpr
result = lowestExpr(p, pmTypeDef) result = lowestExpr(p, pmTypeDef)
proc makeCall(n: PNode): PNode =
if n.kind in nkCallKinds:
result = n
else:
result = newNodeI(nkCall, n.info)
result.add n
proc parseExprStmt(p: var TParser): PNode = proc parseExprStmt(p: var TParser): PNode =
#| exprStmt = lowestExpr #| exprStmt = lowestExpr
#| (( '=' optInd expr ) #| (( '=' optInd expr )
@ -971,9 +978,11 @@ proc parseExprStmt(p: var TParser): PNode =
else: else:
result = a result = a
if p.tok.tokType == tkDo and p.tok.indent < 0: if p.tok.tokType == tkDo and p.tok.indent < 0:
result = makeCall(result)
parseDoBlocks(p, result) parseDoBlocks(p, result)
return result return result
if p.tok.tokType == tkColon: if p.tok.tokType == tkColon and p.tok.indent < 0:
result = makeCall(result)
getTok(p) getTok(p)
skipComment(p, result) skipComment(p, result)
if p.tok.TokType notin {tkOf, tkElif, tkElse, tkExcept}: if p.tok.TokType notin {tkOf, tkElif, tkElse, tkExcept}:
@ -982,7 +991,7 @@ proc parseExprStmt(p: var TParser): PNode =
while sameInd(p): while sameInd(p):
var b: PNode var b: PNode
case p.tok.tokType case p.tok.tokType
of tkOf: of tkOf:
b = newNodeP(nkOfBranch, p) b = newNodeP(nkOfBranch, p)
exprList(p, tkColon, b) exprList(p, tkColon, b)
of tkElif: of tkElif:
@ -1071,7 +1080,11 @@ proc parseReturnOrRaise(p: var TParser, kind: TNodeKind): PNode =
#| continueStmt = 'break' optInd expr? #| continueStmt = 'break' optInd expr?
result = newNodeP(kind, p) result = newNodeP(kind, p)
getTok(p) getTok(p)
if p.tok.indent >= 0 and p.tok.indent <= p.currInd or p.tok.tokType == tkEof: if p.tok.tokType == tkComment:
skipComment(p, result)
addSon(result, ast.emptyNode)
elif p.tok.indent >= 0 and p.tok.indent <= p.currInd or
p.tok.tokType == tkEof:
# NL terminates: # NL terminates:
addSon(result, ast.emptyNode) addSon(result, ast.emptyNode)
else: else:
@ -1094,8 +1107,8 @@ proc parseIfOrWhen(p: var TParser, kind: TNodeKind): PNode =
addSon(branch, parseStmt(p)) addSon(branch, parseStmt(p))
skipComment(p, branch) skipComment(p, branch)
addSon(result, branch) addSon(result, branch)
if p.tok.tokType != tkElif or not sameInd(p): break if p.tok.tokType != tkElif or not sameOrNoInd(p): break
if p.tok.tokType == tkElse and sameInd(p): if p.tok.tokType == tkElse and sameOrNoInd(p):
var branch = newNodeP(nkElse, p) var branch = newNodeP(nkElse, p)
eat(p, tkElse) eat(p, tkElse)
eat(p, tkColon) eat(p, tkColon)
@ -1133,7 +1146,6 @@ proc parseCase(p: var TParser): PNode =
let oldInd = p.currInd let oldInd = p.currInd
if realInd(p): if realInd(p):
p.currInd = p.tok.indent p.currInd = p.tok.indent
getTok(p)
wasIndented = true wasIndented = true
while sameInd(p): while sameInd(p):
@ -1163,16 +1175,16 @@ proc parseCase(p: var TParser): PNode =
p.currInd = oldInd p.currInd = oldInd
proc parseTry(p: var TParser): PNode = proc parseTry(p: var TParser): PNode =
#| tryStmt = 'try' colcom stmt &(IND{=} 'except'|'finally') #| tryStmt = 'try' colcom stmt &(IND{=}? 'except'|'finally')
#| (IND{=} 'except' exprList colcom stmt)* #| (IND{=}? 'except' exprList colcom stmt)*
#| (IND{=} 'finally' colcom stmt)? #| (IND{=}? 'finally' colcom stmt)?
result = newNodeP(nkTryStmt, p) result = newNodeP(nkTryStmt, p)
getTok(p) getTok(p)
eat(p, tkColon) eat(p, tkColon)
skipComment(p, result) skipComment(p, result)
addSon(result, parseStmt(p)) addSon(result, parseStmt(p))
var b: PNode = nil var b: PNode = nil
while sameInd(p): while sameOrNoInd(p):
case p.tok.tokType case p.tok.tokType
of tkExcept: of tkExcept:
b = newNodeP(nkExceptBranch, p) b = newNodeP(nkExceptBranch, p)
@ -1282,7 +1294,7 @@ proc parseGenericParamList(p: var TParser): PNode =
addSon(result, a) addSon(result, a)
if p.tok.tokType notin {tkComma, tkSemicolon}: break if p.tok.tokType notin {tkComma, tkSemicolon}: break
getTok(p) getTok(p)
optInd(p, a) skipComment(p, a)
optPar(p) optPar(p)
eat(p, tkBracketRi) eat(p, tkBracketRi)
@ -1380,10 +1392,12 @@ proc parseEnum(p: var TParser): PNode =
getTok(p) getTok(p)
addSon(result, ast.emptyNode) addSon(result, ast.emptyNode)
optInd(p, result) optInd(p, result)
while true: while true:
var a = parseSymbol(p) var a = parseSymbol(p)
optInd(p, a) if p.tok.indent >= 0 and p.tok.indent <= p.currInd:
if p.tok.tokType == tkEquals: add(result, a)
break
if p.tok.tokType == tkEquals and p.tok.indent < 0:
getTok(p) getTok(p)
optInd(p, a) optInd(p, a)
var b = a var b = a
@ -1391,9 +1405,11 @@ proc parseEnum(p: var TParser): PNode =
addSon(a, b) addSon(a, b)
addSon(a, parseExpr(p)) addSon(a, parseExpr(p))
skipComment(p, a) skipComment(p, a)
if p.tok.tokType == tkComma: if p.tok.tokType == tkComma and p.tok.indent < 0:
getTok(p) getTok(p)
optInd(p, a) rawSkipComment(p, a)
else:
skipComment(p, a)
addSon(result, a) addSon(result, a)
if p.tok.indent >= 0 and p.tok.indent <= p.currInd or if p.tok.indent >= 0 and p.tok.indent <= p.currInd or
p.tok.tokType == tkEof: p.tok.tokType == tkEof:
@ -1476,7 +1492,7 @@ proc parseObjectPart(p: var TParser): PNode =
if realInd(p): if realInd(p):
result = newNodeP(nkRecList, p) result = newNodeP(nkRecList, p)
withInd(p): withInd(p):
skipComment(p, result) rawSkipComment(p, result)
while sameInd(p): while sameInd(p):
case p.tok.tokType case p.tok.tokType
of tkCase, tkWhen, tkSymbol, tkAccent, tkNil: of tkCase, tkWhen, tkSymbol, tkAccent, tkNil:
@ -1514,7 +1530,12 @@ proc parseObject(p: var TParser): PNode =
addSon(result, a) addSon(result, a)
else: else:
addSon(result, ast.emptyNode) addSon(result, ast.emptyNode)
skipComment(p, result) if p.tok.tokType == tkComment:
skipComment(p, result)
# an initial IND{>} HAS to follow:
if not realInd(p):
addSon(result, emptyNode)
return
addSon(result, parseObjectPart(p)) addSon(result, parseObjectPart(p))
proc parseDistinct(p: var TParser): PNode = proc parseDistinct(p: var TParser): PNode =
@ -1551,7 +1572,7 @@ proc parseVarTuple(p: var TParser): PNode =
addSon(result, a) addSon(result, a)
if p.tok.tokType != tkComma: break if p.tok.tokType != tkComma: break
getTok(p) getTok(p)
optInd(p, a) skipComment(p, a)
addSon(result, ast.emptyNode) # no type desc addSon(result, ast.emptyNode) # no type desc
optPar(p) optPar(p)
eat(p, tkParRi) eat(p, tkParRi)
@ -1670,7 +1691,11 @@ proc parseStmt(p: var TParser): PNode =
parMessage(p, errInvalidIndentation) parMessage(p, errInvalidIndentation)
break break
var a = complexOrSimpleStmt(p) var a = complexOrSimpleStmt(p)
if a.kind != nkEmpty: addSon(result, a) if a.kind != nkEmpty:
addSon(result, a)
else:
parMessage(p, errExprExpected, p.tok)
getTok(p)
else: else:
# the case statement is only needed for better error messages: # the case statement is only needed for better error messages:
case p.tok.tokType case p.tok.tokType

View file

@ -230,7 +230,7 @@ proc markIndirect*(c: PContext, s: PSym) {.inline.} =
incl(s.flags, sfAddrTaken) incl(s.flags, sfAddrTaken)
# XXX add to 'c' for global analysis # XXX add to 'c' for global analysis
proc illFormedAst*(n: PNode) = proc illFormedAst*(n: PNode) =
GlobalError(n.info, errIllFormedAstX, renderTree(n, {renderNoComments})) GlobalError(n.info, errIllFormedAstX, renderTree(n, {renderNoComments}))
proc checkSonsLen*(n: PNode, length: int) = proc checkSonsLen*(n: PNode, length: int) =