implemented marker procs for the GC resulting in huge speedups

This commit is contained in:
Araq 2012-03-21 23:10:56 +01:00
commit 03ba0f3e25
6 changed files with 169 additions and 18 deletions

121
compiler/ccgtrav.nim Normal file
View file

@ -0,0 +1,121 @@
#
#
# The Nimrod Compiler
# (c) Copyright 2012 Andreas Rumpf
#
# See the file "copying.txt", included in this
# distribution, for details about the copyright.
#
## Generates traversal procs for the C backend. Traversal procs are only an
## optimization; the GC works without them too.
type
TTraversalClosure {.pure, final.} = object
p: BProc
visitorFrmt: string
proc genTraverseProc(c: var TTraversalClosure, accessor: PRope, typ: PType)
proc genCaseRange(p: BProc, branch: PNode)
proc getTemp(p: BProc, t: PType, result: var TLoc)
proc genTraverseProc(c: var TTraversalClosure, accessor: PRope, n: PNode) =
if n == nil: return
case n.kind
of nkRecList:
for i in countup(0, sonsLen(n) - 1):
genTraverseProc(c, accessor, n.sons[i])
of nkRecCase:
if (n.sons[0].kind != nkSym): InternalError(n.info, "genTraverseProc")
var p = c.p
let disc = n.sons[0].sym
p.s[cpsStmts].appf("switch ($1.$2) {$n", accessor, disc.loc.r)
for i in countup(1, sonsLen(n) - 1):
let branch = n.sons[i]
assert branch.kind in {nkOfBranch, nkElse}
if branch.kind == nkOfBranch:
genCaseRange(c.p, branch)
else:
p.s[cpsStmts].appf("default:$n")
genTraverseProc(c, accessor, lastSon(branch))
p.s[cpsStmts].appf("break;$n")
p.s[cpsStmts].appf("} $n")
of nkSym:
let field = n.sym
genTraverseProc(c, ropef("$1.$2", accessor, field.loc.r), field.loc.t)
else: internalError(n.info, "genTraverseProc()")
proc parentObj(accessor: PRope): PRope {.inline.} =
if gCmd != cmdCompileToCpp:
result = ropef("$1.Sup", accessor)
else:
result = accessor
proc genTraverseProc(c: var TTraversalClosure, accessor: PRope, typ: PType) =
if typ == nil: return
var p = c.p
case typ.kind
of tyGenericInst, tyGenericBody:
genTraverseProc(c, accessor, lastSon(typ))
of tyArrayConstr, tyArray:
let arraySize = lengthOrd(typ.sons[0])
var i: TLoc
getTemp(p, getSysType(tyInt), i)
appf(p.s[cpsStmts], "for ($1 = 0; $1 < $2; $1++) {$n",
i.r, arraySize.toRope)
genTraverseProc(c, ropef("$1[$2]", accessor, i.r), typ.sons[1])
appf(p.s[cpsStmts], "}$n")
of tyObject:
for i in countup(0, sonsLen(typ) - 1):
genTraverseProc(c, accessor.parentObj, typ.sons[i])
if typ.n != nil: genTraverseProc(c, accessor, typ.n)
of tyTuple:
if typ.n != nil:
genTraverseProc(c, accessor, typ.n)
else:
for i in countup(0, sonsLen(typ) - 1):
genTraverseProc(c, ropef("$1.Field$2", accessor, i.toRope), typ.sons[i])
of tyRef, tyString, tySequence:
appcg(p, cpsStmts, c.visitorFrmt, accessor)
else:
# no marker procs for closures yet
nil
proc genTraverseProcSeq(c: var TTraversalClosure, accessor: PRope, typ: PType) =
var p = c.p
assert typ.kind == tySequence
var i: TLoc
getTemp(p, getSysType(tyInt), i)
appf(p.s[cpsStmts], "for ($1 = 0; $1 < $2->$3; $1++) {$n",
i.r, accessor, toRope(if gCmd != cmdCompileToCpp: "Sup.len" else: "len"))
genTraverseProc(c, ropef("$1->data[$2]", accessor, i.r), typ.sons[0])
appf(p.s[cpsStmts], "}$n")
proc genTraverseProc(m: BModule, typ: PType, reason: TTypeInfoReason): PRope =
var c: TTraversalClosure
var p = newProc(nil, m)
result = getGlobalTempName()
case reason
of tiNew: c.visitorFrmt = "#nimGCvisit((void*)$1, op);$n"
else: assert false
let header = ropef("N_NIMCALL(void, $1)(void* p, NI op)", result)
let t = getTypeDesc(m, typ)
p.s[cpsLocals].appf("$1 a;$n", t)
p.s[cpsInit].appf("a = ($1)p;$n", t)
c.p = p
if typ.kind == tySequence:
genTraverseProcSeq(c, "a".toRope, typ)
else:
genTraverseProc(c, "(*a)".toRope, typ.sons[0])
let generatedProc = ropef("$1 {$n$2$3$4}$n",
[header, p.s[cpsLocals], p.s[cpsInit], p.s[cpsStmts]])
m.s[cfsProcHeaders].appf("$1;$n", header)
m.s[cfsProcs].app(generatedProc)

View file

@ -151,7 +151,6 @@ proc ccgIntroducedPtr(s: PSym): bool =
assert skResult != s.kind assert skResult != s.kind
case pt.Kind case pt.Kind
of tyObject: of tyObject:
# XXX quick hack floatSize*2 for the pegs module under 64bit
if (optByRef in s.options) or (getSize(pt) > platform.floatSize * 2): if (optByRef in s.options) or (getSize(pt) > platform.floatSize * 2):
result = true # requested anyway result = true # requested anyway
elif (tfFinal in pt.flags) and (pt.sons[0] == nil): elif (tfFinal in pt.flags) and (pt.sons[0] == nil):
@ -160,7 +159,7 @@ proc ccgIntroducedPtr(s: PSym): bool =
result = true # ordinary objects are always passed by reference, result = true # ordinary objects are always passed by reference,
# otherwise casting doesn't work # otherwise casting doesn't work
of tyTuple: of tyTuple:
result = (getSize(pt) > platform.floatSize) or (optByRef in s.options) result = (getSize(pt) > platform.floatSize*2) or (optByRef in s.options)
else: result = false else: result = false
proc fillResult(param: PSym) = proc fillResult(param: PSym) =
@ -766,6 +765,15 @@ proc fakeClosureType(owner: PSym): PType =
r.addSon(newType(tyTuple, owner)) r.addSon(newType(tyTuple, owner))
result.addSon(r) result.addSon(r)
type
TTypeInfoReason = enum ## for what do we need the type info?
tiNew, ## for 'new'
tiNewSeq, ## for 'newSeq'
tiNonVariantAsgn, ## for generic assignment without variants
tiVariantAsgn ## for generic assignment with variants
include ccgtrav
proc genTypeInfo(m: BModule, typ: PType): PRope = proc genTypeInfo(m: BModule, typ: PType): PRope =
var t = getUniqueType(typ) var t = getUniqueType(typ)
# gNimDat contains all the type information nowadays: # gNimDat contains all the type information nowadays:
@ -787,7 +795,12 @@ proc genTypeInfo(m: BModule, typ: PType): PRope =
genTypeInfoAuxBase(gNimDat, t, result, toRope"0") genTypeInfoAuxBase(gNimDat, t, result, toRope"0")
else: else:
genTupleInfo(gNimDat, fakeClosureType(t.owner), result) genTupleInfo(gNimDat, fakeClosureType(t.owner), result)
of tyRef, tyPtr, tySequence, tyRange: genTypeInfoAux(gNimDat, t, result) of tySequence, tyRef:
genTypeInfoAux(gNimDat, t, result)
if optRefcGC in gGlobalOptions:
let markerProc = genTraverseProc(gNimDat, t, tiNew)
appf(gNimDat.s[cfsTypeInit3], "$1->marker = $2;$n", [result, markerProc])
of tyPtr, tyRange: genTypeInfoAux(gNimDat, t, result)
of tyArrayConstr, tyArray: genArrayInfo(gNimDat, t, result) of tyArrayConstr, tyArray: genArrayInfo(gNimDat, t, result)
of tySet: genSetInfo(gNimDat, t, result) of tySet: genSetInfo(gNimDat, t, result)
of tyEnum: genEnumInfo(gNimDat, t, result) of tyEnum: genEnumInfo(gNimDat, t, result)

View file

@ -338,6 +338,10 @@ proc forAllChildren(cell: PCell, op: TWalkOp) =
sysAssert(cell != nil, "forAllChildren: 1") sysAssert(cell != nil, "forAllChildren: 1")
sysAssert(cell.typ != nil, "forAllChildren: 2") sysAssert(cell.typ != nil, "forAllChildren: 2")
sysAssert cell.typ.kind in {tyRef, tySequence, tyString}, "forAllChildren: 3" sysAssert cell.typ.kind in {tyRef, tySequence, tyString}, "forAllChildren: 3"
let marker = cell.typ.marker
if marker != nil:
marker(cellToUsr(cell), op.int)
else:
case cell.typ.Kind case cell.typ.Kind
of tyRef: # common case of tyRef: # common case
forAllChildrenAux(cellToUsr(cell), cell.typ.base, op) forAllChildrenAux(cellToUsr(cell), cell.typ.base, op)
@ -528,6 +532,9 @@ proc doOperation(p: pointer, op: TWalkOp) =
sysAssert(c.refcount >=% rcIncrement, "doOperation 3") sysAssert(c.refcount >=% rcIncrement, "doOperation 3")
c.refcount = c.refcount -% rcIncrement c.refcount = c.refcount -% rcIncrement
proc nimGCvisit(d: pointer, op: int) {.compilerProc.} =
doOperation(d, TWalkOp(op))
# we now use a much simpler and non-recursive algorithm for cycle removal # we now use a much simpler and non-recursive algorithm for cycle removal
proc collectCycles(gch: var TGcHeap) = proc collectCycles(gch: var TGcHeap) =
var tabSize = 0 var tabSize = 0

View file

@ -60,6 +60,7 @@ type # This should be he same as ast.TTypeKind
base: ptr TNimType base: ptr TNimType
node: ptr TNimNode # valid for tyRecord, tyObject, tyTuple, tyEnum node: ptr TNimNode # valid for tyRecord, tyObject, tyTuple, tyEnum
finalizer: pointer # the finalizer for the type finalizer: pointer # the finalizer for the type
marker: proc (p: pointer, op: int) # marker proc for GC
PNimType = ptr TNimType PNimType = ptr TNimType
# node.len may be the ``first`` element of a set # node.len may be the ``first`` element of a set

View file

@ -10,9 +10,6 @@ version 0.9.0
- Test capture of for loop vars; test generics; - Test capture of for loop vars; test generics;
- test constant closures - test constant closures
- GC: marker procs for native Nimrod GC and Boehm GC; precise stack marking;
escape analysis for string/seq seems to be easy to do too;
even further write barrier specialization
- dead code elim for JS backend; 'of' operator for JS backend - dead code elim for JS backend; 'of' operator for JS backend
- const ptr/ref - const ptr/ref
- unsigned ints and bignums; requires abstract integer literal type: - unsigned ints and bignums; requires abstract integer literal type:
@ -133,6 +130,11 @@ Low priority
need to recompile clients; er ... what about templates, macros or anything need to recompile clients; er ... what about templates, macros or anything
that has inlining semantics? that has inlining semantics?
- codegen should use "NIM_CAST" macro and respect aliasing rules for GCC - codegen should use "NIM_CAST" macro and respect aliasing rules for GCC
- GC: precise stack marking;
escape analysis for string/seq seems to be easy to do too;
even further write barrier specialization
- GC: marker procs Boehm GC
- implement marker procs for assignment and message passing
- warning for implicit openArray -> varargs conversion - warning for implicit openArray -> varargs conversion
- implement explicit varargs; **but** ``len(varargs)`` problem remains! - implement explicit varargs; **but** ``len(varargs)`` problem remains!
--> solve by implicit conversion from varargs to openarray --> solve by implicit conversion from varargs to openarray

View file

@ -33,6 +33,11 @@ Changes affecting backwards compatibility
string versions of the WinAPI. Use the ``-d:useWinAnsi`` switch to revert string versions of the WinAPI. Use the ``-d:useWinAnsi`` switch to revert
back to the old behaviour which uses the Ansi string versions. back to the old behaviour which uses the Ansi string versions.
- ``static`` is now a keyword. - ``static`` is now a keyword.
- Templates now participate in overloading resolution which can break code that
uses templates in subtle ways. Use the new ``immediate`` pragma for templates
to get a template of old behaviour.
- There is now a proper distinction in the type system between ``expr`` and
``PNimrodNode`` which unfortunately breaks the old macro system.
Compiler Additions Compiler Additions
@ -42,6 +47,8 @@ Compiler Additions
- The compiler can detect and evaluate calls that can be evaluated at compile - The compiler can detect and evaluate calls that can be evaluated at compile
time for optimization purposes with the ``--implicitStatic`` command line time for optimization purposes with the ``--implicitStatic`` command line
option or pragma. option or pragma.
- The compiler now generates marker procs that the GC can use instead of RTTI.
This speeds up the GC quite a bit.
Language Additions Language Additions