Merge branch 'new_spawn' of https://github.com/Araq/Nimrod into new_spawn

This commit is contained in:
Araq 2014-06-06 21:11:11 +02:00
commit 4220b1c81d
9 changed files with 305 additions and 212 deletions

View file

@ -1636,7 +1636,7 @@ proc genMagicExpr(p: BProc, e: PNode, d: var TLoc, op: TMagic) =
of mSlurp..mQuoteAst: of mSlurp..mQuoteAst:
localError(e.info, errXMustBeCompileTime, e.sons[0].sym.name.s) localError(e.info, errXMustBeCompileTime, e.sons[0].sym.name.s)
of mSpawn: of mSpawn:
let n = lowerings.wrapProcForSpawn(p.module.module, e[1], e.typ, nil, nil) let n = lowerings.wrapProcForSpawn(p.module.module, e, e.typ, nil, nil)
expr(p, n, d) expr(p, n, d)
of mParallel: of mParallel:
let n = semparallel.liftParallel(p.module.module, e) let n = semparallel.liftParallel(p.module.module, e)

View file

@ -86,7 +86,7 @@ proc indirectAccess*(a: PNode, b: string, info: TLineInfo): PNode =
# returns a[].b as a node # returns a[].b as a node
var deref = newNodeI(nkHiddenDeref, info) var deref = newNodeI(nkHiddenDeref, info)
deref.typ = a.typ.skipTypes(abstractInst).sons[0] deref.typ = a.typ.skipTypes(abstractInst).sons[0]
var t = deref.typ var t = deref.typ.skipTypes(abstractInst)
var field: PSym var field: PSym
while true: while true:
assert t.kind == tyObject assert t.kind == tyObject
@ -94,6 +94,7 @@ proc indirectAccess*(a: PNode, b: string, info: TLineInfo): PNode =
if field != nil: break if field != nil: break
t = t.sons[0] t = t.sons[0]
if t == nil: break if t == nil: break
t = t.skipTypes(abstractInst)
assert field != nil, b assert field != nil, b
addSon(deref, a) addSon(deref, a)
result = newNodeI(nkDotExpr, info) result = newNodeI(nkDotExpr, info)
@ -132,28 +133,33 @@ proc callCodegenProc*(name: string, arg1: PNode;
if arg3 != nil: result.add arg3 if arg3 != nil: result.add arg3
result.typ = sym.typ.sons[0] result.typ = sym.typ.sons[0]
proc callProc(a: PNode): PNode =
result = newNodeI(nkCall, a.info)
result.add a
result.typ = a.typ.sons[0]
# we have 4 cases to consider: # we have 4 cases to consider:
# - a void proc --> nothing to do # - a void proc --> nothing to do
# - a proc returning GC'ed memory --> requires a promise # - a proc returning GC'ed memory --> requires a flowVar
# - a proc returning non GC'ed memory --> pass as hidden 'var' parameter # - a proc returning non GC'ed memory --> pass as hidden 'var' parameter
# - not in a parallel environment --> requires a promise for memory safety # - not in a parallel environment --> requires a flowVar for memory safety
type type
TSpawnResult = enum TSpawnResult = enum
srVoid, srPromise, srByVar srVoid, srFlowVar, srByVar
TPromiseKind = enum TFlowVarKind = enum
promInvalid # invalid type T for 'Promise[T]' fvInvalid # invalid type T for 'FlowVar[T]'
promGC # Promise of a GC'ed type fvGC # FlowVar of a GC'ed type
promBlob # Promise of a blob type fvBlob # FlowVar of a blob type
proc spawnResult(t: PType; inParallel: bool): TSpawnResult = proc spawnResult(t: PType; inParallel: bool): TSpawnResult =
if t.isEmptyType: srVoid if t.isEmptyType: srVoid
elif inParallel and not containsGarbageCollectedRef(t): srByVar elif inParallel and not containsGarbageCollectedRef(t): srByVar
else: srPromise else: srFlowVar
proc promiseKind(t: PType): TPromiseKind = proc flowVarKind(t: PType): TFlowVarKind =
if t.skipTypes(abstractInst).kind in {tyRef, tyString, tySequence}: promGC if t.skipTypes(abstractInst).kind in {tyRef, tyString, tySequence}: fvGC
elif containsGarbageCollectedRef(t): promInvalid elif containsGarbageCollectedRef(t): fvInvalid
else: promBlob else: fvBlob
proc addLocalVar(varSection: PNode; owner: PSym; typ: PType; v: PNode): PSym = proc addLocalVar(varSection: PNode; owner: PSym; typ: PType; v: PNode): PSym =
result = newSym(skTemp, getIdent(genPrefix), owner, varSection.info) result = newSym(skTemp, getIdent(genPrefix), owner, varSection.info)
@ -169,18 +175,18 @@ proc addLocalVar(varSection: PNode; owner: PSym; typ: PType; v: PNode): PSym =
discard """ discard """
We generate roughly this: We generate roughly this:
proc f_wrapper(args) = proc f_wrapper(thread, args) =
barrierEnter(args.barrier) # for parallel statement barrierEnter(args.barrier) # for parallel statement
var a = args.a # thread transfer; deepCopy or shallowCopy or no copy var a = args.a # thread transfer; deepCopy or shallowCopy or no copy
# depending on whether we're in a 'parallel' statement # depending on whether we're in a 'parallel' statement
var b = args.b var b = args.b
var fv = args.fv
args.prom = nimCreatePromise(thread, sizeof(T)) # optional fv.owner = thread # optional
nimPromiseCreateCondVar(args.prom) # optional
nimArgsPassingDone() # signal parent that the work is done nimArgsPassingDone() # signal parent that the work is done
# #
args.prom.blob = f(a, b, ...) args.fv.blob = f(a, b, ...)
nimPromiseSignal(args.prom) nimFlowVarSignal(args.fv)
# - or - # - or -
f(a, b, ...) f(a, b, ...)
@ -192,23 +198,12 @@ stmtList:
scratchObj.b = b scratchObj.b = b
nimSpawn(f_wrapper, addr scratchObj) nimSpawn(f_wrapper, addr scratchObj)
scratchObj.prom # optional scratchObj.fv # optional
""" """
proc createNimCreatePromiseCall(prom, threadParam: PNode): PNode =
let size = newNodeIT(nkCall, prom.info, getSysType(tyInt))
size.add newSymNode(createMagic("sizeof", mSizeOf))
assert prom.typ.kind == tyGenericInst
size.add newNodeIT(nkType, prom.info, prom.typ.sons[1])
let castExpr = newNodeIT(nkCast, prom.info, prom.typ)
castExpr.add emptyNode
castExpr.add callCodeGenProc("nimCreatePromise", threadParam, size)
result = castExpr
proc createWrapperProc(f: PNode; threadParam, argsParam: PSym; proc createWrapperProc(f: PNode; threadParam, argsParam: PSym;
varSection, call, barrier, prom: PNode; varSection, call, barrier, fv: PNode;
spawnKind: TSpawnResult): PSym = spawnKind: TSpawnResult): PSym =
var body = newNodeI(nkStmtList, f.info) var body = newNodeI(nkStmtList, f.info)
var threadLocalBarrier: PSym var threadLocalBarrier: PSym
@ -220,32 +215,32 @@ proc createWrapperProc(f: PNode; threadParam, argsParam: PSym;
body.add callCodeGenProc("barrierEnter", threadLocalBarrier.newSymNode) body.add callCodeGenProc("barrierEnter", threadLocalBarrier.newSymNode)
var threadLocalProm: PSym var threadLocalProm: PSym
if spawnKind == srByVar: if spawnKind == srByVar:
threadLocalProm = addLocalVar(varSection, argsParam.owner, prom.typ, prom) threadLocalProm = addLocalVar(varSection, argsParam.owner, fv.typ, fv)
elif prom != nil: elif fv != nil:
internalAssert prom.typ.kind == tyGenericInst internalAssert fv.typ.kind == tyGenericInst
threadLocalProm = addLocalVar(varSection, argsParam.owner, prom.typ, threadLocalProm = addLocalVar(varSection, argsParam.owner, fv.typ, fv)
createNimCreatePromiseCall(prom, threadParam.newSymNode))
body.add varSection body.add varSection
if prom != nil and spawnKind != srByVar: if fv != nil and spawnKind != srByVar:
body.add newFastAsgnStmt(prom, threadLocalProm.newSymNode) # generate:
if barrier == nil: # fv.owner = threadParam
body.add callCodeGenProc("nimPromiseCreateCondVar", prom) body.add newAsgnStmt(indirectAccess(threadLocalProm.newSymNode,
"owner", fv.info), threadParam.newSymNode)
body.add callCodeGenProc("nimArgsPassingDone", threadParam.newSymNode) body.add callCodeGenProc("nimArgsPassingDone", threadParam.newSymNode)
if spawnKind == srByVar: if spawnKind == srByVar:
body.add newAsgnStmt(genDeref(threadLocalProm.newSymNode), call) body.add newAsgnStmt(genDeref(threadLocalProm.newSymNode), call)
elif prom != nil: elif fv != nil:
let fk = prom.typ.sons[1].promiseKind let fk = fv.typ.sons[1].flowVarKind
if fk == promInvalid: if fk == fvInvalid:
localError(f.info, "cannot create a promise of type: " & localError(f.info, "cannot create a flowVar of type: " &
typeToString(prom.typ.sons[1])) typeToString(fv.typ.sons[1]))
body.add newAsgnStmt(indirectAccess(threadLocalProm.newSymNode, body.add newAsgnStmt(indirectAccess(threadLocalProm.newSymNode,
if fk == promGC: "data" else: "blob", prom.info), call) if fk == fvGC: "data" else: "blob", fv.info), call)
if barrier == nil: if barrier == nil:
# by now 'prom' is shared and thus might have beeen overwritten! we need # by now 'fv' is shared and thus might have beeen overwritten! we need
# to use the thread-local view instead: # to use the thread-local view instead:
body.add callCodeGenProc("nimPromiseSignal", threadLocalProm.newSymNode) body.add callCodeGenProc("nimFlowVarSignal", threadLocalProm.newSymNode)
else: else:
body.add call body.add call
if barrier != nil: if barrier != nil:
@ -404,16 +399,17 @@ proc setupArgsForParallelism(n: PNode; objType: PType; scratchObj: PSym;
indirectAccess(castExpr, field, n.info)) indirectAccess(castExpr, field, n.info))
call.add(threadLocal.newSymNode) call.add(threadLocal.newSymNode)
proc wrapProcForSpawn*(owner: PSym; n: PNode; retType: PType; proc wrapProcForSpawn*(owner: PSym; spawnExpr: PNode; retType: PType;
barrier, dest: PNode = nil): PNode = barrier, dest: PNode = nil): PNode =
# if 'barrier' != nil, then it is in a 'parallel' section and we # if 'barrier' != nil, then it is in a 'parallel' section and we
# generate quite different code # generate quite different code
let n = spawnExpr[1]
let spawnKind = spawnResult(retType, barrier!=nil) let spawnKind = spawnResult(retType, barrier!=nil)
case spawnKind case spawnKind
of srVoid: of srVoid:
internalAssert dest == nil internalAssert dest == nil
result = newNodeI(nkStmtList, n.info) result = newNodeI(nkStmtList, n.info)
of srPromise: of srFlowVar:
internalAssert dest == nil internalAssert dest == nil
result = newNodeIT(nkStmtListExpr, n.info, retType) result = newNodeIT(nkStmtListExpr, n.info, retType)
of srByVar: of srByVar:
@ -482,24 +478,29 @@ proc wrapProcForSpawn*(owner: PSym; n: PNode; retType: PType;
result.add newFastAsgnStmt(newDotExpr(scratchObj, field), barrier) result.add newFastAsgnStmt(newDotExpr(scratchObj, field), barrier)
barrierAsExpr = indirectAccess(castExpr, field, n.info) barrierAsExpr = indirectAccess(castExpr, field, n.info)
var promField, promAsExpr: PNode = nil var fvField, fvAsExpr: PNode = nil
if spawnKind == srPromise: if spawnKind == srFlowVar:
var field = newSym(skField, getIdent"prom", owner, n.info) var field = newSym(skField, getIdent"fv", owner, n.info)
field.typ = retType field.typ = retType
objType.addField(field) objType.addField(field)
promField = newDotExpr(scratchObj, field) fvField = newDotExpr(scratchObj, field)
promAsExpr = indirectAccess(castExpr, field, n.info) fvAsExpr = indirectAccess(castExpr, field, n.info)
# create flowVar:
result.add newFastAsgnStmt(fvField, callProc(spawnExpr[2]))
if barrier == nil:
result.add callCodeGenProc("nimFlowVarCreateCondVar", fvField)
elif spawnKind == srByVar: elif spawnKind == srByVar:
var field = newSym(skField, getIdent"prom", owner, n.info) var field = newSym(skField, getIdent"fv", owner, n.info)
field.typ = newType(tyPtr, objType.owner) field.typ = newType(tyPtr, objType.owner)
field.typ.rawAddSon(retType) field.typ.rawAddSon(retType)
objType.addField(field) objType.addField(field)
promAsExpr = indirectAccess(castExpr, field, n.info) fvAsExpr = indirectAccess(castExpr, field, n.info)
result.add newFastAsgnStmt(newDotExpr(scratchObj, field), genAddrOf(dest)) result.add newFastAsgnStmt(newDotExpr(scratchObj, field), genAddrOf(dest))
let wrapper = createWrapperProc(fn, threadParam, argsParam, varSection, call, let wrapper = createWrapperProc(fn, threadParam, argsParam, varSection, call,
barrierAsExpr, promAsExpr, spawnKind) barrierAsExpr, fvAsExpr, spawnKind)
result.add callCodeGenProc("nimSpawn", wrapper.newSymNode, result.add callCodeGenProc("nimSpawn", wrapper.newSymNode,
genAddrOf(scratchObj.newSymNode)) genAddrOf(scratchObj.newSymNode))
if spawnKind == srPromise: result.add promField if spawnKind == srFlowVar: result.add fvField

View file

@ -646,6 +646,7 @@ proc singlePragma(c: PContext, sym: PSym, n: PNode, i: int,
processDynLib(c, it, sym) processDynLib(c, it, sym)
of wCompilerproc: of wCompilerproc:
noVal(it) # compilerproc may not get a string! noVal(it) # compilerproc may not get a string!
if sfFromGeneric notin sym.flags:
makeExternExport(sym, "$1", it.info) makeExternExport(sym, "$1", it.info)
incl(sym.flags, sfCompilerProc) incl(sym.flags, sfCompilerProc)
incl(sym.flags, sfUsed) # suppress all those stupid warnings incl(sym.flags, sfUsed) # suppress all those stupid warnings

View file

@ -1585,12 +1585,22 @@ proc semShallowCopy(c: PContext, n: PNode, flags: TExprFlags): PNode =
else: else:
result = semDirectOp(c, n, flags) result = semDirectOp(c, n, flags)
proc createPromise(c: PContext; t: PType; info: TLineInfo): PType = proc createFlowVar(c: PContext; t: PType; info: TLineInfo): PType =
result = newType(tyGenericInvokation, c.module) result = newType(tyGenericInvokation, c.module)
addSonSkipIntLit(result, magicsys.getCompilerProc("Promise").typ) addSonSkipIntLit(result, magicsys.getCompilerProc("FlowVar").typ)
addSonSkipIntLit(result, t) addSonSkipIntLit(result, t)
result = instGenericContainer(c, info, result, allowMetaTypes = false) result = instGenericContainer(c, info, result, allowMetaTypes = false)
proc instantiateCreateFlowVarCall(c: PContext; t: PType;
info: TLineInfo): PSym =
let sym = magicsys.getCompilerProc("nimCreateFlowVar")
if sym == nil:
localError(info, errSystemNeeds, "nimCreateFlowVar")
var bindings: TIdTable
initIdTable(bindings)
bindings.idTablePut(sym.ast[genericParamsPos].sons[0].typ, t)
result = c.semGenerateInstance(c, sym, bindings, info)
proc setMs(n: PNode, s: PSym): PNode = proc setMs(n: PNode, s: PSym): PNode =
result = n result = n
n.sons[0] = newSymNode(s) n.sons[0] = newSymNode(s)
@ -1631,7 +1641,8 @@ proc semMagic(c: PContext, n: PNode, s: PSym, flags: TExprFlags): PNode =
if c.inParallelStmt > 0: if c.inParallelStmt > 0:
result.typ = result[1].typ result.typ = result[1].typ
else: else:
result.typ = createPromise(c, result[1].typ, n.info) result.typ = createFlowVar(c, result[1].typ, n.info)
result.add instantiateCreateFlowVarCall(c, result[1].typ, n.info).newSymNode
else: result = semDirectOp(c, n, flags) else: result = semDirectOp(c, n, flags)
proc semWhen(c: PContext, n: PNode, semCheck = true): PNode = proc semWhen(c: PContext, n: PNode, semCheck = true): PNode =

View file

@ -406,19 +406,19 @@ proc transformSpawn(owner: PSym; n, barrier: PNode): PNode =
if result.isNil: if result.isNil:
result = newNodeI(nkStmtList, n.info) result = newNodeI(nkStmtList, n.info)
result.add n result.add n
result.add wrapProcForSpawn(owner, m[1], b.typ, barrier, it[0]) result.add wrapProcForSpawn(owner, m, b.typ, barrier, it[0])
it.sons[it.len-1] = emptyNode it.sons[it.len-1] = emptyNode
if result.isNil: result = n if result.isNil: result = n
of nkAsgn, nkFastAsgn: of nkAsgn, nkFastAsgn:
let b = n[1] let b = n[1]
if getMagic(b) == mSpawn: if getMagic(b) == mSpawn:
let m = transformSlices(b) let m = transformSlices(b)
return wrapProcForSpawn(owner, m[1], b.typ, barrier, n[0]) return wrapProcForSpawn(owner, m, b.typ, barrier, n[0])
result = transformSpawnSons(owner, n, barrier) result = transformSpawnSons(owner, n, barrier)
of nkCallKinds: of nkCallKinds:
if getMagic(n) == mSpawn: if getMagic(n) == mSpawn:
result = transformSlices(n) result = transformSlices(n)
return wrapProcForSpawn(owner, result[1], n.typ, barrier, nil) return wrapProcForSpawn(owner, result, n.typ, barrier, nil)
result = transformSpawnSons(owner, n, barrier) result = transformSpawnSons(owner, n, barrier)
elif n.safeLen > 0: elif n.safeLen > 0:
result = transformSpawnSons(owner, n, barrier) result = transformSpawnSons(owner, n, barrier)

57
doc/spawn.txt Normal file
View file

@ -0,0 +1,57 @@
==========================================================
Parallel & Spawn
==========================================================
Nimrod has two flavors of parallelism:
1) `Structured`:idx parallelism via the ``parallel`` statement.
2) `Unstructured`:idx: parallelism via the standalone ``spawn`` statement.
Somewhat confusingly, ``spawn`` is also used in the ``parallel`` statement
with slightly different semantics. ``spawn`` always takes a call expression of
the form ``f(a, ...)``. Let ``T`` be ``f``'s return type. If ``T`` is ``void``
then ``spawn``'s return type is also ``void``. Within a ``parallel`` section
``spawn``'s return type is ``T``, otherwise it is ``FlowVar[T]``.
The compiler can ensure the location in ``location = spawn f(...)`` is not
read prematurely within a ``parallel`` section and so there is no need for
the overhead of an indirection via ``FlowVar[T]`` to ensure correctness.
Parallel statement
==================
The parallel statement is the preferred mechanism to introduce parallelism
in a Nimrod program. A subset of the Nimrod language is valid within a
``parallel`` section. This subset is checked to be free of data races at
compile time. A sophisticated `disjoint checker`:idx: ensures that no data
races are possible even though shared memory is extensively supported!
The subset is in fact the full language with the following
restrictions / changes:
* ``spawn`` within a ``parallel`` section has special semantics.
* Every location of the form ``a[i]`` and ``a[i..j]`` and ``dest`` where
``dest`` is part of the pattern ``dest = spawn f(...)`` has to be
provable disjoint. This is called the *disjoint check*.
* Every other complex location ``loc`` that is used in a spawned
proc (``spawn f(loc)``) has to immutable for the duration of
the ``parallel``. This is called the *immutability check*. Currently it
is not specified what exactly "complex location" means. We need to make that
an optimization!
* Every array access has to be provable within bounds.
* Slices are optimized so that no copy is performed. This optimization is not
yet performed for ordinary slices outside of a ``parallel`` section.
Spawn statement
===============
A standalone ``spawn`` statement is a simple construct. It executes
the passed expression on the thread pool and returns a `data flow variable`:idx:
``FlowVar[T]`` that can be read from. The reading with the ``^`` operator is
**blocking**. However, one can use ``awaitAny`` to wait on multiple flow variables
at the same time.
Like the ``parallel`` statement data flow variables ensure that no data races
are possible.

View file

@ -40,31 +40,42 @@ proc signal(cv: var CondVar) =
release(cv.L) release(cv.L)
signal(cv.c) signal(cv.c)
const CacheLineSize = 32 # true for most archs
type type
Barrier* {.compilerProc.} = object Barrier {.compilerProc.} = object
entered: int entered: int
cv: CondVar cv: CondVar # condvar takes 3 words at least
cacheAlign: array[0..20, byte] # ensure 'left' is not on the same when sizeof(int) < 8:
# cache line as 'entered' cacheAlign: array[CacheLineSize-4*sizeof(int), byte]
left: int left: int
cacheAlign2: array[CacheLineSize-sizeof(int), byte]
interest: bool ## wether the master is interested in the "all done" event
proc barrierEnter*(b: ptr Barrier) {.compilerProc.} = proc barrierEnter(b: ptr Barrier) {.compilerProc, inline.} =
atomicInc b.entered # due to the signaling between threads, it is ensured we are the only
# one with access to 'entered' so we don't need 'atomicInc' here:
inc b.entered
# also we need no 'fence' instructions here as soon 'nimArgsPassingDone'
# will be called which already will perform a fence for us.
proc barrierLeave*(b: ptr Barrier) {.compilerProc.} = proc barrierLeave(b: ptr Barrier) {.compilerProc, inline.} =
atomicInc b.left atomicInc b.left
# these can only be equal if 'closeBarrier' already signaled its interest when not defined(x86): fence()
# in this event: if b.interest and b.left == b.entered: signal(b.cv)
if b.left == b.entered: signal(b.cv)
proc openBarrier*(b: ptr Barrier) {.compilerProc.} = proc openBarrier(b: ptr Barrier) {.compilerProc, inline.} =
b.entered = 0 b.entered = 0
b.cv = createCondVar() b.left = 0
b.left = -1 b.interest = false
proc closeBarrier*(b: ptr Barrier) {.compilerProc.} = proc closeBarrier(b: ptr Barrier) {.compilerProc.} =
# signal interest in the "all done" event: fence()
atomicInc b.left if b.left != b.entered:
b.cv = createCondVar()
fence()
b.interest = true
fence()
while b.left != b.entered: await(b.cv) while b.left != b.entered: await(b.cv)
destroyCondVar(b.cv) destroyCondVar(b.cv)
@ -79,25 +90,26 @@ type
cv: CondVar cv: CondVar
idx: int idx: int
RawPromise* = ptr RawPromiseObj ## untyped base class for 'Promise[T]' RawFlowVar* = ref RawFlowVarObj ## untyped base class for 'FlowVar[T]'
RawPromiseObj {.inheritable.} = object # \ RawFlowVarObj = object of TObject
# we allocate this with the thread local allocator; this
# is possible since we already need to do the GC_unref
# on the owning thread
ready, usesCondVar: bool ready, usesCondVar: bool
cv: CondVar #\ cv: CondVar #\
# for 'awaitAny' support # for 'awaitAny' support
ai: ptr AwaitInfo ai: ptr AwaitInfo
idx: int idx: int
data: PObject # we incRef and unref it to keep it alive data: pointer # we incRef and unref it to keep it alive
owner: ptr Worker owner: pointer # ptr Worker
next: RawPromise
align: float64 # a float for proper alignment
Promise* {.compilerProc.} [T] = ptr object of RawPromiseObj FlowVarObj[T] = object of RawFlowVarObj
blob: T ## the underlying value, if available. Note that usually blob: T
## you should not access this field directly! However it can
## sometimes be more efficient than getting the value via ``^``. FlowVar*{.compilerProc.}[T] = ref FlowVarObj[T] ## a data flow variable
ToFreeQueue = object
len: int
lock: TLock
empty: TCond
data: array[512, pointer]
WorkerProc = proc (thread, args: pointer) {.nimcall, gcsafe.} WorkerProc = proc (thread, args: pointer) {.nimcall, gcsafe.}
Worker = object Worker = object
@ -109,109 +121,117 @@ type
ready: bool # put it here for correct alignment! ready: bool # put it here for correct alignment!
initialized: bool # whether it has even been initialized initialized: bool # whether it has even been initialized
shutdown: bool # the pool requests to shut down this worker thread shutdown: bool # the pool requests to shut down this worker thread
promiseLock: TLock q: ToFreeQueue
head: RawPromise
proc finished*(prom: RawPromise) = proc await*(fv: RawFlowVar) =
## This MUST be called for every created promise to free its associated ## waits until the value for the flowVar arrives. Usually it is not necessary
## resources. Note that the default reading operation ``^`` is destructive ## to call this explicitly.
## and calls ``finished``. if fv.usesCondVar:
doAssert prom.ai.isNil, "promise is still attached to an 'awaitAny'" fv.usesCondVar = false
assert prom.next == nil await(fv.cv)
let w = prom.owner destroyCondVar(fv.cv)
acquire(w.promiseLock)
prom.next = w.head
w.head = prom
release(w.promiseLock)
proc cleanPromises(w: ptr Worker) = proc finished(fv: RawFlowVar) =
var it = w.head doAssert fv.ai.isNil, "flowVar is still attached to an 'awaitAny'"
acquire(w.promiseLock) # we have to protect against the rare cases where the owner of the flowVar
while it != nil: # simply disregards the flowVar and yet the "flowVarr" has not yet written
let nxt = it.next # anything to it:
if it.usesCondVar: destroyCondVar(it.cv) await(fv)
if it.data != nil: GC_unref(it.data) if fv.data.isNil: return
dealloc(it) let owner = cast[ptr Worker](fv.owner)
it = nxt let q = addr(owner.q)
w.head = nil var waited = false
release(w.promiseLock) while true:
acquire(q.lock)
if q.len < q.data.len:
q.data[q.len] = fv.data
inc q.len
release(q.lock)
break
else:
# the queue is exhausted! We block until it has been cleaned:
release(q.lock)
wait(q.empty, q.lock)
waited = true
fv.data = nil
# wakeup other potentially waiting threads:
if waited: signal(q.empty)
proc nimCreatePromise(owner: pointer; blobSize: int): RawPromise {. proc cleanFlowVars(w: ptr Worker) =
compilerProc.} = let q = addr(w.q)
result = cast[RawPromise](alloc0(RawPromiseObj.sizeof + blobSize)) acquire(q.lock)
result.owner = cast[ptr Worker](owner) for i in 0 .. <q.len:
GC_unref(cast[PObject](q.data[i]))
q.len = 0
release(q.lock)
signal(q.empty)
proc nimPromiseCreateCondVar(prom: RawPromise) {.compilerProc.} = proc fvFinalizer[T](fv: FlowVar[T]) = finished(fv)
prom.cv = createCondVar()
prom.usesCondVar = true
proc nimPromiseSignal(prom: RawPromise) {.compilerProc.} = proc nimCreateFlowVar[T](): FlowVar[T] {.compilerProc.} =
if prom.ai != nil: new(result, fvFinalizer)
acquire(prom.ai.cv.L)
prom.ai.idx = prom.idx
inc prom.ai.cv.counter
release(prom.ai.cv.L)
signal(prom.ai.cv.c)
if prom.usesCondVar: signal(prom.cv)
proc await*[T](prom: Promise[T]) = proc nimFlowVarCreateCondVar(fv: RawFlowVar) {.compilerProc.} =
## waits until the value for the promise arrives. fv.cv = createCondVar()
if prom.usesCondVar: await(prom.cv) fv.usesCondVar = true
proc awaitAndThen*[T](prom: Promise[T]; action: proc (x: T) {.closure.}) = proc nimFlowVarSignal(fv: RawFlowVar) {.compilerProc.} =
## blocks until the ``prom`` is available and then passes its value if fv.ai != nil:
acquire(fv.ai.cv.L)
fv.ai.idx = fv.idx
inc fv.ai.cv.counter
release(fv.ai.cv.L)
signal(fv.ai.cv.c)
if fv.usesCondVar: signal(fv.cv)
proc awaitAndThen*[T](fv: FlowVar[T]; action: proc (x: T) {.closure.}) =
## blocks until the ``fv`` is available and then passes its value
## to ``action``. Note that due to Nimrod's parameter passing semantics this ## to ``action``. Note that due to Nimrod's parameter passing semantics this
## means that ``T`` doesn't need to be copied and so ``awaitAndThen`` can ## means that ``T`` doesn't need to be copied and so ``awaitAndThen`` can
## sometimes be more efficient than ``^``. ## sometimes be more efficient than ``^``.
if prom.usesCondVar: await(prom) await(fv)
when T is string or T is seq: when T is string or T is seq:
action(cast[T](prom.data)) action(cast[T](fv.data))
elif T is ref: elif T is ref:
{.error: "'awaitAndThen' not available for Promise[ref]".} {.error: "'awaitAndThen' not available for FlowVar[ref]".}
else: else:
action(prom.blob) action(fv.blob)
finished(prom) finished(fv)
proc `^`*[T](prom: Promise[ref T]): foreign ptr T = proc `^`*[T](fv: FlowVar[ref T]): foreign ptr T =
## blocks until the value is available and then returns this value. Note ## blocks until the value is available and then returns this value.
## this reading is destructive for reasons of efficiency and convenience. await(fv)
## This calls ``finished(prom)``. result = cast[foreign ptr T](fv.data)
if prom.usesCondVar: await(prom)
result = cast[foreign ptr T](prom.data)
finished(prom)
proc `^`*[T](prom: Promise[T]): T = proc `^`*[T](fv: FlowVar[T]): T =
## blocks until the value is available and then returns this value. Note ## blocks until the value is available and then returns this value.
## this reading is destructive for reasons of efficiency and convenience. await(fv)
## This calls ``finished(prom)``.
if prom.usesCondVar: await(prom)
when T is string or T is seq: when T is string or T is seq:
result = cast[T](prom.data) result = cast[T](fv.data)
else: else:
result = prom.blob result = fv.blob
finished(prom)
proc awaitAny*(promises: openArray[RawPromise]): int = proc awaitAny*(flowVars: openArray[RawFlowVar]): int =
# awaits any of the given promises. Returns the index of one promise for which ## awaits any of the given flowVars. Returns the index of one flowVar for
## a value arrived. A promise only supports one call to 'awaitAny' at the ## which a value arrived. A flowVar only supports one call to 'awaitAny' at
## same time. That means if you await([a,b]) and await([b,c]) the second ## the same time. That means if you await([a,b]) and await([b,c]) the second
## call will only await 'c'. If there is no promise left to be able to wait ## call will only await 'c'. If there is no flowVar left to be able to wait
## on, -1 is returned. ## on, -1 is returned.
## **Note**: This results in non-deterministic behaviour and so should be ## **Note**: This results in non-deterministic behaviour and so should be
## avoided. ## avoided.
var ai: AwaitInfo var ai: AwaitInfo
ai.cv = createCondVar() ai.cv = createCondVar()
var conflicts = 0 var conflicts = 0
for i in 0 .. promises.high: for i in 0 .. flowVars.high:
if cas(addr promises[i].ai, nil, addr ai): if cas(addr flowVars[i].ai, nil, addr ai):
promises[i].idx = i flowVars[i].idx = i
else: else:
inc conflicts inc conflicts
if conflicts < promises.len: if conflicts < flowVars.len:
await(ai.cv) await(ai.cv)
result = ai.idx result = ai.idx
for i in 0 .. promises.high: for i in 0 .. flowVars.high:
discard cas(addr promises[i].ai, addr ai, nil) discard cas(addr flowVars[i].ai, addr ai, nil)
else: else:
result = -1 result = -1
destroyCondVar(ai.cv) destroyCondVar(ai.cv)
@ -239,7 +259,7 @@ proc slave(w: ptr Worker) {.thread.} =
await(w.taskArrived) await(w.taskArrived)
assert(not w.ready) assert(not w.ready)
w.f(w, w.data) w.f(w, w.data)
if w.head != nil: w.cleanPromises if w.q.len != 0: w.cleanFlowVars
if w.shutdown: if w.shutdown:
w.shutdown = false w.shutdown = false
atomicDec currentPoolSize atomicDec currentPoolSize
@ -260,8 +280,9 @@ var
proc activateThread(i: int) {.noinline.} = proc activateThread(i: int) {.noinline.} =
workersData[i].taskArrived = createCondVar() workersData[i].taskArrived = createCondVar()
workersData[i].taskStarted = createCondVar() workersData[i].taskStarted = createCondVar()
initLock workersData[i].promiseLock
workersData[i].initialized = true workersData[i].initialized = true
initCond(workersData[i].q.empty)
initLock(workersData[i].q.lock)
createThread(workers[i], slave, addr(workersData[i])) createThread(workers[i], slave, addr(workersData[i]))
proc setup() = proc setup() =
@ -278,14 +299,16 @@ proc preferSpawn*(): bool =
proc spawn*(call: expr): expr {.magic: "Spawn".} proc spawn*(call: expr): expr {.magic: "Spawn".}
## always spawns a new task, so that the 'call' is never executed on ## always spawns a new task, so that the 'call' is never executed on
## the calling thread. 'call' has to be proc call 'p(...)' where 'p' ## the calling thread. 'call' has to be proc call 'p(...)' where 'p'
## is gcsafe and has 'void' as the return type. ## is gcsafe and has a return type that is either 'void' or compatible
## with ``FlowVar[T]``.
template spawnX*(call: expr): expr = template spawnX*(call: expr): expr =
## spawns a new task if a CPU core is ready, otherwise executes the ## spawns a new task if a CPU core is ready, otherwise executes the
## call in the calling thread. Usually it is advised to ## call in the calling thread. Usually it is advised to
## use 'spawn' in order to not block the producer for an unknown ## use 'spawn' in order to not block the producer for an unknown
## amount of time. 'call' has to be proc call 'p(...)' where 'p' ## amount of time. 'call' has to be proc call 'p(...)' where 'p'
## is gcsafe and has 'void' as the return type. ## is gcsafe and has a return type that is either 'void' or compatible
## with ``FlowVar[T]``.
(if preferSpawn(): spawn call else: call) (if preferSpawn(): spawn call else: call)
proc parallel*(body: stmt) {.magic: "Parallel".} proc parallel*(body: stmt) {.magic: "Parallel".}

View file

@ -10,7 +10,9 @@
## Atomic operations for Nimrod. ## Atomic operations for Nimrod.
{.push stackTrace:off.} {.push stackTrace:off.}
when (defined(gcc) or defined(llvm_gcc)) and hasThreadSupport: const someGcc = defined(gcc) or defined(llvm_gcc) or defined(clang)
when someGcc and hasThreadSupport:
type type
AtomMemModel* = enum AtomMemModel* = enum
ATOMIC_RELAXED, ## No barriers or synchronization. ATOMIC_RELAXED, ## No barriers or synchronization.
@ -153,41 +155,16 @@ when (defined(gcc) or defined(llvm_gcc)) and hasThreadSupport:
## A value of 0 indicates typical alignment should be used. The compiler may also ## A value of 0 indicates typical alignment should be used. The compiler may also
## ignore this parameter. ## ignore this parameter.
template fence*() = atomicThreadFence(ATOMIC_SEQ_CST)
elif defined(vcc) and hasThreadSupport: elif defined(vcc) and hasThreadSupport:
proc addAndFetch*(p: ptr int, val: int): int {. proc addAndFetch*(p: ptr int, val: int): int {.
importc: "NimXadd", nodecl.} importc: "NimXadd", nodecl.}
else: else:
proc addAndFetch*(p: ptr int, val: int): int {.inline.} = proc addAndFetch*(p: ptr int, val: int): int {.inline.} =
inc(p[], val) inc(p[], val)
result = p[] result = p[]
# atomic compare and swap (CAS) funcitons to implement lock-free algorithms
#if defined(windows) and not defined(gcc) and hasThreadSupport:
# proc InterlockedCompareExchangePointer(mem: ptr pointer,
# newValue: pointer, comparand: pointer) : pointer {.nodecl,
# importc: "InterlockedCompareExchangePointer", header:"windows.h".}
# proc compareAndSwap*[T](mem: ptr T,
# expected: T, newValue: T): bool {.inline.}=
# ## Returns true if successfully set value at mem to newValue when value
# ## at mem == expected
# return InterlockedCompareExchangePointer(addr(mem),
# addr(newValue), addr(expected))[] == expected
#elif not hasThreadSupport:
# proc compareAndSwap*[T](mem: ptr T,
# expected: T, newValue: T): bool {.inline.} =
# ## Returns true if successfully set value at mem to newValue when value
# ## at mem == expected
# var oldval = mem[]
# if oldval == expected:
# mem[] = newValue
# return true
# return false
# Some convenient functions
proc atomicInc*(memLoc: var int, x: int = 1): int = proc atomicInc*(memLoc: var int, x: int = 1): int =
when defined(gcc) and hasThreadSupport: when defined(gcc) and hasThreadSupport:
result = atomic_add_fetch(memLoc.addr, x, ATOMIC_RELAXED) result = atomic_add_fetch(memLoc.addr, x, ATOMIC_RELAXED)
@ -205,7 +182,7 @@ proc atomicDec*(memLoc: var int, x: int = 1): int =
dec(memLoc, x) dec(memLoc, x)
result = memLoc result = memLoc
when defined(windows) and not defined(gcc): when defined(windows) and not someGcc:
proc interlockedCompareExchange(p: pointer; exchange, comparand: int32): int32 proc interlockedCompareExchange(p: pointer; exchange, comparand: int32): int32
{.importc: "InterlockedCompareExchange", header: "<windows.h>", cdecl.} {.importc: "InterlockedCompareExchange", header: "<windows.h>", cdecl.}
@ -219,7 +196,7 @@ else:
# XXX is this valid for 'int'? # XXX is this valid for 'int'?
when (defined(x86) or defined(amd64)) and defined(gcc): when (defined(x86) or defined(amd64)) and (defined(gcc) or defined(llvm_gcc)):
proc cpuRelax {.inline.} = proc cpuRelax {.inline.} =
{.emit: """asm volatile("pause" ::: "memory");""".} {.emit: """asm volatile("pause" ::: "memory");""".}
elif (defined(x86) or defined(amd64)) and defined(vcc): elif (defined(x86) or defined(amd64)) and defined(vcc):
@ -231,4 +208,10 @@ elif false:
proc cpuRelax {.inline.} = os.sleep(1) proc cpuRelax {.inline.} = os.sleep(1)
when not defined(fence) and hasThreadSupport:
# XXX fixme
proc fence*() {.inline.} =
var dummy: bool
discard cas(addr dummy, false, true)
{.pop.} {.pop.}

View file

@ -1,6 +1,23 @@
version 0.9.6 version 0.9.6
============= =============
Concurrency
-----------
- document the new 'spawn' and 'parallel' statements
- implement 'deepCopy' builtin
- implement 'foo[1..4] = spawn(f[4..7])'
- the disjoint checker needs to deal with 'a = spawn f(); g = spawn f()'
- support for exception propagation
- Minor: The copying of the 'ref Promise' into the thead local storage only
happens to work due to the write barrier's implementation
- 'gcsafe' inferrence needs to be fixed
- implement lock levels --> first without the more complex race avoidance
Misc
----
- fix the bug that keeps 'defer' template from working - fix the bug that keeps 'defer' template from working
- make '--implicitStatic:on' the default - make '--implicitStatic:on' the default
- fix the tuple unpacking in lambda bug - fix the tuple unpacking in lambda bug