reorganize

This commit is contained in:
Siu Kwan Lam 2012-08-13 15:53:09 -07:00
commit f9d7195413
25 changed files with 0 additions and 2535 deletions

View file

@ -1,23 +0,0 @@
llvm_cbuilder
-------------
A llvm-py Builder wrapper for writing in slightly higher-level constructs.
This is aiming for two usecases:
1. Emit LLVM code in a more human-readable way;
2. Writing low-level code that you can't do it properly/portably with C, e.g
template (generic), atomic operations, memory ordering...
Parallel Vectorize
------------------
`parallel_vectorize.py` implements a set of code generator that bases on
llvm_cbuilder to create specialized parallel ufunc.
See `test_parallel_vectorize_numpy*.py` for testing and demo of the code
with the new `numpy.fromfunc()`.

View file

@ -1,586 +0,0 @@
{
"metadata": {
"name": "llvm_cbuilder_intro"
},
"nbformat": 2,
"worksheets": [
{
"cells": [
{
"cell_type": "markdown",
"source": [
"llvm_cbuilder: A Quick Tour",
"---------------------------",
"",
"Writing code generation logic in pure llvmpy requires one to think like a compiler -- ",
"dealing with basic-blocks, branches, SSA, etc.",
"The resulting code looks like assembly code, with no apparent hint about the control-flow.",
"",
"llvm_cbuilder is originally designed to simplify the translation of C code into llvmpy.",
"It provides a simple API for mimicking C programming in Python.",
"With llvm_cbulder, one can write very low-level code, probably even lower-level than C.",
"At the same time, one can use if-else, loop, structures and a bit of OOP."
]
},
{
"cell_type": "markdown",
"source": [
"A Simple Example",
"----------------",
"",
"Let's see some action.",
"We will define a function that calculates the square of a double-precision float."
]
},
{
"cell_type": "code",
"collapsed": true,
"input": [
"from llvm.core import *",
"from llvm_cbuilder import *",
"import llvm_cbuilder.shortnames as C",
"",
"class Square(CDefinition):",
" # prototype: double square(double x)",
" _name_ = 'square' # function name",
" _retty_ = C.double",
" _argtys_ = [ ('x', C.double) ]",
" ",
" def body(self, x):",
" y = x * x # just write out the expression",
" self.ret(y)"
],
"language": "python",
"outputs": [],
"prompt_number": 1
},
{
"cell_type": "markdown",
"source": [
"Notice how numerical expressions can be written naturally."
]
},
{
"cell_type": "markdown",
"source": [
"Let's see the emitted code:"
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"m = Module.new('my_module')",
"llvm_square = Square()(m) # define square() in my_module",
"print(m)"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"; ModuleID = 'my_module'",
"",
"define double @square(double %x) {",
"decl:",
" %x1 = alloca double",
" br label %body",
"",
"body: ; preds = %decl",
" store double %x, double* %x1",
" %0 = load double* %x1",
" %1 = load double* %x1",
" %2 = fmul double %0, %1",
" ret double %2",
"}",
""
]
}
],
"prompt_number": 2
},
{
"cell_type": "markdown",
"source": [
"Let's generate a ctype function object to call `square()` in the Python code:"
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"exe = CExecutor(m)",
"square = exe.get_ctype_function(llvm_square, \"double, double\")",
"result = square(1.2)",
"print(result)"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"1.44"
]
}
],
"prompt_number": 3
},
{
"cell_type": "markdown",
"source": [
"Control Flow Constructs",
"-----------------------",
"",
"The main strength of llvm_cbuilder is the control-flow constructs. ",
"They use the python \"with\" statement to setup new code blocks to",
"contain different paths of the control flow.",
"",
"The following example demonstrates both the if-else and loop contructs:"
]
},
{
"cell_type": "code",
"collapsed": true,
"input": [
"class IsPrime(CDefinition):",
" # prototype int isprime(int x)",
" _name_ = 'isprime'",
" _retty_ = C.int",
" _argtys_ = [ ('x', C.int) ]",
" ",
" def body(self, x):",
" two = self.constant(C.int, 2)",
" true = one = self.constant(C.int, 1)",
" false = zero = self.constant(C.int, 0)",
" ",
" with self.ifelse( x <= two ) as ifelse:",
" with ifelse.then():",
" self.ret(true)",
" ",
" with self.ifelse( (x % two) == zero ) as ifelse:",
" with ifelse.then():",
" self.ret(false)",
" ",
" idx = self.var(C.int, 3, name='idx')",
" with self.loop() as loop:",
" with loop.condition() as setcond:",
" setcond( idx < x )",
" ",
" with loop.body():",
" with self.ifelse( (x % idx) == zero ) as ifelse:",
" with ifelse.then():",
" self.ret(false)",
" idx += two",
" self.ret(true)"
],
"language": "python",
"outputs": [],
"prompt_number": 4
},
{
"cell_type": "markdown",
"source": [
"The code above is quite verbose.",
"It is like writing in Pascal or Ada. ",
"But, it is still easier than writing in llvmpy directly.",
"Take a look at the generated LLVM IR below."
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"llvm_isprime = IsPrime()(m)",
"print(llvm_isprime)"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"",
"define i32 @isprime(i32 %x) {",
"decl:",
" %x1 = alloca i32",
" %idx = alloca i32",
" br label %body",
"",
"body: ; preds = %decl",
" store i32 %x, i32* %x1",
" %0 = load i32* %x1",
" %1 = icmp sle i32 %0, 2",
" br i1 %1, label %if.then, label %if.else",
"",
"if.then: ; preds = %body",
" ret i32 1",
"",
"if.else: ; preds = %body",
" br label %if.end",
"",
"if.end: ; preds = %if.else",
" %2 = load i32* %x1",
" %3 = srem i32 %2, 2",
" %4 = icmp eq i32 %3, 0",
" br i1 %4, label %if.then2, label %if.else3",
"",
"if.then2: ; preds = %if.end",
" ret i32 0",
"",
"if.else3: ; preds = %if.end",
" br label %if.end4",
"",
"if.end4: ; preds = %if.else3",
" store i32 3, i32* %idx",
" br label %loop.cond",
"",
"loop.cond: ; preds = %if.end7, %if.end4",
" %5 = load i32* %idx",
" %6 = load i32* %x1",
" %7 = icmp slt i32 %5, %6",
" br i1 %7, label %loop.body, label %loop.end",
"",
"loop.body: ; preds = %loop.cond",
" %8 = load i32* %x1",
" %9 = load i32* %idx",
" %10 = srem i32 %8, %9",
" %11 = icmp eq i32 %10, 0",
" br i1 %11, label %if.then5, label %if.else6",
"",
"loop.end: ; preds = %loop.cond",
" ret i32 1",
"",
"if.then5: ; preds = %loop.body",
" ret i32 0",
"",
"if.else6: ; preds = %loop.body",
" br label %if.end7",
"",
"if.end7: ; preds = %if.else6",
" %12 = load i32* %idx",
" %13 = add i32 %12, 2",
" store i32 %13, i32* %idx",
" br label %loop.cond",
"}",
""
]
}
],
"prompt_number": 5
},
{
"cell_type": "markdown",
"source": [
"We'll setup a ctype function object to try it out:"
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"isprime = exe.get_ctype_function(llvm_isprime, 'int, int')",
"prime_100 = filter(isprime, range(2, 100))",
"print(prime_100)"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"[2, 3, 5, 7, 11, 13, 17, 19, 23, 29, 31, 37, 41, 43, 47, 53, 59, 61, 67, 71, 73, 79, 83, 89, 97]"
]
}
],
"prompt_number": 6
},
{
"cell_type": "markdown",
"source": [
"Structures",
"----------",
"",
"It is possible to create structures in llvm_cbuilder. ",
"Beware that structures in LLVM are unlike those in C.",
"LLVM type system allows structures to be literal or identified.",
"Literal structures are equivalent iff they have the same elements.",
"Identified structures are equivalent iff they have the same name.",
"In llvm_cbuilder, all structures are, by default, literal types.",
"",
"Here's an example:"
]
},
{
"cell_type": "code",
"collapsed": true,
"input": [
"class Vector2D(CStruct):",
" _fields_ = [",
" ('x', C.float),",
" ('y', C.float),",
" ]"
],
"language": "python",
"outputs": [],
"prompt_number": 7
},
{
"cell_type": "markdown",
"source": [
"That's all you need for a structure. Very much like defining structures with ctypes.",
"",
"We can also bind methods to structures which inline code to the caller."
]
},
{
"cell_type": "code",
"collapsed": true,
"input": [
"class Vector2D(CStruct):",
" _fields_ = [",
" ('x', C.float),",
" ('y', C.float),",
" ]",
" ",
" def add(self, other):",
" self.x += other.x",
" self.y += other.y"
],
"language": "python",
"outputs": [],
"prompt_number": 8
},
{
"cell_type": "markdown",
"source": [
"Setup a a test function:"
]
},
{
"cell_type": "code",
"collapsed": true,
"input": [
"class TestVector(CDefinition):",
" _name_ = 'testvector'",
" _retty_ = C.float",
" _argtys_ = [('x', C.float), ",
" ('y', C.float)] ",
" ",
" def body(self, x, y):",
" va = self.var(Vector2D)",
" vb = self.var(Vector2D)",
" va.x.assign(self.constant(C.float, 10))",
" va.y.assign(self.constant(C.float, 20))",
" vb.x.assign(x)",
" vb.y.assign(y)",
" ",
" va.add(vb)",
" ",
" return self.ret(va.x * va.y)",
" "
],
"language": "python",
"outputs": [],
"prompt_number": 9
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"llvm_testvec = TestVector()(m)",
"testvec = exe.get_ctype_function(llvm_testvec, \"float, float, float\")",
"print(testvec(-8, 5)) # should output 50"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"50.0"
]
}
],
"prompt_number": 10
},
{
"cell_type": "markdown",
"source": [
"Generic Programming",
"-------------------",
"",
"llvm_cbuilder supports generic function (or template function for C++).",
"When a `CDefinition` defines the `specialize` class method,",
"its constructor will invoke `specialize`.",
"All parameters passed to the constructor are forwarded to `specialize`.",
"",
"The constructor of a generic CDefinition creates a new dynamic class with the original CDefinition subclass as the parent.",
"Thus, `specialize` can modify class attributes without affecting the parent."
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"class GenericSquare(CDefinition):",
" @classmethod",
" def specialize(cls, data_type):",
" cls._name_ = '.'.join(['square', str(data_type)]) # set function name",
" cls._retty_ = data_type # set return type",
" cls._argtys_ = [ ('x', data_type) ] # set argument type",
" ",
" def body(self, x):",
" y = x * x # just write out the expression",
" self.ret(y)",
" ",
"llvm_square_int = GenericSquare(data_type=C.int)(m)",
"llvm_square_float = GenericSquare(data_type=C.float)(m)",
"",
"print(llvm_square_int)",
"print(llvm_square_float)"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"",
"define i32 @square.i32(i32 %x) {",
"decl:",
" %x1 = alloca i32",
" br label %body",
"",
"body: ; preds = %decl",
" store i32 %x, i32* %x1",
" %0 = load i32* %x1",
" %1 = load i32* %x1",
" %2 = mul i32 %0, %1",
" ret i32 %2",
"}",
"",
"",
"define float @square.float(float %x) {",
"decl:",
" %x1 = alloca float",
" br label %body",
"",
"body: ; preds = %decl",
" store float %x, float* %x1",
" %0 = load float* %x1",
" %1 = load float* %x1",
" %2 = fmul float %0, %1",
" ret float %2",
"}",
""
]
}
],
"prompt_number": 11
},
{
"cell_type": "markdown",
"source": [
"External Functions",
"------------------",
"",
"`CExternal` is a convenient class to help accessing externally-defined functions.",
"",
"We define an external interface to `sqrtf()` in libm."
]
},
{
"cell_type": "code",
"collapsed": true,
"input": [
"class LibM(CExternal):",
" sqrtf = Type.function(C.float, [C.float])"
],
"language": "python",
"outputs": [],
"prompt_number": 12
},
{
"cell_type": "markdown",
"source": [
"All class attributes that are `llvm.core.FunctionType` in `CExternal` ",
"are converted to a `CFunc` instance.",
"`CFunc` instances are callable.",
"`CFunc.__call__` generates code that perform the corresponding LLVM function call.",
"",
"The following example demonstrates the usage of `CExternal` and `CFunc`."
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"class TestSqrtf(CDefinition):",
" _name_ = 'test_sqrtf'",
" _retty_ = C.float",
" _argtys_ = [('x', C.float)]",
" ",
" def body(self, x):",
" libm = LibM(self) # init the API",
" y = libm.sqrtf(x) # call sqrtf",
" self.ret(y)",
" ",
"llvm_sqrtf = TestSqrtf()(m)",
"sqrtf = exe.get_ctype_function(llvm_sqrtf, 'float, float')",
"print(sqrtf(144))"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"12.0"
]
}
],
"prompt_number": 13
},
{
"cell_type": "markdown",
"source": [
"Casting",
"-------",
"",
"Unlike C, llvm_cbuilder does not automatically cast variables.",
"It is up to the user to explicit cast types.",
"All binary operations require both operands to be the same type.",
"All values in llvm_cbuilder has a `cast(destty)` method.",
"",
"For example:"
]
},
{
"cell_type": "code",
"collapsed": true,
"input": [
"class CastExample(CDefinition):",
" _name_ = 'cast_example'",
" _retty_ = C.float",
" _argtys_ = [('x', C.int)]",
" ",
" def body(self, x):",
" y = x.cast(C.float) # cast integer x to float",
" self.ret(y)"
],
"language": "python",
"outputs": [],
"prompt_number": 14
},
{
"cell_type": "markdown",
"source": [
"More... (TODO)"
]
}
]
}
]
}

Binary file not shown.

View file

@ -1,251 +0,0 @@
{
"metadata": {
"name": "parallel_vectorize"
},
"nbformat": 2,
"worksheets": [
{
"cells": [
{
"cell_type": "markdown",
"source": [
"Parallel Vectorize",
"------------------",
"",
"The `parallel_vectorize.py` module contains a set of llvmpy code generators",
"for creating mulithreaded _ufunc_. ",
"It depends on the new `numpy.fromfunc` for turning arbitrary function pointers into _ufunc_.",
"",
"From LLVM Function",
"------------------",
"",
"The `parallel_vectorize_from_func` method generates multithreaded _ufunc_ from LLVM functions.",
"",
"First, we will implement a workload function:"
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"from llvm_cbuilder import *",
"from llvm_cbuilder import shortnames as C",
"from llvm.core import *",
"",
"# Implement a workload",
"class Square(CDefinition):",
" _name_ = 'square'",
" _retty_ = C.double # 1 output: double",
" _argtys_ = [('x', C.double)] # 1 input: double",
" ",
" def body(self, x):",
" self.ret(x * x)",
"",
"m = Module.new('my_module')",
"llvm_square = Square()(m) # Generate a llvm function",
"print(llvm_square) "
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"",
"define double @square(double %x) {",
"decl:",
" %x1 = alloca double",
" br label %body",
"",
"body: ; preds = %decl",
" store double %x, double* %x1",
" %0 = load double* %x1",
" %1 = load double* %x1",
" %2 = fmul double %0, %1",
" ret double %2",
"}",
""
]
}
],
"prompt_number": 1
},
{
"cell_type": "markdown",
"source": [
"Then, we will generate a _ufunc_ from `llvm_square`:"
]
},
{
"cell_type": "code",
"collapsed": true,
"input": [
"from llvm.ee import *",
"engine = EngineBuilder.new(m).create() # Generate JIT engine",
"",
"from parallel_vectorize import parallel_vectorize_from_func",
"ufunc_square = parallel_vectorize_from_func(llvm_square, engine) # Generate UFunc"
],
"language": "python",
"outputs": [],
"prompt_number": 2
},
{
"cell_type": "markdown",
"source": [
"We are ready to use `ufunc_square` as a regular _ufunc_."
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"import numpy as np",
"A = np.arange(10., dtype=np.double)",
"ufunc_square(A)"
],
"language": "python",
"outputs": [
{
"output_type": "pyout",
"prompt_number": 3,
"text": [
"array([ 0., 1., 4., 9., 16., 25., 36., 49., 64., 81.])"
]
}
],
"prompt_number": 3
},
{
"cell_type": "markdown",
"source": [
"Here's another example that uses three inputs:"
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"class SumOfThree(CDefinition):",
" _name_ = 'sum.of.three'",
" _retty_ = C.int",
" _argtys_ = [('x', C.int),",
" ('y', C.int),",
" ('z', C.int)]",
" def body(self, x, y, z):",
" self.ret( x + y + z )",
"",
"llvm_sum3 = SumOfThree()(m)",
"ufunc_sum3 = parallel_vectorize_from_func(llvm_sum3, engine)",
"A = np.arange(10, dtype=np.int32)",
"B = A * 10",
"C = A * 100",
"ufunc_sum3(A, B, C)"
],
"language": "python",
"outputs": [
{
"output_type": "pyout",
"prompt_number": 4,
"text": [
"array([ 0, 111, 222, 333, 444, 555, 666, 777, 888, 999], dtype=int32)"
]
}
],
"prompt_number": 4
},
{
"cell_type": "markdown",
"source": [
"* * *",
"",
"Internals",
"---------",
"",
"There are four functions behind each multithreaded _ufunc_.",
"",
"1. the workload function (user defined);",
"2. the thread worker function (`UFuncCoreGeneric`);",
"3. the thread manager function (`ParallelUFuncPlatform`);",
"4. the ufunc entry point function (`SpecializedParallelUFunc`).",
"",
"**UFuncCoreGeneric** specializes to a llvm function type.",
"**It currently understands simple builtin scalar types (integers, float, double) only as arguments and return-type for the workload function.**",
"It sends work items to the workload function and performs work-stealing when it has finished its own workqueue.",
"Work-stealing uses atomic compare-exchange (or CAS) instruction to acquire ownership of a workqueue.",
"Work-stealing is implemented in the `UFuncCore._do_work_stealing`.",
"It can be disabled on platform that does not support atomic operations.",
"",
"**ParallelUFuncPlatform** specializes to the maximum number of threads. ",
"It divides all works equally among all threads.",
"Each thread executes the function generated by `UFuncCoreGeneric` once.",
"",
"**SpecializedParallelUFunc** is the specialized _ufunc_ entry point for a specific combination of ",
"workload, UFuncCoreGeneric and ParallelUFuncPlatform.",
"",
"Here's an example that uses `SpecializedParallelUFunc` directly for the `SumOfThree` workload."
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"import parallel_vectorize as pv",
"# specialize",
"def_spuf = pv.SpecializedParallelUFunc(pv.ParallelUFuncPlatform(num_thread=2),",
" pv.UFuncCoreGeneric(llvm_sum3.type.pointee),",
" CFuncRef(llvm_sum3))",
"# define",
"spuf = def_spuf(m)",
"print(spuf.name)"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"specialized_parallel_ufunc_2_ufunc_worker.i32.i32.i32.i32_sum.of.three"
]
}
],
"prompt_number": 5
},
{
"cell_type": "markdown",
"source": [
"`CFuncRef` also accepts arbitrary function pointer as long as the function type is provided."
]
},
{
"cell_type": "code",
"collapsed": false,
"input": [
"# specialize",
"fnty = llvm_sum3.type.pointee",
"sum3ptr = engine.get_pointer_to_function(llvm_sum3)",
"print(\"as function pointer: %x\" % sum3ptr)",
"def_spuf = pv.SpecializedParallelUFunc(pv.ParallelUFuncPlatform(num_thread=2),",
" pv.UFuncCoreGeneric(fnty),",
" CFuncRef('sum3.as.ptr', fnty, sum3ptr)) # name, type, ptr",
"# define",
"spuf = def_spuf(m)",
"print(spuf.name)"
],
"language": "python",
"outputs": [
{
"output_type": "stream",
"stream": "stdout",
"text": [
"as function pointer: 7f0bfc090740",
"specialized_parallel_ufunc_2_ufunc_worker.i32.i32.i32.i32_sum3.as.ptr"
]
}
],
"prompt_number": 6
}
]
}
]
}

Binary file not shown.

View file

@ -1,461 +0,0 @@
'''
This file implements the code-generator for parallel-vectorize.
ParallelUFunc is the platform independent base class for generating
the thread dispatcher. This thread dispatcher launches threads
that execute the generated function of UFuncCore.
UFuncCore is subclassed to specialize for the input/output types.
The actual workload is invoked inside the function generated by UFuncCore.
UFuncCore also defines a work-stealing mechanism that allows idle threads
to steal works from other threads.
'''
from llvm.core import *
from llvm.passes import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import sys
class WorkQueue(CStruct):
'''structure for workqueue for parallel-ufunc.
'''
_fields_ = [
('next', C.intp), # next index of work item
('last', C.intp), # last index of work item (exlusive)
('lock', C.int), # for locking the workqueue
]
def Lock(self):
'''inline the lock procedure.
'''
with self.parent.loop() as loop:
with loop.condition() as setcond:
unlocked = self.parent.constant(self.lock.type, 0)
locked = self.parent.constant(self.lock.type, 1)
res = self.lock.reference().atomic_cmpxchg(unlocked, locked,
ordering='acquire')
setcond( res != unlocked )
with loop.body():
pass
def Unlock(self):
'''inline the unlock procedure.
'''
unlocked = self.parent.constant(self.lock.type, 0)
locked = self.parent.constant(self.lock.type, 1)
res = self.lock.reference().atomic_cmpxchg(locked, unlocked,
ordering='release')
with self.parent.ifelse( res != locked ) as ifelse:
with ifelse.then():
# This shall kill the program
self.parent.unreachable()
class ContextCommon(CStruct):
'''structure for thread-shared context information in parallel-ufunc.
'''
_fields_ = [
# loop ufunc args
('args', C.pointer(C.char_p)),
('dimensions', C.pointer(C.intp)),
('steps', C.pointer(C.intp)),
('data', C.void_p),
# specifics for work queues
('func', C.void_p),
('num_thread', C.int),
('workqueues', C.pointer(WorkQueue.llvm_type())),
]
class Context(CStruct):
'''structure for thread-specific context information in parallel-ufunc.
'''
_fields_ = [
('common', C.pointer(ContextCommon.llvm_type())),
('id', C.int),
('completed', C.intp),
]
class ParallelUFunc(CDefinition):
'''the generic parallel vectorize mechanism
Can be specialized to the maximum number of threads on the platform.
Platform dependent threading function is implemented in
def _dispatch_worker(self, worker, contexts, num_thread):
...
which should be implemented in subclass or mixin.
'''
_argtys_ = [
('func', C.void_p),
('worker', C.void_p),
('args', C.pointer(C.char_p)),
('dimensions', C.pointer(C.intp)),
('steps', C.pointer(C.intp)),
('data', C.void_p),
]
@classmethod
def specialize(cls, num_thread):
'''specialize to the maximum # of thread
'''
cls._name_ = 'parallel_ufunc_%d' % num_thread
cls.ThreadCount = num_thread
def body(self, func, worker, args, dimensions, steps, data):
# Setup variables
ThreadCount = self.ThreadCount
common = self.var(ContextCommon, name='common')
workqueues = self.array(WorkQueue, ThreadCount, name='workqueues')
contexts = self.array(Context, ThreadCount, name='contexts')
num_thread = self.var(C.int, ThreadCount, name='num_thread')
# Initialize ContextCommon
common.args.assign(args)
common.dimensions.assign(dimensions)
common.steps.assign(steps)
common.data.assign(data)
common.func.assign(func)
common.num_thread.assign(num_thread.cast(C.int))
common.workqueues.assign(workqueues.reference())
# Determine chunksize, initial count of work-items per thread.
# If total_work >= num_thread, equally divide the works.
# If total_work % num_thread != 0, the last thread does all remaining works.
# If total_work < num_thread, each thread does one work,
# and set num_thread to total_work
N = dimensions[0]
ChunkSize = self.var_copy(N / num_thread.cast(N.type))
ChunkSize_NULL = self.constant_null(ChunkSize.type)
with self.ifelse(ChunkSize == ChunkSize_NULL) as ifelse:
with ifelse.then():
ChunkSize.assign(self.constant(ChunkSize.type, 1))
num_thread.assign(N.cast(num_thread.type))
# Populate workqueue for all threads
self._populate_workqueues(workqueues, N, ChunkSize, num_thread)
# Populate contexts for all threads
self._populate_context(contexts, common, num_thread)
# Dispatch worker threads
self._dispatch_worker(worker, contexts, num_thread)
## DEBUG ONLY ##
# Check for race condition
if True:
total_completed = self.var(C.intp, 0, name='total_completed')
for t in range(ThreadCount):
cur_ctxt = contexts[t].as_struct(Context)
total_completed += cur_ctxt.completed
# self.debug(cur_ctxt.id, 'completed', cur_ctxt.completed)
with self.ifelse( total_completed == N ) as ifelse:
with ifelse.then():
# self.debug("All is well!")
pass # keep quite if all is well
with ifelse.otherwise():
self.debug("ERROR: race occurred! Trigger segfault")
self.unreachable()
# Return
self.ret()
def _populate_workqueues(self, workqueues, N, ChunkSize, num_thread):
'''loop over all threads and populate the workqueue for each of them.
'''
ONE = self.constant(num_thread.type, 1)
with self.for_range(num_thread) as (loop, i):
cur_wq = workqueues[i].as_struct(WorkQueue)
cur_wq.next.assign(i.cast(ChunkSize.type) * ChunkSize)
cur_wq.last.assign((i + ONE).cast(ChunkSize.type) * ChunkSize)
cur_wq.lock.assign(self.constant(C.int, 0))
# end loop
last_wq = workqueues[num_thread - ONE].as_struct(WorkQueue)
last_wq.last.assign(N)
def _populate_context(self, contexts, common, num_thread):
'''loop over all threads and populate contexts for each of them.
'''
ONE = self.constant(num_thread.type, 1)
with self.for_range(num_thread) as (loop, i):
cur_ctxt = contexts[i].as_struct(Context)
cur_ctxt.common.assign(common.reference())
cur_ctxt.id.assign(i)
cur_ctxt.completed.assign(
self.constant_null(cur_ctxt.completed.type))
class ParallelUFuncPosixMixin(object):
'''ParallelUFunc mixin that implements _dispatch_worker to use pthread.
'''
def _dispatch_worker(self, worker, contexts, num_thread):
api = PThreadAPI(self)
NULL = self.constant_null(C.void_p)
threads = self.array(api.pthread_t, num_thread, name='threads')
# self.debug("launch threads")
# TODO error handling
ONE = self.constant(num_thread.type, 1)
with self.for_range(num_thread) as (loop, i):
api.pthread_create(threads[i].reference(), NULL, worker,
contexts[i].reference().cast(C.void_p))
with self.for_range(num_thread) as (loop, i):
api.pthread_join(threads[i], NULL)
class UFuncCore(CDefinition):
'''core work of a ufunc worker thread
Subclass to implement UFuncCore._do_work
Generates the workqueue handling and work stealing and invoke
the work function for each work item.
'''
_name_ = 'ufunc_worker'
_argtys_ = [
('context', C.pointer(Context.llvm_type())),
]
def body(self, context):
context = context.as_struct(Context)
common = context.common.as_struct(ContextCommon)
tid = context.id
# self.debug("start thread", tid, "/", common.num_thread)
workqueue = common.workqueues[tid].as_struct(WorkQueue)
self._do_workqueue(common, workqueue, tid, context.completed)
self._do_work_stealing(common, tid, context.completed) # optional
self.ret()
def _do_workqueue(self, common, workqueue, tid, completed):
'''process local workqueue.
'''
ZERO = self.constant_null(C.int)
with self.forever() as loop:
workqueue.Lock()
# Critical section
item = self.var_copy(workqueue.next, name='item')
workqueue.next += self.constant(item.type, 1)
last = self.var_copy(workqueue.last, name='last')
# Release
workqueue.Unlock()
with self.ifelse( item >= last ) as ifelse:
with ifelse.then():
loop.break_loop()
self._do_work(common, item, tid)
completed += self.constant(completed.type, 1)
def _do_work_stealing(self, common, tid, completed):
'''steal work from other workqueues.
'''
# self.debug("start work stealing", tid)
steal_continue = self.var(C.int, 1)
STEAL_STOP = self.constant_null(steal_continue.type)
# Loop until all workqueues are done.
with self.loop() as loop:
with loop.condition() as setcond:
setcond( steal_continue != STEAL_STOP )
with loop.body():
steal_continue.assign(STEAL_STOP)
self._do_work_stealing_innerloop(common, steal_continue, tid,
completed)
def _do_work_stealing_innerloop(self, common, steal_continue, tid,
completed):
'''loop over all other threads and try to steal work.
'''
with self.for_range(common.num_thread) as (loop, i):
with self.ifelse( i != tid ) as ifelse:
with ifelse.then():
otherqueue = common.workqueues[i].as_struct(WorkQueue)
self._do_work_stealing_check(common, otherqueue,
steal_continue, tid,
completed)
def _do_work_stealing_check(self, common, otherqueue, steal_continue, tid,
completed):
'''check the workqueue for any remaining work and steal it.
'''
otherqueue.Lock()
# Acquired
ONE = self.constant(otherqueue.last.type, 1)
STEAL_CONTINUE = self.constant(steal_continue.type, 1)
with self.ifelse(otherqueue.next < otherqueue.last) as ifelse:
with ifelse.then():
otherqueue.last -= ONE
item = self.var_copy(otherqueue.last)
otherqueue.Unlock()
# Released
self._do_work(common, item, tid)
completed += self.constant(completed.type, 1)
# Mark incomplete thread
steal_continue.assign(STEAL_CONTINUE)
with ifelse.otherwise():
otherqueue.Unlock()
# Released
def _do_work(self, common, item, tid):
'''prepare to call the actual work function
Implementation depends on number and type of arguments.
'''
raise NotImplementedError
class SpecializedParallelUFunc(CDefinition):
'''a generic ufunc that wraps ParallelUFunc, UFuncCore and the workload
'''
_argtys_ = [
('args', C.pointer(C.char_p)),
('dimensions', C.pointer(C.intp)),
('steps', C.pointer(C.intp)),
('data', C.void_p),
]
def body(self, args, dimensions, steps, data,):
pufunc = self.depends(self.PUFuncDef)
core = self.depends(self.CoreDef)
func = self.depends(self.FuncDef)
to_void_p = lambda x: x.cast(C.void_p)
pufunc(to_void_p(func), to_void_p(core), args, dimensions, steps, data)
self.ret()
@classmethod
def specialize(cls, pufunc_def, core_def, func_def):
'''specialize to a combination of ParallelUFunc, UFuncCore and workload
'''
cls._name_ = 'specialized_%s_%s_%s'% (pufunc_def, core_def, func_def)
cls.PUFuncDef = pufunc_def
cls.CoreDef = core_def
cls.FuncDef = func_def
class PThreadAPI(CExternal):
'''external declaration of pthread API
'''
pthread_t = C.void_p
pthread_create = Type.function(C.int,
[C.pointer(pthread_t), # thread_t
C.void_p, # thread attr
C.void_p, # function
C.void_p]) # arg
pthread_join = Type.function(C.int, [C.void_p, C.void_p])
class UFuncCoreGeneric(UFuncCore):
'''A generic ufunc core worker from LLVM function type
'''
def _do_work(self, common, item, tid):
ufunc_type = Type.function(self.RETTY, self.ARGTYS)
ufunc_ptr = CFunc(self, common.func.cast(C.pointer(ufunc_type)).value)
get_offset = lambda B, S, T: B[item * S].reference().cast(C.pointer(T))
indata = []
for i, argty in enumerate(self.ARGTYS):
ptr = get_offset(common.args[i], common.steps[i], argty)
indata.append(ptr.load())
out_index = len(self.ARGTYS)
outptr = get_offset(common.args[out_index], common.steps[out_index],
self.RETTY)
res = ufunc_ptr(*indata)
outptr.store(res)
@classmethod
def specialize(cls, fntype):
'''specialize to a LLVM function type
fntype : a LLVM function type (llvm.core.FunctionType)
'''
cls._name_ = '.'.join([cls._name_] +
map(str, [fntype.return_type] + fntype.args))
cls.RETTY = fntype.return_type
cls.ARGTYS = tuple(fntype.args)
if sys.platform not in ['win32']:
class ParallelUFuncPlatform(ParallelUFunc, ParallelUFuncPosixMixin):
pass
else:
raise NotImplementedError("Threading for %s" % sys.platform)
def parallel_vectorize_from_func(lfunc, engine=None):
'''create ufunc from a llvm.core.Function
If engine is given, return a function object which can be called
from python. (This needs Jay's numpy.fromfunc).
Otherwise, return the specialized ufunc as a llvm.core.Function
'''
import multiprocessing
NUM_CPU = multiprocessing.cpu_count()
fntype = lfunc.type.pointee
def_spuf = SpecializedParallelUFunc(
ParallelUFuncPlatform(num_thread=NUM_CPU),
UFuncCoreGeneric(fntype),
CFuncRef(lfunc))
spuf = def_spuf(lfunc.module)
if engine is None:
return spuf
else:
import numpy as np
fptr = engine.get_pointer_to_function(spuf)
inct = len(fntype.args)
outct = 1
# TODO refactor
typemap = {
'i8' : np.int8,
'i16' : np.int16,
'i32' : np.int32,
'i64' : np.int64,
'float' : np.float32,
'double' : np.float64,
}
try:
ptr_t = long
except:
ptr_t = int
assert False, "Having check this yet"
get_typenum = lambda T:np.dtype(typemap[str(T)]).num
assert fntype.return_type != C.void
tys = list(map(get_typenum, list(fntype.args) + [fntype.return_type]))
# Becareful that fromfunc does not provide full error checking yet.
# If typenum is out-of-bound, we have nasty memory corruptions.
# For instance, -1 for typenum will cause segfault.
# If elements of type-list (2nd arg) is tuple instead,
# there will also memory corruption. (Seems like code rewrite.)
return np.fromfunc([ptr_t(fptr)], [tys], inct, outct, [None])

View file

@ -1,130 +0,0 @@
from parallel_vectorize import *
class Work_D_D(CDefinition):
_name_ = 'work_d_d'
_retty_ = C.double
_argtys_ = [
('inval', C.double),
]
def body(self, inval):
self.ret(inval / self.constant(inval.type, 2.345))
class UFuncCore_D_D(UFuncCore):
'''
Specialize UFuncCore for double input, double output.
'''
_name_ = UFuncCore._name_ + '_d_d'
def _do_work(self, common, item, tid):
ufunc_type = Type.function(C.double, [C.double])
ufunc_ptr = CFunc(self, common.func.cast(C.pointer(ufunc_type)).value)
inbase = common.args[0]
outbase = common.args[1]
instep = common.steps[0]
outstep = common.steps[1]
indata = inbase[item * instep].reference().cast(C.pointer(C.double))
outdata = outbase[item * outstep].reference().cast(C.pointer(C.double))
res = ufunc_ptr(indata.load())
outdata.store(res)
class ParallelUFuncPosix(ParallelUFunc, ParallelUFuncPosixMixin):
pass
class Tester(CDefinition):
'''
Generate test.
'''
_name_ = 'tester'
def body(self):
# depends
module = self.function.module
ThreadCount = 2
ArgCount = 2
WorkCount = 10000
spufdef = SpecializedParallelUFunc(ParallelUFuncPosix(num_thread=2),
UFuncCore_D_D(),
Work_D_D())
sppufunc = self.depends(spufdef)
# real work
NULL = self.constant_null(C.void_p)
args = self.array(C.char_p, 2, name='args')
args_double = []
for t in range(ThreadCount):
args_for_thread = self.array(C.double, WorkCount)
args[t].assign(args_for_thread.cast(C.char_p))
args_double.append(args_for_thread)
dims = self.array(C.intp, 1, name='dims')
dims[0].assign(self.constant(C.intp, WorkCount))
steps = self.array(C.intp, ArgCount, name='steps')
for c in range(ArgCount):
steps[c].assign(self.constant(C.intp, 8))
# populate data
inbase = args_double[0]
i = self.var(C.intp, 0)
with self.loop() as loop:
with loop.condition() as setcond:
setcond( i < self.constant(C.intp, WorkCount) )
with loop.body():
inbase[i].assign(i.cast(C.double))
i += self.constant(C.intp, 1)
# call parallel ufunc
sppufunc(args, dims, steps, NULL)
# check error
outbase = args_double[-1]
with self.for_range(self.constant(C.intp, WorkCount)) as (loop, i):
test = outbase[i] != (inbase[i] / self.constant(C.double, 2.345))
with self.ifelse( test ) as ifelse:
with ifelse.then():
self.debug("Invalid data at i =", i, outbase[i], inbase[i])
self.ret()
def main():
module = Module.new(__name__)
mpm = PassManager.new()
pmbuilder = PassManagerBuilder.new()
pmbuilder.opt_level = 3
pmbuilder.populate(mpm)
fntester = Tester.define(module)
# print(module)
module.verify()
mpm.run(module)
print('optimized'.center(80,'-'))
print(module)
# run
print('run')
exe = CExecutor(module)
func = exe.get_ctype_function(fntester, 'void')
func()
# Will not reach here is race condition occurred
print('Good')
if __name__ == '__main__':
main()

View file

@ -1,57 +0,0 @@
'''
Test parallel-vectorize with numpy.fromfunc.
Uses the work load from test_parallel_vectorize.
'''
from test_parallel_vectorize import *
import numpy as np
def main():
module = Module.new(__name__)
spufdef = SpecializedParallelUFunc(ParallelUFuncPosix(num_thread=2),
UFuncCore_D_D(),
Work_D_D())
sppufunc = spufdef(module)
module.verify()
mpm = PassManager.new()
pmbuilder = PassManagerBuilder.new()
pmbuilder.opt_level = 3
pmbuilder.populate(mpm)
mpm.run(module)
# print module
# run
exe = CExecutor(module)
funcptr = exe.engine.get_pointer_to_function(sppufunc)
print("Function pointer: %x" % funcptr)
ptr_t = long # py2 only
# Becareful that fromfunc does not provide full error checking yet.
# If typenum is out-of-bound, we have nasty memory corruptions.
# For instance, -1 for typenum will cause segfault.
# If elements of type-list (2nd arg) is tuple instead,
# there will also memory corruption. (Seems like code rewrite.)
typenum = np.dtype(np.double).num
ufunc = np.fromfunc([ptr_t(funcptr)], [[typenum, typenum]], 1, 1, [None])
x = np.linspace(0., 10., 1000)
x.dtype=np.double
# print x
ans = ufunc(x)
# print ans
if not ( ans == x/2.345 ).all():
raise ValueError('Computation failed')
else:
print('Good')
if __name__ == '__main__':
main()

View file

@ -1,69 +0,0 @@
'''
Test parallel-vectorize with numpy.fromfunc.
Uses the work load from test_parallel_vectorize.
This time we pass a function pointer.
'''
from test_parallel_vectorize import *
import numpy as np
def main():
module = Module.new(__name__)
exe = CExecutor(module)
workdef = Work_D_D()
workfunc = workdef(module)
# get pointer to workfunc
workfunc_ptr = exe.engine.get_pointer_to_function(workfunc)
workdecl = CFuncRef(workfunc.name, workfunc.type.pointee, workfunc_ptr)
spufdef = SpecializedParallelUFunc(ParallelUFuncPosix(num_thread=2),
UFuncCore_D_D(),
workdecl)
sppufunc = spufdef(module)
sppufunc.verify()
print(sppufunc)
module.verify()
mpm = PassManager.new()
pmbuilder = PassManagerBuilder.new()
pmbuilder.opt_level = 3
pmbuilder.populate(mpm)
mpm.run(module)
print(module)
# run
funcptr = exe.engine.get_pointer_to_function(sppufunc)
print("Function pointer: %x" % funcptr)
ptr_t = long # py2 only
# Becareful that fromfunc does not provide full error checking yet.
# If typenum is out-of-bound, we have nasty memory corruptions.
# For instance, -1 for typenum will cause segfault.
# If elements of type-list (2nd arg) is tuple instead,
# there will also memory corruption. (Seems like code rewrite.)
typenum = np.dtype(np.double).num
ufunc = np.fromfunc([ptr_t(funcptr)], [[typenum, typenum]], 1, 1, [None])
x = np.linspace(0., 10., 1000)
x.dtype=np.double
# print x
ans = ufunc(x)
# print ans
if not ( ans == x/2.345 ).all():
raise ValueError('Computation failed')
else:
print('Good')
if __name__ == '__main__':
main()

View file

@ -1,56 +0,0 @@
from parallel_vectorize import *
from llvm_cbuilder import shortnames as C
from llvm.core import *
import numpy as np
import unittest
from random import random
class OneOne(CDefinition):
def body(self, inval):
self.ret( (inval * inval).cast(self.OUT_TYPE) )
@classmethod
def specialize(cls, itype, otype):
cls._name_ = '.'.join(map(str, ['oneone', itype, otype]))
cls._retty_ = otype
cls._argtys_ = [
('inval', itype),
]
cls.OUT_TYPE = otype
class TestParallelVectorize(unittest.TestCase):
def test_parallelvectorize_d_d(self):
self.template(C.double, C.double)
def test_parallelvectorize_d_f(self):
self.template(C.double, C.float)
def template(self, itype, otype):
module = Module.new(__name__)
exe = CExecutor(module)
def_oneone = OneOne(itype, otype)
oneone = def_oneone(module)
ufunc = parallel_vectorize_from_func(oneone, exe.engine)
# print(module)
module.verify()
x = np.linspace(.0, 10., 1000)
x.dtype = np.double
ans = ufunc(x)
gold = x * x
for x, y in zip(ans, gold):
if y != 0:
err = abs(x - y)/y
self.assertLess(err, 1e-6)
else:
self.assertEqual(x, y)
if __name__ == '__main__':
unittest.main()

View file

@ -1,58 +0,0 @@
from parallel_vectorize import *
from llvm_cbuilder import shortnames as C
from llvm.core import *
import numpy as np
import unittest
from random import random
class TwoOne(CDefinition):
def body(self, a, b):
self.ret( (a * b).cast(self.OUT_TYPE) )
@classmethod
def specialize(cls, itype1, itype2, otype):
cls._name_ = '.'.join(map(str, ['oneone', itype1, itype2, otype]))
cls._retty_ = otype
cls._argtys_ = [
('a', itype1),
('b', itype2),
]
cls.OUT_TYPE = otype
class TestParallelVectorize(unittest.TestCase):
def test_parallelvectorize_dd_d(self):
self.template(C.double, C.double, C.double)
def test_parallelvectorize_dd_f(self):
self.template(C.double, C.double, C.float)
def template(self, itype1, itype2, otype):
module = Module.new(__name__)
exe = CExecutor(module)
def_twoone = TwoOne(itype1, itype2, otype)
twoone = def_twoone(module)
ufunc = parallel_vectorize_from_func(twoone, exe.engine)
# print(module)
module.verify()
A = np.linspace(.0, 10., 1000)
A.dtype = np.double
B = np.linspace(-10., 0., 1000)
B.dtype = np.double
ans = ufunc(A, B)
gold = A * B
for x, y in zip(ans, gold):
if y != 0:
err = abs(x - y)/y
self.assertLess(err, 1e-6)
else:
self.assertEqual(x, y)
if __name__ == '__main__':
unittest.main()

View file

@ -1,115 +0,0 @@
'''
Base on the test_pthread.py and extend to use atomic instructions
'''
from llvm.core import *
from llvm.passes import *
from llvm.ee import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import unittest, logging
# logging.basicConfig(level=logging.DEBUG)
NUM_OF_THREAD = 4
REPEAT = 10000
def gen_test_worker(mod):
cb = CBuilder.new_function(mod, 'worker', C.void, [C.pointer(C.int)])
pval = cb.args[0]
one = cb.constant(pval.type.pointee, 1)
ct = cb.var(C.int, 0)
limit = cb.constant(C.int, REPEAT)
with cb.loop() as loop:
with loop.condition() as setcond:
setcond( ct < limit )
with loop.body():
cb.atomic_add(pval, one, 'acq_rel')
ct += one
cb.ret()
cb.close()
return cb.function
def gen_test_pthread(mod):
cb = CBuilder.new_function(mod, 'manager', C.int, [C.int])
arg = cb.args[0]
worker_func = cb.get_function_named('worker')
pthread_create = cb.get_function_named('pthread_create')
pthread_join = cb.get_function_named('pthread_join')
NULL = cb.constant_null(C.void_p)
cast_to_null = lambda x: x.cast(C.void_p)
threads = cb.array(C.void_p, NUM_OF_THREAD)
for tid in range(NUM_OF_THREAD):
pthread_create_args = [threads[tid].reference(),
NULL,
worker_func,
arg.reference()]
pthread_create(*map(cast_to_null, pthread_create_args))
worker_func(arg.reference())
for tid in range(NUM_OF_THREAD):
pthread_join_args = threads[tid], NULL
pthread_join(*map(cast_to_null, pthread_join_args))
cb.ret(arg)
cb.close()
return cb.function
class TestAtomicAdd(unittest.TestCase):
def test_atomic_add(self):
mod = Module.new(__name__)
# add pthread functions
mod.add_function(Type.function(C.int,
[C.void_p, C.void_p, C.void_p, C.void_p]),
'pthread_create')
mod.add_function(Type.function(C.int,
[C.void_p, C.void_p]),
'pthread_join')
lf_test_worker = gen_test_worker(mod)
lf_test_pthread = gen_test_pthread(mod)
logging.debug(mod)
mod.verify()
# optimize
fpm = FunctionPassManager.new(mod)
mpm = PassManager.new()
pmb = PassManagerBuilder.new()
pmb.vectorize = True
pmb.opt_level = 3
pmb.populate(fpm)
pmb.populate(mpm)
fpm.run(lf_test_worker)
fpm.run(lf_test_pthread)
mpm.run(mod)
logging.debug(mod)
mod.verify()
# run
exe = CExecutor(mod)
exe.engine.get_pointer_to_function(mod.get_function_named('worker'))
func = exe.get_ctype_function(lf_test_pthread, 'int, int')
inarg = 1234
gold = inarg + (NUM_OF_THREAD + 1) * REPEAT
for _ in range(1000): # run many many times to catch race condition
self.assertEqual(func(inarg), gold, "Unexpected race condition")
if __name__ == '__main__':
unittest.main()

View file

@ -1,122 +0,0 @@
'''
Base on the test_pthread.py and extend to use atomic instructions
'''
from llvm.core import *
from llvm.passes import *
from llvm.ee import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import unittest, logging
# logging.basicConfig(level=logging.DEBUG)
NUM_OF_THREAD = 4
REPEAT = 10000
def gen_test_worker(mod):
cb = CBuilder.new_function(mod, 'worker', C.void, [C.pointer(C.int)])
pval = cb.args[0]
one = cb.constant(pval.type.pointee, 1)
ct = cb.var(C.int, 0)
limit = cb.constant(C.int, REPEAT)
with cb.loop() as loop:
with loop.condition() as setcond:
setcond( ct < limit )
with loop.body():
oldval = pval.atomic_load('acquire')
updated = oldval + one
castmp = pval.atomic_cmpxchg(oldval, updated, 'release')
with cb.ifelse( castmp == oldval ) as ifelse:
with ifelse.then():
ct += one
cb.ret()
cb.close()
return cb.function
def gen_test_pthread(mod):
cb = CBuilder.new_function(mod, 'manager', C.int, [C.int])
arg = cb.args[0]
worker_func = cb.get_function_named('worker')
pthread_create = cb.get_function_named('pthread_create')
pthread_join = cb.get_function_named('pthread_join')
NULL = cb.constant_null(C.void_p)
cast_to_null = lambda x: x.cast(C.void_p)
threads = cb.array(C.void_p, NUM_OF_THREAD)
for tid in range(NUM_OF_THREAD):
pthread_create_args = [threads[tid].reference(),
NULL,
worker_func,
arg.reference()]
pthread_create(*map(cast_to_null, pthread_create_args))
worker_func(arg.reference())
for tid in range(NUM_OF_THREAD):
pthread_join_args = threads[tid], NULL
pthread_join(*map(cast_to_null, pthread_join_args))
cb.ret(arg)
cb.close()
return cb.function
class TestAtomicCmpXchg(unittest.TestCase):
def test_atomic_cmpxchg(self):
mod = Module.new(__name__)
# add pthread functions
mod.add_function(Type.function(C.int,
[C.void_p, C.void_p, C.void_p, C.void_p]),
'pthread_create')
mod.add_function(Type.function(C.int,
[C.void_p, C.void_p]),
'pthread_join')
lf_test_worker = gen_test_worker(mod)
lf_test_pthread = gen_test_pthread(mod)
logging.debug(mod)
mod.verify()
# optimize
fpm = FunctionPassManager.new(mod)
mpm = PassManager.new()
pmb = PassManagerBuilder.new()
pmb.vectorize = True
pmb.opt_level = 3
pmb.populate(fpm)
pmb.populate(mpm)
fpm.run(lf_test_worker)
fpm.run(lf_test_pthread)
mpm.run(mod)
logging.debug(mod)
mod.verify()
# run
exe = CExecutor(mod)
exe.engine.get_pointer_to_function(mod.get_function_named('worker'))
func = exe.get_ctype_function(lf_test_pthread, 'int, int')
inarg = 1234
gold = inarg + (NUM_OF_THREAD + 1) * REPEAT
for _ in range(1000): # run many many times to catch race condition
res = func(inarg)
self.assertEqual(res, gold,
"Unexpected race condition: res = %d" % res)
if __name__ == '__main__':
unittest.main()

View file

@ -1,17 +0,0 @@
from llvm.core import *
from llvm_cbuilder import *
from llvm_cbuilder import shortnames as C
import unittest
class TestCstrCollide(unittest.TestCase):
def test_same_string(self):
mod = Module.new(__name__)
cb = CBuilder.new_function(mod, 'test_cstr_collide', C.void, [])
a = cb.constant_string("hello")
b = cb.constant_string("hello")
self.assertEqual(a.value, b.value)
if __name__ == '__main__':
unittest.main()

View file

@ -1,125 +0,0 @@
from llvm.core import *
from llvm.passes import *
from llvm.ee import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import unittest, logging
def is_prime(x):
if x <= 2:
return True
if (x % 2) == 0:
return False
for y in range(2, int(1 + x**0.5)):
if (x % y) == 0:
return False
return True
def gen_is_prime(mod):
functype = Type.function(C.int, [C.int])
func = mod.add_function(functype, 'isprime')
cb = CBuilder(func)
arg = cb.args[0]
two = cb.constant(C.int, 2)
true = one = cb.constant(C.int, 1)
false = zero = cb.constant(C.int, 0)
with cb.ifelse( arg <= two ) as ifelse:
with ifelse.then():
cb.ret(true)
with cb.ifelse( (arg % two) == zero ) as ifelse:
with ifelse.then():
cb.ret(false)
idx = cb.var(C.int, 3, name='idx')
with cb.loop() as loop:
with loop.condition() as setcond:
setcond( idx < arg )
with loop.body():
with cb.ifelse( (arg % idx) == zero ) as ifelse:
with ifelse.then():
cb.ret(false)
# increment
idx += two
cb.ret(true)
cb.close()
return func
def gen_is_prime_fast(mod):
functype = Type.function(C.int, [C.int])
func = mod.add_function(functype, 'isprime_fast')
cb = CBuilder(func)
arg = cb.args[0]
two = cb.constant(C.int, 2)
true = one = cb.constant(C.int, 1)
false = zero = cb.constant(C.int, 0)
with cb.ifelse( arg <= two ) as ifelse:
with ifelse.then():
cb.ret(true)
with cb.ifelse( (arg % two) == zero ) as ifelse:
with ifelse.then():
cb.ret(false)
idx = cb.var(C.int, 3, name='idx')
sqrt = cb.get_intrinsic(INTR_SQRT, [C.float])
looplimit = one + sqrt(arg.cast(C.float)).cast(C.int)
with cb.loop() as loop:
with loop.condition() as setcond:
setcond( idx < looplimit )
with loop.body():
with cb.ifelse( (arg % idx) == zero ) as ifelse:
with ifelse.then():
cb.ret(false)
# increment
idx += two
cb.ret(true)
cb.close()
return func
class TestIsPrime(unittest.TestCase):
def test_isprime(self):
mod = Module.new(__name__)
lf_isprime = gen_is_prime(mod)
logging.debug(mod)
mod.verify()
exe = CExecutor(mod)
func = exe.get_ctype_function(lf_isprime, 'bool, int')
for x in range(2, 1000):
msg = "Failed at x = %d" % x
self.assertEqual(func(x), is_prime(x), msg)
def test_isprime_fast(self):
mod = Module.new(__name__)
lf_isprime = gen_is_prime_fast(mod)
logging.debug(mod)
mod.verify()
exe = CExecutor(mod)
func = exe.get_ctype_function(lf_isprime, 'bool, int')
for x in range(2, 1000):
msg = "Failed at x = %d" % x
self.assertEqual(func(x), is_prime(x), msg)
if __name__ == '__main__':
unittest.main()

View file

@ -1,135 +0,0 @@
from llvm.core import *
from llvm.passes import *
from llvm.ee import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import unittest, logging
def loopbreak(d):
z = 0
for x in range(100):
for y in range(100):
z += x + y
if z > 50:
break
z -= d
return z
def gen_loopbreak(mod):
functype = Type.function(C.int, [C.int])
func = mod.add_function(functype, 'loopbreak')
cb = CBuilder(func)
d = cb.args[0]
x = cb.var(C.int)
y = cb.var(C.int)
z = cb.var(C.int)
one = cb.constant(C.int, 1)
zero = cb.constant(C.int, 0)
limit = cb.constant(C.int, 100)
fifty = cb.constant(C.int, 50)
z.assign(zero)
x.assign(zero)
with cb.loop() as outer:
with outer.condition() as setcond:
setcond( x < limit )
with outer.body():
y.assign(zero)
with cb.loop() as inner:
with inner.condition() as setcond:
setcond( y < limit )
with inner.body():
z += x + y
with cb.ifelse( z > fifty ) as ifelse:
with ifelse.then():
inner.break_loop()
y += one
z -= d
x += one
cb.ret(z)
cb.close()
return func
def loopcontinue(d):
z = 0
for x in range(100):
for y in range(100):
z += x + y
if z > 50:
continue
z += d
return z
def gen_loopcontinue(mod):
functype = Type.function(C.int, [C.int])
func = mod.add_function(functype, 'loopcontinue')
cb = CBuilder(func)
d = cb.args[0]
x = cb.var(C.int)
y = cb.var(C.int)
z = cb.var(C.int)
one = cb.constant(C.int, 1)
zero = cb.constant(C.int, 0)
limit = cb.constant(C.int, 100)
fifty = cb.constant(C.int, 50)
z.assign(zero)
x.assign(zero)
with cb.loop() as outer:
with outer.condition() as setcond:
setcond( x < limit )
with outer.body():
y.assign(zero)
with cb.loop() as inner:
with inner.condition() as setcond:
setcond( y < limit )
with inner.body():
z += x + y
y += one
with cb.ifelse( z > fifty ) as ifelse:
with ifelse.then():
inner.continue_loop()
z += d
x += one
cb.ret(z)
cb.close()
return func
class TestLoopControl(unittest.TestCase):
def test_loopbreak(self):
mod = Module.new(__name__)
lfunc = gen_loopbreak(mod)
logging.debug(mod)
mod.verify()
exe = CExecutor(mod)
func = exe.get_ctype_function(lfunc, 'int, int')
for x in range(100):
self.assertEqual(func(x), loopbreak(x))
def test_loopcontinue(self):
mod = Module.new(__name__)
lfunc = gen_loopcontinue(mod)
logging.debug(mod)
mod.verify()
exe = CExecutor(mod)
func = exe.get_ctype_function(lfunc, 'int, int')
for x in range(100):
self.assertEqual(func(x), loopcontinue(x))
if __name__ == '__main__':
unittest.main()

View file

@ -1,128 +0,0 @@
from llvm.core import *
from llvm.passes import *
from llvm.ee import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import unittest, logging
def nestedloop1(d):
z = 0
for x in range(100):
for y in range(100):
z += x * d + int(y / d)
return z
def gen_nestedloop1(mod):
functype = Type.function(C.int, [C.int])
func = mod.add_function(functype, 'nestedloop1')
cb = CBuilder(func)
d = cb.args[0]
x = cb.var(C.int)
y = cb.var(C.int)
z = cb.var(C.int)
one = cb.constant(C.int, 1)
zero = cb.constant(C.int, 0)
limit = cb.constant(C.int, 100)
z.assign(zero)
x.assign(zero)
with cb.loop() as outer:
with outer.condition() as setcond:
setcond( x < limit )
with outer.body():
y.assign(zero)
with cb.loop() as inner:
with inner.condition() as setcond:
setcond( y < limit )
with inner.body():
z += x * d + y / d
y += one
x += one
cb.ret(z)
cb.close()
return func
def nestedloop2(d):
z = 0
for x in range(1, 100):
for y in range(1, 100):
if x > y:
z += int(x / y) * d
else:
z += int(y / x) * d
return z
def gen_nestedloop2(mod):
functype = Type.function(C.int, [C.int])
func = mod.add_function(functype, 'nestedloop2')
cb = CBuilder(func)
d = cb.args[0]
x = cb.var(C.int)
y = cb.var(C.int)
z = cb.var(C.int)
one = cb.constant(C.int, 1)
zero = cb.constant(C.int, 0)
limit = cb.constant(C.int, 100)
z.assign(zero)
x.assign(one)
with cb.loop() as outer:
with outer.condition() as setcond:
setcond( x < limit )
with outer.body():
y.assign(one)
with cb.loop() as inner:
with inner.condition() as setcond:
setcond( y < limit )
with inner.body():
with cb.ifelse(x > y) as ifelse:
with ifelse.then():
z += x / y * d
with ifelse.otherwise():
z += y / x * d
y += one
x += one
cb.ret(z)
cb.close()
return func
class TestNestedLoop(unittest.TestCase):
def test_nestedloop1(self):
mod = Module.new(__name__)
lfunc = gen_nestedloop1(mod)
logging.debug(mod)
mod.verify()
exe = CExecutor(mod)
func = exe.get_ctype_function(lfunc, 'int, int')
for x in range(1, 100):
self.assertEqual(func(x), int(nestedloop1(x)))
def test_nestedloop2(self):
mod = Module.new(__name__)
lfunc = gen_nestedloop2(mod)
logging.debug(mod)
mod.verify()
exe = CExecutor(mod)
func = exe.get_ctype_function(lfunc, 'int, int')
for x in range(1, 100):
self.assertEqual(func(x), int(nestedloop2(x)))
if __name__ == '__main__':
unittest.main()

View file

@ -1,60 +0,0 @@
from llvm.core import *
from llvm.passes import *
from llvm.ee import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import sys, unittest, logging
from subprocess import Popen, PIPE
def gen_debugprint(mod):
functype = Type.function(C.void, [])
func = mod.add_function(functype, 'debugprint')
cb = CBuilder(func)
fmt = cb.constant_string("Show %d %.3f %.3e\n")
an_int = cb.constant(C.int, 123)
a_float = cb.constant(C.double, 1.234)
a_double = cb.constant(C.double, 1e-31)
cb.printf(fmt, an_int, a_float, a_double)
cb.debug('an_int =', an_int, 'a_float =', a_float, 'a_double =', a_double)
cb.ret()
cb.close()
return func
def main_debugprint():
# generate code
mod = Module.new(__name__)
lfunc = gen_debugprint(mod)
logging.debug(mod)
mod.verify()
# run
exe = CExecutor(mod)
func = exe.get_ctype_function(lfunc, 'void')
func()
class TestPrint(unittest.TestCase):
def test_debugprint(self):
p = Popen(["python", "test_print.py", "-child"], stdout=PIPE)
p.wait()
lines = p.stdout.read().decode().splitlines(False)
expect = [
'Show 123 1.234 1.000e-31',
'an_int = 123 a_float = 1.234000e+00 a_double = 1.000000e-31',
]
self.assertEqual(expect, lines)
p.stdout.close()
if __name__ == '__main__':
try:
if sys.argv[1] == '-child':
main_debugprint()
except IndexError:
unittest.main()

View file

@ -1,90 +0,0 @@
from llvm.core import *
from llvm.passes import *
from llvm.ee import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import unittest, logging
# logging.basicConfig(level=logging.DEBUG)
NUM_OF_THREAD = 4
def gen_test_worker(mod):
cb = CBuilder.new_function(mod, 'worker', C.void, [C.pointer(C.int)])
pval = cb.args[0]
val = pval.load()
one = cb.constant(val.type, 1)
pval.store(val + one)
cb.ret()
cb.close()
def gen_test_pthread(mod):
cb = CBuilder.new_function(mod, 'manager', C.int, [C.int])
arg = cb.args[0]
worker_func = cb.get_function_named('worker')
pthread_create = cb.get_function_named('pthread_create')
pthread_join = cb.get_function_named('pthread_join')
NULL = cb.constant_null(C.void_p)
cast_to_null = lambda x: x.cast(C.void_p)
threads = cb.array(C.void_p, NUM_OF_THREAD)
for tid in range(NUM_OF_THREAD):
pthread_create_args = [threads[tid].reference(),
NULL,
worker_func,
arg.reference()]
pthread_create(*map(cast_to_null, pthread_create_args))
worker_func(arg.reference())
for tid in range(NUM_OF_THREAD):
pthread_join_args = threads[tid], NULL
pthread_join(*map(cast_to_null, pthread_join_args))
cb.ret(arg)
cb.close()
return cb.function
class TestPThread(unittest.TestCase):
def test_pthread(self):
mod = Module.new(__name__)
# add pthread functions
mod.add_function(Type.function(C.int,
[C.void_p, C.void_p, C.void_p, C.void_p]),
'pthread_create')
mod.add_function(Type.function(C.int,
[C.void_p, C.void_p]),
'pthread_join')
gen_test_worker(mod)
lf_test_pthread = gen_test_pthread(mod)
logging.debug(mod)
mod.verify()
exe = CExecutor(mod)
exe.engine.get_pointer_to_function(mod.get_function_named('worker'))
func = exe.get_ctype_function(lf_test_pthread, 'int, int')
inarg = 1234
gold = inarg + NUM_OF_THREAD + 1
self.assertLessEqual(func(inarg), gold)
# Cannot determine the exact return value due to untamed race condition
count_race = 0
for _ in range(2**12):
if func(inarg) != gold:
count_race += 1
if count_race > 0:
logging.info("Race condition occured %d times.", count_race)
logging.info("Race condition is expected.")
if __name__ == '__main__':
unittest.main()

View file

@ -1,52 +0,0 @@
from llvm.core import *
from llvm_cbuilder import *
import llvm_cbuilder.shortnames as C
import unittest, ctypes
class Vector2D(CStruct):
_fields_ = [
('x', C.float),
('y', C.float),
]
class Vector2DCtype(ctypes.Structure):
_fields_ = [
('x', ctypes.c_float),
('y', ctypes.c_float),
]
def gen_vector2d_dist(mod):
functype = Type.function(C.float, [C.pointer(Vector2D.llvm_type())])
func = mod.add_function(functype, 'vector2d_dist')
cb = CBuilder(func)
vec = cb.var(Vector2D, cb.args[0].load())
dist = vec.x * vec.x + vec.y * vec.y
cb.ret(dist)
cb.close()
return func
class TestStruct(unittest.TestCase):
def test_vector2d_dist(self):
# prepare module
mod = Module.new('mod')
lfunc = gen_vector2d_dist(mod)
mod.verify()
# run
exe = CExecutor(mod)
func = exe.get_ctype_function(lfunc, ctypes.c_float, ctypes.POINTER(Vector2DCtype))
from random import random
pydist = lambda x, y: x * x + y * y
for _ in range(100):
x, y = random(), random()
vec = Vector2DCtype(x=x, y=y)
ans = func(ctypes.pointer(vec))
gold = pydist(x, y)
self.assertLess(abs(ans-gold)/gold, 1e-6)
if __name__ == '__main__':
unittest.main()