reorganize
This commit is contained in:
parent
1009daf3fa
commit
f9d7195413
25 changed files with 0 additions and 2535 deletions
23
README.md
23
README.md
|
|
@ -1,23 +0,0 @@
|
|||
llvm_cbuilder
|
||||
-------------
|
||||
|
||||
A llvm-py Builder wrapper for writing in slightly higher-level constructs.
|
||||
This is aiming for two usecases:
|
||||
|
||||
1. Emit LLVM code in a more human-readable way;
|
||||
|
||||
2. Writing low-level code that you can't do it properly/portably with C, e.g
|
||||
template (generic), atomic operations, memory ordering...
|
||||
|
||||
|
||||
|
||||
Parallel Vectorize
|
||||
------------------
|
||||
|
||||
`parallel_vectorize.py` implements a set of code generator that bases on
|
||||
llvm_cbuilder to create specialized parallel ufunc.
|
||||
|
||||
See `test_parallel_vectorize_numpy*.py` for testing and demo of the code
|
||||
with the new `numpy.fromfunc()`.
|
||||
|
||||
|
||||
|
|
@ -1,586 +0,0 @@
|
|||
{
|
||||
"metadata": {
|
||||
"name": "llvm_cbuilder_intro"
|
||||
},
|
||||
"nbformat": 2,
|
||||
"worksheets": [
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"llvm_cbuilder: A Quick Tour",
|
||||
"---------------------------",
|
||||
"",
|
||||
"Writing code generation logic in pure llvmpy requires one to think like a compiler -- ",
|
||||
"dealing with basic-blocks, branches, SSA, etc.",
|
||||
"The resulting code looks like assembly code, with no apparent hint about the control-flow.",
|
||||
"",
|
||||
"llvm_cbuilder is originally designed to simplify the translation of C code into llvmpy.",
|
||||
"It provides a simple API for mimicking C programming in Python.",
|
||||
"With llvm_cbulder, one can write very low-level code, probably even lower-level than C.",
|
||||
"At the same time, one can use if-else, loop, structures and a bit of OOP."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"A Simple Example",
|
||||
"----------------",
|
||||
"",
|
||||
"Let's see some action.",
|
||||
"We will define a function that calculates the square of a double-precision float."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": true,
|
||||
"input": [
|
||||
"from llvm.core import *",
|
||||
"from llvm_cbuilder import *",
|
||||
"import llvm_cbuilder.shortnames as C",
|
||||
"",
|
||||
"class Square(CDefinition):",
|
||||
" # prototype: double square(double x)",
|
||||
" _name_ = 'square' # function name",
|
||||
" _retty_ = C.double",
|
||||
" _argtys_ = [ ('x', C.double) ]",
|
||||
" ",
|
||||
" def body(self, x):",
|
||||
" y = x * x # just write out the expression",
|
||||
" self.ret(y)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [],
|
||||
"prompt_number": 1
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Notice how numerical expressions can be written naturally."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Let's see the emitted code:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"m = Module.new('my_module')",
|
||||
"llvm_square = Square()(m) # define square() in my_module",
|
||||
"print(m)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"; ModuleID = 'my_module'",
|
||||
"",
|
||||
"define double @square(double %x) {",
|
||||
"decl:",
|
||||
" %x1 = alloca double",
|
||||
" br label %body",
|
||||
"",
|
||||
"body: ; preds = %decl",
|
||||
" store double %x, double* %x1",
|
||||
" %0 = load double* %x1",
|
||||
" %1 = load double* %x1",
|
||||
" %2 = fmul double %0, %1",
|
||||
" ret double %2",
|
||||
"}",
|
||||
""
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 2
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Let's generate a ctype function object to call `square()` in the Python code:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"exe = CExecutor(m)",
|
||||
"square = exe.get_ctype_function(llvm_square, \"double, double\")",
|
||||
"result = square(1.2)",
|
||||
"print(result)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"1.44"
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 3
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Control Flow Constructs",
|
||||
"-----------------------",
|
||||
"",
|
||||
"The main strength of llvm_cbuilder is the control-flow constructs. ",
|
||||
"They use the python \"with\" statement to setup new code blocks to",
|
||||
"contain different paths of the control flow.",
|
||||
"",
|
||||
"The following example demonstrates both the if-else and loop contructs:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": true,
|
||||
"input": [
|
||||
"class IsPrime(CDefinition):",
|
||||
" # prototype int isprime(int x)",
|
||||
" _name_ = 'isprime'",
|
||||
" _retty_ = C.int",
|
||||
" _argtys_ = [ ('x', C.int) ]",
|
||||
" ",
|
||||
" def body(self, x):",
|
||||
" two = self.constant(C.int, 2)",
|
||||
" true = one = self.constant(C.int, 1)",
|
||||
" false = zero = self.constant(C.int, 0)",
|
||||
" ",
|
||||
" with self.ifelse( x <= two ) as ifelse:",
|
||||
" with ifelse.then():",
|
||||
" self.ret(true)",
|
||||
" ",
|
||||
" with self.ifelse( (x % two) == zero ) as ifelse:",
|
||||
" with ifelse.then():",
|
||||
" self.ret(false)",
|
||||
" ",
|
||||
" idx = self.var(C.int, 3, name='idx')",
|
||||
" with self.loop() as loop:",
|
||||
" with loop.condition() as setcond:",
|
||||
" setcond( idx < x )",
|
||||
" ",
|
||||
" with loop.body():",
|
||||
" with self.ifelse( (x % idx) == zero ) as ifelse:",
|
||||
" with ifelse.then():",
|
||||
" self.ret(false)",
|
||||
" idx += two",
|
||||
" self.ret(true)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [],
|
||||
"prompt_number": 4
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"The code above is quite verbose.",
|
||||
"It is like writing in Pascal or Ada. ",
|
||||
"But, it is still easier than writing in llvmpy directly.",
|
||||
"Take a look at the generated LLVM IR below."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"llvm_isprime = IsPrime()(m)",
|
||||
"print(llvm_isprime)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"",
|
||||
"define i32 @isprime(i32 %x) {",
|
||||
"decl:",
|
||||
" %x1 = alloca i32",
|
||||
" %idx = alloca i32",
|
||||
" br label %body",
|
||||
"",
|
||||
"body: ; preds = %decl",
|
||||
" store i32 %x, i32* %x1",
|
||||
" %0 = load i32* %x1",
|
||||
" %1 = icmp sle i32 %0, 2",
|
||||
" br i1 %1, label %if.then, label %if.else",
|
||||
"",
|
||||
"if.then: ; preds = %body",
|
||||
" ret i32 1",
|
||||
"",
|
||||
"if.else: ; preds = %body",
|
||||
" br label %if.end",
|
||||
"",
|
||||
"if.end: ; preds = %if.else",
|
||||
" %2 = load i32* %x1",
|
||||
" %3 = srem i32 %2, 2",
|
||||
" %4 = icmp eq i32 %3, 0",
|
||||
" br i1 %4, label %if.then2, label %if.else3",
|
||||
"",
|
||||
"if.then2: ; preds = %if.end",
|
||||
" ret i32 0",
|
||||
"",
|
||||
"if.else3: ; preds = %if.end",
|
||||
" br label %if.end4",
|
||||
"",
|
||||
"if.end4: ; preds = %if.else3",
|
||||
" store i32 3, i32* %idx",
|
||||
" br label %loop.cond",
|
||||
"",
|
||||
"loop.cond: ; preds = %if.end7, %if.end4",
|
||||
" %5 = load i32* %idx",
|
||||
" %6 = load i32* %x1",
|
||||
" %7 = icmp slt i32 %5, %6",
|
||||
" br i1 %7, label %loop.body, label %loop.end",
|
||||
"",
|
||||
"loop.body: ; preds = %loop.cond",
|
||||
" %8 = load i32* %x1",
|
||||
" %9 = load i32* %idx",
|
||||
" %10 = srem i32 %8, %9",
|
||||
" %11 = icmp eq i32 %10, 0",
|
||||
" br i1 %11, label %if.then5, label %if.else6",
|
||||
"",
|
||||
"loop.end: ; preds = %loop.cond",
|
||||
" ret i32 1",
|
||||
"",
|
||||
"if.then5: ; preds = %loop.body",
|
||||
" ret i32 0",
|
||||
"",
|
||||
"if.else6: ; preds = %loop.body",
|
||||
" br label %if.end7",
|
||||
"",
|
||||
"if.end7: ; preds = %if.else6",
|
||||
" %12 = load i32* %idx",
|
||||
" %13 = add i32 %12, 2",
|
||||
" store i32 %13, i32* %idx",
|
||||
" br label %loop.cond",
|
||||
"}",
|
||||
""
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 5
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"We'll setup a ctype function object to try it out:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"isprime = exe.get_ctype_function(llvm_isprime, 'int, int')",
|
||||
"prime_100 = filter(isprime, range(2, 100))",
|
||||
"print(prime_100)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"[2, 3, 5, 7, 11, 13, 17, 19, 23, 29, 31, 37, 41, 43, 47, 53, 59, 61, 67, 71, 73, 79, 83, 89, 97]"
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 6
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Structures",
|
||||
"----------",
|
||||
"",
|
||||
"It is possible to create structures in llvm_cbuilder. ",
|
||||
"Beware that structures in LLVM are unlike those in C.",
|
||||
"LLVM type system allows structures to be literal or identified.",
|
||||
"Literal structures are equivalent iff they have the same elements.",
|
||||
"Identified structures are equivalent iff they have the same name.",
|
||||
"In llvm_cbuilder, all structures are, by default, literal types.",
|
||||
"",
|
||||
"Here's an example:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": true,
|
||||
"input": [
|
||||
"class Vector2D(CStruct):",
|
||||
" _fields_ = [",
|
||||
" ('x', C.float),",
|
||||
" ('y', C.float),",
|
||||
" ]"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [],
|
||||
"prompt_number": 7
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"That's all you need for a structure. Very much like defining structures with ctypes.",
|
||||
"",
|
||||
"We can also bind methods to structures which inline code to the caller."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": true,
|
||||
"input": [
|
||||
"class Vector2D(CStruct):",
|
||||
" _fields_ = [",
|
||||
" ('x', C.float),",
|
||||
" ('y', C.float),",
|
||||
" ]",
|
||||
" ",
|
||||
" def add(self, other):",
|
||||
" self.x += other.x",
|
||||
" self.y += other.y"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [],
|
||||
"prompt_number": 8
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Setup a a test function:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": true,
|
||||
"input": [
|
||||
"class TestVector(CDefinition):",
|
||||
" _name_ = 'testvector'",
|
||||
" _retty_ = C.float",
|
||||
" _argtys_ = [('x', C.float), ",
|
||||
" ('y', C.float)] ",
|
||||
" ",
|
||||
" def body(self, x, y):",
|
||||
" va = self.var(Vector2D)",
|
||||
" vb = self.var(Vector2D)",
|
||||
" va.x.assign(self.constant(C.float, 10))",
|
||||
" va.y.assign(self.constant(C.float, 20))",
|
||||
" vb.x.assign(x)",
|
||||
" vb.y.assign(y)",
|
||||
" ",
|
||||
" va.add(vb)",
|
||||
" ",
|
||||
" return self.ret(va.x * va.y)",
|
||||
" "
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [],
|
||||
"prompt_number": 9
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"llvm_testvec = TestVector()(m)",
|
||||
"testvec = exe.get_ctype_function(llvm_testvec, \"float, float, float\")",
|
||||
"print(testvec(-8, 5)) # should output 50"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"50.0"
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 10
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Generic Programming",
|
||||
"-------------------",
|
||||
"",
|
||||
"llvm_cbuilder supports generic function (or template function for C++).",
|
||||
"When a `CDefinition` defines the `specialize` class method,",
|
||||
"its constructor will invoke `specialize`.",
|
||||
"All parameters passed to the constructor are forwarded to `specialize`.",
|
||||
"",
|
||||
"The constructor of a generic CDefinition creates a new dynamic class with the original CDefinition subclass as the parent.",
|
||||
"Thus, `specialize` can modify class attributes without affecting the parent."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"class GenericSquare(CDefinition):",
|
||||
" @classmethod",
|
||||
" def specialize(cls, data_type):",
|
||||
" cls._name_ = '.'.join(['square', str(data_type)]) # set function name",
|
||||
" cls._retty_ = data_type # set return type",
|
||||
" cls._argtys_ = [ ('x', data_type) ] # set argument type",
|
||||
" ",
|
||||
" def body(self, x):",
|
||||
" y = x * x # just write out the expression",
|
||||
" self.ret(y)",
|
||||
" ",
|
||||
"llvm_square_int = GenericSquare(data_type=C.int)(m)",
|
||||
"llvm_square_float = GenericSquare(data_type=C.float)(m)",
|
||||
"",
|
||||
"print(llvm_square_int)",
|
||||
"print(llvm_square_float)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"",
|
||||
"define i32 @square.i32(i32 %x) {",
|
||||
"decl:",
|
||||
" %x1 = alloca i32",
|
||||
" br label %body",
|
||||
"",
|
||||
"body: ; preds = %decl",
|
||||
" store i32 %x, i32* %x1",
|
||||
" %0 = load i32* %x1",
|
||||
" %1 = load i32* %x1",
|
||||
" %2 = mul i32 %0, %1",
|
||||
" ret i32 %2",
|
||||
"}",
|
||||
"",
|
||||
"",
|
||||
"define float @square.float(float %x) {",
|
||||
"decl:",
|
||||
" %x1 = alloca float",
|
||||
" br label %body",
|
||||
"",
|
||||
"body: ; preds = %decl",
|
||||
" store float %x, float* %x1",
|
||||
" %0 = load float* %x1",
|
||||
" %1 = load float* %x1",
|
||||
" %2 = fmul float %0, %1",
|
||||
" ret float %2",
|
||||
"}",
|
||||
""
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 11
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"External Functions",
|
||||
"------------------",
|
||||
"",
|
||||
"`CExternal` is a convenient class to help accessing externally-defined functions.",
|
||||
"",
|
||||
"We define an external interface to `sqrtf()` in libm."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": true,
|
||||
"input": [
|
||||
"class LibM(CExternal):",
|
||||
" sqrtf = Type.function(C.float, [C.float])"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [],
|
||||
"prompt_number": 12
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"All class attributes that are `llvm.core.FunctionType` in `CExternal` ",
|
||||
"are converted to a `CFunc` instance.",
|
||||
"`CFunc` instances are callable.",
|
||||
"`CFunc.__call__` generates code that perform the corresponding LLVM function call.",
|
||||
"",
|
||||
"The following example demonstrates the usage of `CExternal` and `CFunc`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"class TestSqrtf(CDefinition):",
|
||||
" _name_ = 'test_sqrtf'",
|
||||
" _retty_ = C.float",
|
||||
" _argtys_ = [('x', C.float)]",
|
||||
" ",
|
||||
" def body(self, x):",
|
||||
" libm = LibM(self) # init the API",
|
||||
" y = libm.sqrtf(x) # call sqrtf",
|
||||
" self.ret(y)",
|
||||
" ",
|
||||
"llvm_sqrtf = TestSqrtf()(m)",
|
||||
"sqrtf = exe.get_ctype_function(llvm_sqrtf, 'float, float')",
|
||||
"print(sqrtf(144))"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"12.0"
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 13
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Casting",
|
||||
"-------",
|
||||
"",
|
||||
"Unlike C, llvm_cbuilder does not automatically cast variables.",
|
||||
"It is up to the user to explicit cast types.",
|
||||
"All binary operations require both operands to be the same type.",
|
||||
"All values in llvm_cbuilder has a `cast(destty)` method.",
|
||||
"",
|
||||
"For example:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": true,
|
||||
"input": [
|
||||
"class CastExample(CDefinition):",
|
||||
" _name_ = 'cast_example'",
|
||||
" _retty_ = C.float",
|
||||
" _argtys_ = [('x', C.int)]",
|
||||
" ",
|
||||
" def body(self, x):",
|
||||
" y = x.cast(C.float) # cast integer x to float",
|
||||
" self.ret(y)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [],
|
||||
"prompt_number": 14
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"More... (TODO)"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
Binary file not shown.
|
|
@ -1,251 +0,0 @@
|
|||
{
|
||||
"metadata": {
|
||||
"name": "parallel_vectorize"
|
||||
},
|
||||
"nbformat": 2,
|
||||
"worksheets": [
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Parallel Vectorize",
|
||||
"------------------",
|
||||
"",
|
||||
"The `parallel_vectorize.py` module contains a set of llvmpy code generators",
|
||||
"for creating mulithreaded _ufunc_. ",
|
||||
"It depends on the new `numpy.fromfunc` for turning arbitrary function pointers into _ufunc_.",
|
||||
"",
|
||||
"From LLVM Function",
|
||||
"------------------",
|
||||
"",
|
||||
"The `parallel_vectorize_from_func` method generates multithreaded _ufunc_ from LLVM functions.",
|
||||
"",
|
||||
"First, we will implement a workload function:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"from llvm_cbuilder import *",
|
||||
"from llvm_cbuilder import shortnames as C",
|
||||
"from llvm.core import *",
|
||||
"",
|
||||
"# Implement a workload",
|
||||
"class Square(CDefinition):",
|
||||
" _name_ = 'square'",
|
||||
" _retty_ = C.double # 1 output: double",
|
||||
" _argtys_ = [('x', C.double)] # 1 input: double",
|
||||
" ",
|
||||
" def body(self, x):",
|
||||
" self.ret(x * x)",
|
||||
"",
|
||||
"m = Module.new('my_module')",
|
||||
"llvm_square = Square()(m) # Generate a llvm function",
|
||||
"print(llvm_square) "
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"",
|
||||
"define double @square(double %x) {",
|
||||
"decl:",
|
||||
" %x1 = alloca double",
|
||||
" br label %body",
|
||||
"",
|
||||
"body: ; preds = %decl",
|
||||
" store double %x, double* %x1",
|
||||
" %0 = load double* %x1",
|
||||
" %1 = load double* %x1",
|
||||
" %2 = fmul double %0, %1",
|
||||
" ret double %2",
|
||||
"}",
|
||||
""
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 1
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Then, we will generate a _ufunc_ from `llvm_square`:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": true,
|
||||
"input": [
|
||||
"from llvm.ee import *",
|
||||
"engine = EngineBuilder.new(m).create() # Generate JIT engine",
|
||||
"",
|
||||
"from parallel_vectorize import parallel_vectorize_from_func",
|
||||
"ufunc_square = parallel_vectorize_from_func(llvm_square, engine) # Generate UFunc"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [],
|
||||
"prompt_number": 2
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"We are ready to use `ufunc_square` as a regular _ufunc_."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"import numpy as np",
|
||||
"A = np.arange(10., dtype=np.double)",
|
||||
"ufunc_square(A)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "pyout",
|
||||
"prompt_number": 3,
|
||||
"text": [
|
||||
"array([ 0., 1., 4., 9., 16., 25., 36., 49., 64., 81.])"
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 3
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"Here's another example that uses three inputs:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"class SumOfThree(CDefinition):",
|
||||
" _name_ = 'sum.of.three'",
|
||||
" _retty_ = C.int",
|
||||
" _argtys_ = [('x', C.int),",
|
||||
" ('y', C.int),",
|
||||
" ('z', C.int)]",
|
||||
" def body(self, x, y, z):",
|
||||
" self.ret( x + y + z )",
|
||||
"",
|
||||
"llvm_sum3 = SumOfThree()(m)",
|
||||
"ufunc_sum3 = parallel_vectorize_from_func(llvm_sum3, engine)",
|
||||
"A = np.arange(10, dtype=np.int32)",
|
||||
"B = A * 10",
|
||||
"C = A * 100",
|
||||
"ufunc_sum3(A, B, C)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "pyout",
|
||||
"prompt_number": 4,
|
||||
"text": [
|
||||
"array([ 0, 111, 222, 333, 444, 555, 666, 777, 888, 999], dtype=int32)"
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 4
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"* * *",
|
||||
"",
|
||||
"Internals",
|
||||
"---------",
|
||||
"",
|
||||
"There are four functions behind each multithreaded _ufunc_.",
|
||||
"",
|
||||
"1. the workload function (user defined);",
|
||||
"2. the thread worker function (`UFuncCoreGeneric`);",
|
||||
"3. the thread manager function (`ParallelUFuncPlatform`);",
|
||||
"4. the ufunc entry point function (`SpecializedParallelUFunc`).",
|
||||
"",
|
||||
"**UFuncCoreGeneric** specializes to a llvm function type.",
|
||||
"**It currently understands simple builtin scalar types (integers, float, double) only as arguments and return-type for the workload function.**",
|
||||
"It sends work items to the workload function and performs work-stealing when it has finished its own workqueue.",
|
||||
"Work-stealing uses atomic compare-exchange (or CAS) instruction to acquire ownership of a workqueue.",
|
||||
"Work-stealing is implemented in the `UFuncCore._do_work_stealing`.",
|
||||
"It can be disabled on platform that does not support atomic operations.",
|
||||
"",
|
||||
"**ParallelUFuncPlatform** specializes to the maximum number of threads. ",
|
||||
"It divides all works equally among all threads.",
|
||||
"Each thread executes the function generated by `UFuncCoreGeneric` once.",
|
||||
"",
|
||||
"**SpecializedParallelUFunc** is the specialized _ufunc_ entry point for a specific combination of ",
|
||||
"workload, UFuncCoreGeneric and ParallelUFuncPlatform.",
|
||||
"",
|
||||
"Here's an example that uses `SpecializedParallelUFunc` directly for the `SumOfThree` workload."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"import parallel_vectorize as pv",
|
||||
"# specialize",
|
||||
"def_spuf = pv.SpecializedParallelUFunc(pv.ParallelUFuncPlatform(num_thread=2),",
|
||||
" pv.UFuncCoreGeneric(llvm_sum3.type.pointee),",
|
||||
" CFuncRef(llvm_sum3))",
|
||||
"# define",
|
||||
"spuf = def_spuf(m)",
|
||||
"print(spuf.name)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"specialized_parallel_ufunc_2_ufunc_worker.i32.i32.i32.i32_sum.of.three"
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 5
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"`CFuncRef` also accepts arbitrary function pointer as long as the function type is provided."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"collapsed": false,
|
||||
"input": [
|
||||
"# specialize",
|
||||
"fnty = llvm_sum3.type.pointee",
|
||||
"sum3ptr = engine.get_pointer_to_function(llvm_sum3)",
|
||||
"print(\"as function pointer: %x\" % sum3ptr)",
|
||||
"def_spuf = pv.SpecializedParallelUFunc(pv.ParallelUFuncPlatform(num_thread=2),",
|
||||
" pv.UFuncCoreGeneric(fnty),",
|
||||
" CFuncRef('sum3.as.ptr', fnty, sum3ptr)) # name, type, ptr",
|
||||
"# define",
|
||||
"spuf = def_spuf(m)",
|
||||
"print(spuf.name)"
|
||||
],
|
||||
"language": "python",
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "stream",
|
||||
"stream": "stdout",
|
||||
"text": [
|
||||
"as function pointer: 7f0bfc090740",
|
||||
"specialized_parallel_ufunc_2_ufunc_worker.i32.i32.i32.i32_sum3.as.ptr"
|
||||
]
|
||||
}
|
||||
],
|
||||
"prompt_number": 6
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
Binary file not shown.
|
|
@ -1,461 +0,0 @@
|
|||
'''
|
||||
This file implements the code-generator for parallel-vectorize.
|
||||
|
||||
ParallelUFunc is the platform independent base class for generating
|
||||
the thread dispatcher. This thread dispatcher launches threads
|
||||
that execute the generated function of UFuncCore.
|
||||
UFuncCore is subclassed to specialize for the input/output types.
|
||||
The actual workload is invoked inside the function generated by UFuncCore.
|
||||
UFuncCore also defines a work-stealing mechanism that allows idle threads
|
||||
to steal works from other threads.
|
||||
'''
|
||||
|
||||
from llvm.core import *
|
||||
from llvm.passes import *
|
||||
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
|
||||
import sys
|
||||
|
||||
class WorkQueue(CStruct):
|
||||
'''structure for workqueue for parallel-ufunc.
|
||||
'''
|
||||
|
||||
_fields_ = [
|
||||
('next', C.intp), # next index of work item
|
||||
('last', C.intp), # last index of work item (exlusive)
|
||||
('lock', C.int), # for locking the workqueue
|
||||
]
|
||||
|
||||
|
||||
def Lock(self):
|
||||
'''inline the lock procedure.
|
||||
'''
|
||||
with self.parent.loop() as loop:
|
||||
with loop.condition() as setcond:
|
||||
unlocked = self.parent.constant(self.lock.type, 0)
|
||||
locked = self.parent.constant(self.lock.type, 1)
|
||||
|
||||
res = self.lock.reference().atomic_cmpxchg(unlocked, locked,
|
||||
ordering='acquire')
|
||||
setcond( res != unlocked )
|
||||
|
||||
with loop.body():
|
||||
pass
|
||||
|
||||
def Unlock(self):
|
||||
'''inline the unlock procedure.
|
||||
'''
|
||||
unlocked = self.parent.constant(self.lock.type, 0)
|
||||
locked = self.parent.constant(self.lock.type, 1)
|
||||
|
||||
res = self.lock.reference().atomic_cmpxchg(locked, unlocked,
|
||||
ordering='release')
|
||||
|
||||
with self.parent.ifelse( res != locked ) as ifelse:
|
||||
with ifelse.then():
|
||||
# This shall kill the program
|
||||
self.parent.unreachable()
|
||||
|
||||
|
||||
class ContextCommon(CStruct):
|
||||
'''structure for thread-shared context information in parallel-ufunc.
|
||||
'''
|
||||
_fields_ = [
|
||||
# loop ufunc args
|
||||
('args', C.pointer(C.char_p)),
|
||||
('dimensions', C.pointer(C.intp)),
|
||||
('steps', C.pointer(C.intp)),
|
||||
('data', C.void_p),
|
||||
# specifics for work queues
|
||||
('func', C.void_p),
|
||||
('num_thread', C.int),
|
||||
('workqueues', C.pointer(WorkQueue.llvm_type())),
|
||||
]
|
||||
|
||||
class Context(CStruct):
|
||||
'''structure for thread-specific context information in parallel-ufunc.
|
||||
'''
|
||||
_fields_ = [
|
||||
('common', C.pointer(ContextCommon.llvm_type())),
|
||||
('id', C.int),
|
||||
('completed', C.intp),
|
||||
]
|
||||
|
||||
class ParallelUFunc(CDefinition):
|
||||
'''the generic parallel vectorize mechanism
|
||||
|
||||
Can be specialized to the maximum number of threads on the platform.
|
||||
|
||||
|
||||
Platform dependent threading function is implemented in
|
||||
|
||||
def _dispatch_worker(self, worker, contexts, num_thread):
|
||||
...
|
||||
|
||||
which should be implemented in subclass or mixin.
|
||||
'''
|
||||
|
||||
_argtys_ = [
|
||||
('func', C.void_p),
|
||||
('worker', C.void_p),
|
||||
('args', C.pointer(C.char_p)),
|
||||
('dimensions', C.pointer(C.intp)),
|
||||
('steps', C.pointer(C.intp)),
|
||||
('data', C.void_p),
|
||||
]
|
||||
|
||||
@classmethod
|
||||
def specialize(cls, num_thread):
|
||||
'''specialize to the maximum # of thread
|
||||
'''
|
||||
cls._name_ = 'parallel_ufunc_%d' % num_thread
|
||||
cls.ThreadCount = num_thread
|
||||
|
||||
def body(self, func, worker, args, dimensions, steps, data):
|
||||
# Setup variables
|
||||
ThreadCount = self.ThreadCount
|
||||
common = self.var(ContextCommon, name='common')
|
||||
workqueues = self.array(WorkQueue, ThreadCount, name='workqueues')
|
||||
contexts = self.array(Context, ThreadCount, name='contexts')
|
||||
|
||||
num_thread = self.var(C.int, ThreadCount, name='num_thread')
|
||||
|
||||
# Initialize ContextCommon
|
||||
common.args.assign(args)
|
||||
common.dimensions.assign(dimensions)
|
||||
common.steps.assign(steps)
|
||||
common.data.assign(data)
|
||||
common.func.assign(func)
|
||||
common.num_thread.assign(num_thread.cast(C.int))
|
||||
common.workqueues.assign(workqueues.reference())
|
||||
|
||||
# Determine chunksize, initial count of work-items per thread.
|
||||
# If total_work >= num_thread, equally divide the works.
|
||||
# If total_work % num_thread != 0, the last thread does all remaining works.
|
||||
# If total_work < num_thread, each thread does one work,
|
||||
# and set num_thread to total_work
|
||||
N = dimensions[0]
|
||||
ChunkSize = self.var_copy(N / num_thread.cast(N.type))
|
||||
ChunkSize_NULL = self.constant_null(ChunkSize.type)
|
||||
with self.ifelse(ChunkSize == ChunkSize_NULL) as ifelse:
|
||||
with ifelse.then():
|
||||
ChunkSize.assign(self.constant(ChunkSize.type, 1))
|
||||
num_thread.assign(N.cast(num_thread.type))
|
||||
|
||||
# Populate workqueue for all threads
|
||||
self._populate_workqueues(workqueues, N, ChunkSize, num_thread)
|
||||
|
||||
# Populate contexts for all threads
|
||||
self._populate_context(contexts, common, num_thread)
|
||||
|
||||
# Dispatch worker threads
|
||||
self._dispatch_worker(worker, contexts, num_thread)
|
||||
|
||||
## DEBUG ONLY ##
|
||||
# Check for race condition
|
||||
if True:
|
||||
total_completed = self.var(C.intp, 0, name='total_completed')
|
||||
for t in range(ThreadCount):
|
||||
cur_ctxt = contexts[t].as_struct(Context)
|
||||
total_completed += cur_ctxt.completed
|
||||
# self.debug(cur_ctxt.id, 'completed', cur_ctxt.completed)
|
||||
|
||||
with self.ifelse( total_completed == N ) as ifelse:
|
||||
with ifelse.then():
|
||||
# self.debug("All is well!")
|
||||
pass # keep quite if all is well
|
||||
with ifelse.otherwise():
|
||||
self.debug("ERROR: race occurred! Trigger segfault")
|
||||
self.unreachable()
|
||||
|
||||
# Return
|
||||
self.ret()
|
||||
|
||||
def _populate_workqueues(self, workqueues, N, ChunkSize, num_thread):
|
||||
'''loop over all threads and populate the workqueue for each of them.
|
||||
'''
|
||||
ONE = self.constant(num_thread.type, 1)
|
||||
with self.for_range(num_thread) as (loop, i):
|
||||
cur_wq = workqueues[i].as_struct(WorkQueue)
|
||||
cur_wq.next.assign(i.cast(ChunkSize.type) * ChunkSize)
|
||||
cur_wq.last.assign((i + ONE).cast(ChunkSize.type) * ChunkSize)
|
||||
cur_wq.lock.assign(self.constant(C.int, 0))
|
||||
# end loop
|
||||
last_wq = workqueues[num_thread - ONE].as_struct(WorkQueue)
|
||||
last_wq.last.assign(N)
|
||||
|
||||
def _populate_context(self, contexts, common, num_thread):
|
||||
'''loop over all threads and populate contexts for each of them.
|
||||
'''
|
||||
ONE = self.constant(num_thread.type, 1)
|
||||
with self.for_range(num_thread) as (loop, i):
|
||||
cur_ctxt = contexts[i].as_struct(Context)
|
||||
cur_ctxt.common.assign(common.reference())
|
||||
cur_ctxt.id.assign(i)
|
||||
cur_ctxt.completed.assign(
|
||||
self.constant_null(cur_ctxt.completed.type))
|
||||
|
||||
class ParallelUFuncPosixMixin(object):
|
||||
'''ParallelUFunc mixin that implements _dispatch_worker to use pthread.
|
||||
'''
|
||||
def _dispatch_worker(self, worker, contexts, num_thread):
|
||||
api = PThreadAPI(self)
|
||||
NULL = self.constant_null(C.void_p)
|
||||
|
||||
threads = self.array(api.pthread_t, num_thread, name='threads')
|
||||
|
||||
# self.debug("launch threads")
|
||||
# TODO error handling
|
||||
|
||||
ONE = self.constant(num_thread.type, 1)
|
||||
with self.for_range(num_thread) as (loop, i):
|
||||
api.pthread_create(threads[i].reference(), NULL, worker,
|
||||
contexts[i].reference().cast(C.void_p))
|
||||
|
||||
with self.for_range(num_thread) as (loop, i):
|
||||
api.pthread_join(threads[i], NULL)
|
||||
|
||||
class UFuncCore(CDefinition):
|
||||
'''core work of a ufunc worker thread
|
||||
|
||||
Subclass to implement UFuncCore._do_work
|
||||
|
||||
Generates the workqueue handling and work stealing and invoke
|
||||
the work function for each work item.
|
||||
'''
|
||||
_name_ = 'ufunc_worker'
|
||||
_argtys_ = [
|
||||
('context', C.pointer(Context.llvm_type())),
|
||||
]
|
||||
|
||||
def body(self, context):
|
||||
context = context.as_struct(Context)
|
||||
common = context.common.as_struct(ContextCommon)
|
||||
tid = context.id
|
||||
|
||||
# self.debug("start thread", tid, "/", common.num_thread)
|
||||
workqueue = common.workqueues[tid].as_struct(WorkQueue)
|
||||
|
||||
self._do_workqueue(common, workqueue, tid, context.completed)
|
||||
self._do_work_stealing(common, tid, context.completed) # optional
|
||||
|
||||
self.ret()
|
||||
|
||||
def _do_workqueue(self, common, workqueue, tid, completed):
|
||||
'''process local workqueue.
|
||||
'''
|
||||
ZERO = self.constant_null(C.int)
|
||||
|
||||
with self.forever() as loop:
|
||||
workqueue.Lock()
|
||||
# Critical section
|
||||
item = self.var_copy(workqueue.next, name='item')
|
||||
workqueue.next += self.constant(item.type, 1)
|
||||
last = self.var_copy(workqueue.last, name='last')
|
||||
# Release
|
||||
workqueue.Unlock()
|
||||
|
||||
with self.ifelse( item >= last ) as ifelse:
|
||||
with ifelse.then():
|
||||
loop.break_loop()
|
||||
|
||||
self._do_work(common, item, tid)
|
||||
completed += self.constant(completed.type, 1)
|
||||
|
||||
def _do_work_stealing(self, common, tid, completed):
|
||||
'''steal work from other workqueues.
|
||||
'''
|
||||
# self.debug("start work stealing", tid)
|
||||
steal_continue = self.var(C.int, 1)
|
||||
STEAL_STOP = self.constant_null(steal_continue.type)
|
||||
|
||||
# Loop until all workqueues are done.
|
||||
with self.loop() as loop:
|
||||
with loop.condition() as setcond:
|
||||
setcond( steal_continue != STEAL_STOP )
|
||||
|
||||
with loop.body():
|
||||
steal_continue.assign(STEAL_STOP)
|
||||
self._do_work_stealing_innerloop(common, steal_continue, tid,
|
||||
completed)
|
||||
|
||||
def _do_work_stealing_innerloop(self, common, steal_continue, tid,
|
||||
completed):
|
||||
'''loop over all other threads and try to steal work.
|
||||
'''
|
||||
with self.for_range(common.num_thread) as (loop, i):
|
||||
with self.ifelse( i != tid ) as ifelse:
|
||||
with ifelse.then():
|
||||
otherqueue = common.workqueues[i].as_struct(WorkQueue)
|
||||
self._do_work_stealing_check(common, otherqueue,
|
||||
steal_continue, tid,
|
||||
completed)
|
||||
|
||||
def _do_work_stealing_check(self, common, otherqueue, steal_continue, tid,
|
||||
completed):
|
||||
'''check the workqueue for any remaining work and steal it.
|
||||
'''
|
||||
otherqueue.Lock()
|
||||
# Acquired
|
||||
ONE = self.constant(otherqueue.last.type, 1)
|
||||
STEAL_CONTINUE = self.constant(steal_continue.type, 1)
|
||||
with self.ifelse(otherqueue.next < otherqueue.last) as ifelse:
|
||||
with ifelse.then():
|
||||
otherqueue.last -= ONE
|
||||
item = self.var_copy(otherqueue.last)
|
||||
|
||||
otherqueue.Unlock()
|
||||
# Released
|
||||
|
||||
self._do_work(common, item, tid)
|
||||
completed += self.constant(completed.type, 1)
|
||||
|
||||
# Mark incomplete thread
|
||||
steal_continue.assign(STEAL_CONTINUE)
|
||||
|
||||
with ifelse.otherwise():
|
||||
otherqueue.Unlock()
|
||||
# Released
|
||||
|
||||
def _do_work(self, common, item, tid):
|
||||
'''prepare to call the actual work function
|
||||
|
||||
Implementation depends on number and type of arguments.
|
||||
'''
|
||||
raise NotImplementedError
|
||||
|
||||
class SpecializedParallelUFunc(CDefinition):
|
||||
'''a generic ufunc that wraps ParallelUFunc, UFuncCore and the workload
|
||||
'''
|
||||
_argtys_ = [
|
||||
('args', C.pointer(C.char_p)),
|
||||
('dimensions', C.pointer(C.intp)),
|
||||
('steps', C.pointer(C.intp)),
|
||||
('data', C.void_p),
|
||||
]
|
||||
|
||||
def body(self, args, dimensions, steps, data,):
|
||||
pufunc = self.depends(self.PUFuncDef)
|
||||
core = self.depends(self.CoreDef)
|
||||
func = self.depends(self.FuncDef)
|
||||
to_void_p = lambda x: x.cast(C.void_p)
|
||||
pufunc(to_void_p(func), to_void_p(core), args, dimensions, steps, data)
|
||||
self.ret()
|
||||
|
||||
@classmethod
|
||||
def specialize(cls, pufunc_def, core_def, func_def):
|
||||
'''specialize to a combination of ParallelUFunc, UFuncCore and workload
|
||||
'''
|
||||
cls._name_ = 'specialized_%s_%s_%s'% (pufunc_def, core_def, func_def)
|
||||
cls.PUFuncDef = pufunc_def
|
||||
cls.CoreDef = core_def
|
||||
cls.FuncDef = func_def
|
||||
|
||||
class PThreadAPI(CExternal):
|
||||
'''external declaration of pthread API
|
||||
'''
|
||||
pthread_t = C.void_p
|
||||
|
||||
pthread_create = Type.function(C.int,
|
||||
[C.pointer(pthread_t), # thread_t
|
||||
C.void_p, # thread attr
|
||||
C.void_p, # function
|
||||
C.void_p]) # arg
|
||||
|
||||
pthread_join = Type.function(C.int, [C.void_p, C.void_p])
|
||||
|
||||
|
||||
class UFuncCoreGeneric(UFuncCore):
|
||||
'''A generic ufunc core worker from LLVM function type
|
||||
'''
|
||||
def _do_work(self, common, item, tid):
|
||||
ufunc_type = Type.function(self.RETTY, self.ARGTYS)
|
||||
ufunc_ptr = CFunc(self, common.func.cast(C.pointer(ufunc_type)).value)
|
||||
|
||||
get_offset = lambda B, S, T: B[item * S].reference().cast(C.pointer(T))
|
||||
|
||||
indata = []
|
||||
for i, argty in enumerate(self.ARGTYS):
|
||||
ptr = get_offset(common.args[i], common.steps[i], argty)
|
||||
indata.append(ptr.load())
|
||||
|
||||
out_index = len(self.ARGTYS)
|
||||
outptr = get_offset(common.args[out_index], common.steps[out_index],
|
||||
self.RETTY)
|
||||
|
||||
res = ufunc_ptr(*indata)
|
||||
outptr.store(res)
|
||||
|
||||
@classmethod
|
||||
def specialize(cls, fntype):
|
||||
'''specialize to a LLVM function type
|
||||
|
||||
fntype : a LLVM function type (llvm.core.FunctionType)
|
||||
'''
|
||||
cls._name_ = '.'.join([cls._name_] +
|
||||
map(str, [fntype.return_type] + fntype.args))
|
||||
|
||||
cls.RETTY = fntype.return_type
|
||||
cls.ARGTYS = tuple(fntype.args)
|
||||
|
||||
|
||||
if sys.platform not in ['win32']:
|
||||
class ParallelUFuncPlatform(ParallelUFunc, ParallelUFuncPosixMixin):
|
||||
pass
|
||||
else:
|
||||
raise NotImplementedError("Threading for %s" % sys.platform)
|
||||
|
||||
|
||||
|
||||
def parallel_vectorize_from_func(lfunc, engine=None):
|
||||
'''create ufunc from a llvm.core.Function
|
||||
|
||||
If engine is given, return a function object which can be called
|
||||
from python. (This needs Jay's numpy.fromfunc).
|
||||
Otherwise, return the specialized ufunc as a llvm.core.Function
|
||||
'''
|
||||
import multiprocessing
|
||||
NUM_CPU = multiprocessing.cpu_count()
|
||||
|
||||
fntype = lfunc.type.pointee
|
||||
def_spuf = SpecializedParallelUFunc(
|
||||
ParallelUFuncPlatform(num_thread=NUM_CPU),
|
||||
UFuncCoreGeneric(fntype),
|
||||
CFuncRef(lfunc))
|
||||
spuf = def_spuf(lfunc.module)
|
||||
if engine is None:
|
||||
return spuf
|
||||
else:
|
||||
import numpy as np
|
||||
|
||||
fptr = engine.get_pointer_to_function(spuf)
|
||||
inct = len(fntype.args)
|
||||
outct = 1
|
||||
|
||||
# TODO refactor
|
||||
typemap = {
|
||||
'i8' : np.int8,
|
||||
'i16' : np.int16,
|
||||
'i32' : np.int32,
|
||||
'i64' : np.int64,
|
||||
'float' : np.float32,
|
||||
'double' : np.float64,
|
||||
}
|
||||
try:
|
||||
ptr_t = long
|
||||
except:
|
||||
ptr_t = int
|
||||
assert False, "Having check this yet"
|
||||
|
||||
get_typenum = lambda T:np.dtype(typemap[str(T)]).num
|
||||
assert fntype.return_type != C.void
|
||||
tys = list(map(get_typenum, list(fntype.args) + [fntype.return_type]))
|
||||
# Becareful that fromfunc does not provide full error checking yet.
|
||||
# If typenum is out-of-bound, we have nasty memory corruptions.
|
||||
# For instance, -1 for typenum will cause segfault.
|
||||
# If elements of type-list (2nd arg) is tuple instead,
|
||||
# there will also memory corruption. (Seems like code rewrite.)
|
||||
return np.fromfunc([ptr_t(fptr)], [tys], inct, outct, [None])
|
||||
|
||||
|
|
@ -1,130 +0,0 @@
|
|||
from parallel_vectorize import *
|
||||
|
||||
|
||||
class Work_D_D(CDefinition):
|
||||
_name_ = 'work_d_d'
|
||||
_retty_ = C.double
|
||||
_argtys_ = [
|
||||
('inval', C.double),
|
||||
]
|
||||
def body(self, inval):
|
||||
self.ret(inval / self.constant(inval.type, 2.345))
|
||||
|
||||
class UFuncCore_D_D(UFuncCore):
|
||||
'''
|
||||
Specialize UFuncCore for double input, double output.
|
||||
'''
|
||||
_name_ = UFuncCore._name_ + '_d_d'
|
||||
def _do_work(self, common, item, tid):
|
||||
ufunc_type = Type.function(C.double, [C.double])
|
||||
ufunc_ptr = CFunc(self, common.func.cast(C.pointer(ufunc_type)).value)
|
||||
|
||||
inbase = common.args[0]
|
||||
outbase = common.args[1]
|
||||
|
||||
instep = common.steps[0]
|
||||
outstep = common.steps[1]
|
||||
|
||||
indata = inbase[item * instep].reference().cast(C.pointer(C.double))
|
||||
outdata = outbase[item * outstep].reference().cast(C.pointer(C.double))
|
||||
|
||||
res = ufunc_ptr(indata.load())
|
||||
outdata.store(res)
|
||||
|
||||
class ParallelUFuncPosix(ParallelUFunc, ParallelUFuncPosixMixin):
|
||||
pass
|
||||
|
||||
class Tester(CDefinition):
|
||||
'''
|
||||
Generate test.
|
||||
'''
|
||||
_name_ = 'tester'
|
||||
|
||||
def body(self):
|
||||
# depends
|
||||
module = self.function.module
|
||||
|
||||
ThreadCount = 2
|
||||
ArgCount = 2
|
||||
WorkCount = 10000
|
||||
|
||||
|
||||
spufdef = SpecializedParallelUFunc(ParallelUFuncPosix(num_thread=2),
|
||||
UFuncCore_D_D(),
|
||||
Work_D_D())
|
||||
|
||||
sppufunc = self.depends(spufdef)
|
||||
|
||||
# real work
|
||||
NULL = self.constant_null(C.void_p)
|
||||
|
||||
args = self.array(C.char_p, 2, name='args')
|
||||
|
||||
args_double = []
|
||||
for t in range(ThreadCount):
|
||||
args_for_thread = self.array(C.double, WorkCount)
|
||||
args[t].assign(args_for_thread.cast(C.char_p))
|
||||
args_double.append(args_for_thread)
|
||||
|
||||
dims = self.array(C.intp, 1, name='dims')
|
||||
dims[0].assign(self.constant(C.intp, WorkCount))
|
||||
|
||||
steps = self.array(C.intp, ArgCount, name='steps')
|
||||
|
||||
for c in range(ArgCount):
|
||||
steps[c].assign(self.constant(C.intp, 8))
|
||||
|
||||
# populate data
|
||||
inbase = args_double[0]
|
||||
|
||||
i = self.var(C.intp, 0)
|
||||
with self.loop() as loop:
|
||||
with loop.condition() as setcond:
|
||||
setcond( i < self.constant(C.intp, WorkCount) )
|
||||
with loop.body():
|
||||
inbase[i].assign(i.cast(C.double))
|
||||
i += self.constant(C.intp, 1)
|
||||
|
||||
# call parallel ufunc
|
||||
sppufunc(args, dims, steps, NULL)
|
||||
|
||||
# check error
|
||||
outbase = args_double[-1]
|
||||
with self.for_range(self.constant(C.intp, WorkCount)) as (loop, i):
|
||||
test = outbase[i] != (inbase[i] / self.constant(C.double, 2.345))
|
||||
with self.ifelse( test ) as ifelse:
|
||||
with ifelse.then():
|
||||
self.debug("Invalid data at i =", i, outbase[i], inbase[i])
|
||||
|
||||
self.ret()
|
||||
|
||||
def main():
|
||||
module = Module.new(__name__)
|
||||
|
||||
mpm = PassManager.new()
|
||||
pmbuilder = PassManagerBuilder.new()
|
||||
pmbuilder.opt_level = 3
|
||||
pmbuilder.populate(mpm)
|
||||
|
||||
fntester = Tester.define(module)
|
||||
|
||||
# print(module)
|
||||
module.verify()
|
||||
|
||||
mpm.run(module)
|
||||
|
||||
print('optimized'.center(80,'-'))
|
||||
print(module)
|
||||
|
||||
# run
|
||||
print('run')
|
||||
exe = CExecutor(module)
|
||||
func = exe.get_ctype_function(fntester, 'void')
|
||||
|
||||
func()
|
||||
# Will not reach here is race condition occurred
|
||||
print('Good')
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
||||
|
|
@ -1,57 +0,0 @@
|
|||
'''
|
||||
Test parallel-vectorize with numpy.fromfunc.
|
||||
Uses the work load from test_parallel_vectorize.
|
||||
'''
|
||||
|
||||
from test_parallel_vectorize import *
|
||||
|
||||
import numpy as np
|
||||
|
||||
def main():
|
||||
module = Module.new(__name__)
|
||||
|
||||
spufdef = SpecializedParallelUFunc(ParallelUFuncPosix(num_thread=2),
|
||||
UFuncCore_D_D(),
|
||||
Work_D_D())
|
||||
|
||||
sppufunc = spufdef(module)
|
||||
|
||||
module.verify()
|
||||
|
||||
mpm = PassManager.new()
|
||||
pmbuilder = PassManagerBuilder.new()
|
||||
pmbuilder.opt_level = 3
|
||||
pmbuilder.populate(mpm)
|
||||
|
||||
mpm.run(module)
|
||||
# print module
|
||||
|
||||
# run
|
||||
|
||||
exe = CExecutor(module)
|
||||
funcptr = exe.engine.get_pointer_to_function(sppufunc)
|
||||
print("Function pointer: %x" % funcptr)
|
||||
|
||||
ptr_t = long # py2 only
|
||||
|
||||
# Becareful that fromfunc does not provide full error checking yet.
|
||||
# If typenum is out-of-bound, we have nasty memory corruptions.
|
||||
# For instance, -1 for typenum will cause segfault.
|
||||
# If elements of type-list (2nd arg) is tuple instead,
|
||||
# there will also memory corruption. (Seems like code rewrite.)
|
||||
typenum = np.dtype(np.double).num
|
||||
ufunc = np.fromfunc([ptr_t(funcptr)], [[typenum, typenum]], 1, 1, [None])
|
||||
|
||||
x = np.linspace(0., 10., 1000)
|
||||
x.dtype=np.double
|
||||
# print x
|
||||
ans = ufunc(x)
|
||||
# print ans
|
||||
|
||||
if not ( ans == x/2.345 ).all():
|
||||
raise ValueError('Computation failed')
|
||||
else:
|
||||
print('Good')
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
|
@ -1,69 +0,0 @@
|
|||
'''
|
||||
Test parallel-vectorize with numpy.fromfunc.
|
||||
Uses the work load from test_parallel_vectorize.
|
||||
This time we pass a function pointer.
|
||||
'''
|
||||
|
||||
from test_parallel_vectorize import *
|
||||
|
||||
import numpy as np
|
||||
|
||||
def main():
|
||||
module = Module.new(__name__)
|
||||
exe = CExecutor(module)
|
||||
|
||||
workdef = Work_D_D()
|
||||
workfunc = workdef(module)
|
||||
|
||||
# get pointer to workfunc
|
||||
workfunc_ptr = exe.engine.get_pointer_to_function(workfunc)
|
||||
|
||||
workdecl = CFuncRef(workfunc.name, workfunc.type.pointee, workfunc_ptr)
|
||||
|
||||
spufdef = SpecializedParallelUFunc(ParallelUFuncPosix(num_thread=2),
|
||||
UFuncCore_D_D(),
|
||||
workdecl)
|
||||
|
||||
|
||||
sppufunc = spufdef(module)
|
||||
sppufunc.verify()
|
||||
print(sppufunc)
|
||||
module.verify()
|
||||
|
||||
mpm = PassManager.new()
|
||||
pmbuilder = PassManagerBuilder.new()
|
||||
pmbuilder.opt_level = 3
|
||||
pmbuilder.populate(mpm)
|
||||
|
||||
mpm.run(module)
|
||||
print(module)
|
||||
|
||||
|
||||
# run
|
||||
|
||||
funcptr = exe.engine.get_pointer_to_function(sppufunc)
|
||||
print("Function pointer: %x" % funcptr)
|
||||
|
||||
ptr_t = long # py2 only
|
||||
|
||||
# Becareful that fromfunc does not provide full error checking yet.
|
||||
# If typenum is out-of-bound, we have nasty memory corruptions.
|
||||
# For instance, -1 for typenum will cause segfault.
|
||||
# If elements of type-list (2nd arg) is tuple instead,
|
||||
# there will also memory corruption. (Seems like code rewrite.)
|
||||
typenum = np.dtype(np.double).num
|
||||
ufunc = np.fromfunc([ptr_t(funcptr)], [[typenum, typenum]], 1, 1, [None])
|
||||
|
||||
x = np.linspace(0., 10., 1000)
|
||||
x.dtype=np.double
|
||||
# print x
|
||||
ans = ufunc(x)
|
||||
# print ans
|
||||
|
||||
if not ( ans == x/2.345 ).all():
|
||||
raise ValueError('Computation failed')
|
||||
else:
|
||||
print('Good')
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
|
@ -1,56 +0,0 @@
|
|||
from parallel_vectorize import *
|
||||
from llvm_cbuilder import shortnames as C
|
||||
from llvm.core import *
|
||||
import numpy as np
|
||||
import unittest
|
||||
from random import random
|
||||
|
||||
class OneOne(CDefinition):
|
||||
|
||||
def body(self, inval):
|
||||
self.ret( (inval * inval).cast(self.OUT_TYPE) )
|
||||
|
||||
@classmethod
|
||||
def specialize(cls, itype, otype):
|
||||
cls._name_ = '.'.join(map(str, ['oneone', itype, otype]))
|
||||
cls._retty_ = otype
|
||||
cls._argtys_ = [
|
||||
('inval', itype),
|
||||
]
|
||||
cls.OUT_TYPE = otype
|
||||
|
||||
|
||||
class TestParallelVectorize(unittest.TestCase):
|
||||
def test_parallelvectorize_d_d(self):
|
||||
self.template(C.double, C.double)
|
||||
|
||||
def test_parallelvectorize_d_f(self):
|
||||
self.template(C.double, C.float)
|
||||
|
||||
def template(self, itype, otype):
|
||||
module = Module.new(__name__)
|
||||
exe = CExecutor(module)
|
||||
|
||||
def_oneone = OneOne(itype, otype)
|
||||
oneone = def_oneone(module)
|
||||
ufunc = parallel_vectorize_from_func(oneone, exe.engine)
|
||||
# print(module)
|
||||
module.verify()
|
||||
|
||||
x = np.linspace(.0, 10., 1000)
|
||||
x.dtype = np.double
|
||||
|
||||
ans = ufunc(x)
|
||||
gold = x * x
|
||||
|
||||
for x, y in zip(ans, gold):
|
||||
if y != 0:
|
||||
err = abs(x - y)/y
|
||||
self.assertLess(err, 1e-6)
|
||||
else:
|
||||
self.assertEqual(x, y)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
||||
|
|
@ -1,58 +0,0 @@
|
|||
from parallel_vectorize import *
|
||||
from llvm_cbuilder import shortnames as C
|
||||
from llvm.core import *
|
||||
import numpy as np
|
||||
import unittest
|
||||
from random import random
|
||||
|
||||
class TwoOne(CDefinition):
|
||||
|
||||
def body(self, a, b):
|
||||
self.ret( (a * b).cast(self.OUT_TYPE) )
|
||||
|
||||
@classmethod
|
||||
def specialize(cls, itype1, itype2, otype):
|
||||
cls._name_ = '.'.join(map(str, ['oneone', itype1, itype2, otype]))
|
||||
cls._retty_ = otype
|
||||
cls._argtys_ = [
|
||||
('a', itype1),
|
||||
('b', itype2),
|
||||
]
|
||||
cls.OUT_TYPE = otype
|
||||
|
||||
|
||||
class TestParallelVectorize(unittest.TestCase):
|
||||
def test_parallelvectorize_dd_d(self):
|
||||
self.template(C.double, C.double, C.double)
|
||||
|
||||
def test_parallelvectorize_dd_f(self):
|
||||
self.template(C.double, C.double, C.float)
|
||||
|
||||
def template(self, itype1, itype2, otype):
|
||||
module = Module.new(__name__)
|
||||
exe = CExecutor(module)
|
||||
|
||||
def_twoone = TwoOne(itype1, itype2, otype)
|
||||
twoone = def_twoone(module)
|
||||
ufunc = parallel_vectorize_from_func(twoone, exe.engine)
|
||||
# print(module)
|
||||
module.verify()
|
||||
|
||||
A = np.linspace(.0, 10., 1000)
|
||||
A.dtype = np.double
|
||||
B = np.linspace(-10., 0., 1000)
|
||||
B.dtype = np.double
|
||||
|
||||
ans = ufunc(A, B)
|
||||
gold = A * B
|
||||
|
||||
for x, y in zip(ans, gold):
|
||||
if y != 0:
|
||||
err = abs(x - y)/y
|
||||
self.assertLess(err, 1e-6)
|
||||
else:
|
||||
self.assertEqual(x, y)
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
||||
|
|
@ -1,115 +0,0 @@
|
|||
'''
|
||||
Base on the test_pthread.py and extend to use atomic instructions
|
||||
'''
|
||||
|
||||
from llvm.core import *
|
||||
from llvm.passes import *
|
||||
from llvm.ee import *
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
import unittest, logging
|
||||
|
||||
# logging.basicConfig(level=logging.DEBUG)
|
||||
|
||||
NUM_OF_THREAD = 4
|
||||
REPEAT = 10000
|
||||
|
||||
def gen_test_worker(mod):
|
||||
cb = CBuilder.new_function(mod, 'worker', C.void, [C.pointer(C.int)])
|
||||
pval = cb.args[0]
|
||||
one = cb.constant(pval.type.pointee, 1)
|
||||
|
||||
ct = cb.var(C.int, 0)
|
||||
limit = cb.constant(C.int, REPEAT)
|
||||
with cb.loop() as loop:
|
||||
with loop.condition() as setcond:
|
||||
setcond( ct < limit )
|
||||
|
||||
with loop.body():
|
||||
cb.atomic_add(pval, one, 'acq_rel')
|
||||
ct += one
|
||||
|
||||
cb.ret()
|
||||
cb.close()
|
||||
return cb.function
|
||||
|
||||
def gen_test_pthread(mod):
|
||||
cb = CBuilder.new_function(mod, 'manager', C.int, [C.int])
|
||||
arg = cb.args[0]
|
||||
|
||||
worker_func = cb.get_function_named('worker')
|
||||
pthread_create = cb.get_function_named('pthread_create')
|
||||
pthread_join = cb.get_function_named('pthread_join')
|
||||
|
||||
|
||||
NULL = cb.constant_null(C.void_p)
|
||||
cast_to_null = lambda x: x.cast(C.void_p)
|
||||
|
||||
threads = cb.array(C.void_p, NUM_OF_THREAD)
|
||||
|
||||
for tid in range(NUM_OF_THREAD):
|
||||
pthread_create_args = [threads[tid].reference(),
|
||||
NULL,
|
||||
worker_func,
|
||||
arg.reference()]
|
||||
pthread_create(*map(cast_to_null, pthread_create_args))
|
||||
|
||||
worker_func(arg.reference())
|
||||
|
||||
for tid in range(NUM_OF_THREAD):
|
||||
pthread_join_args = threads[tid], NULL
|
||||
pthread_join(*map(cast_to_null, pthread_join_args))
|
||||
|
||||
|
||||
cb.ret(arg)
|
||||
cb.close()
|
||||
return cb.function
|
||||
|
||||
class TestAtomicAdd(unittest.TestCase):
|
||||
def test_atomic_add(self):
|
||||
mod = Module.new(__name__)
|
||||
# add pthread functions
|
||||
|
||||
mod.add_function(Type.function(C.int,
|
||||
[C.void_p, C.void_p, C.void_p, C.void_p]),
|
||||
'pthread_create')
|
||||
|
||||
mod.add_function(Type.function(C.int,
|
||||
[C.void_p, C.void_p]),
|
||||
'pthread_join')
|
||||
|
||||
lf_test_worker = gen_test_worker(mod)
|
||||
lf_test_pthread = gen_test_pthread(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
# optimize
|
||||
fpm = FunctionPassManager.new(mod)
|
||||
mpm = PassManager.new()
|
||||
pmb = PassManagerBuilder.new()
|
||||
pmb.vectorize = True
|
||||
pmb.opt_level = 3
|
||||
pmb.populate(fpm)
|
||||
pmb.populate(mpm)
|
||||
|
||||
fpm.run(lf_test_worker)
|
||||
fpm.run(lf_test_pthread)
|
||||
mpm.run(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
# run
|
||||
exe = CExecutor(mod)
|
||||
exe.engine.get_pointer_to_function(mod.get_function_named('worker'))
|
||||
func = exe.get_ctype_function(lf_test_pthread, 'int, int')
|
||||
|
||||
inarg = 1234
|
||||
gold = inarg + (NUM_OF_THREAD + 1) * REPEAT
|
||||
|
||||
for _ in range(1000): # run many many times to catch race condition
|
||||
self.assertEqual(func(inarg), gold, "Unexpected race condition")
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
||||
|
|
@ -1,122 +0,0 @@
|
|||
'''
|
||||
Base on the test_pthread.py and extend to use atomic instructions
|
||||
'''
|
||||
|
||||
from llvm.core import *
|
||||
from llvm.passes import *
|
||||
from llvm.ee import *
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
import unittest, logging
|
||||
|
||||
# logging.basicConfig(level=logging.DEBUG)
|
||||
|
||||
NUM_OF_THREAD = 4
|
||||
REPEAT = 10000
|
||||
|
||||
def gen_test_worker(mod):
|
||||
cb = CBuilder.new_function(mod, 'worker', C.void, [C.pointer(C.int)])
|
||||
pval = cb.args[0]
|
||||
one = cb.constant(pval.type.pointee, 1)
|
||||
|
||||
ct = cb.var(C.int, 0)
|
||||
limit = cb.constant(C.int, REPEAT)
|
||||
with cb.loop() as loop:
|
||||
with loop.condition() as setcond:
|
||||
setcond( ct < limit )
|
||||
|
||||
with loop.body():
|
||||
oldval = pval.atomic_load('acquire')
|
||||
updated = oldval + one
|
||||
castmp = pval.atomic_cmpxchg(oldval, updated, 'release')
|
||||
|
||||
with cb.ifelse( castmp == oldval ) as ifelse:
|
||||
with ifelse.then():
|
||||
ct += one
|
||||
|
||||
cb.ret()
|
||||
cb.close()
|
||||
return cb.function
|
||||
|
||||
def gen_test_pthread(mod):
|
||||
cb = CBuilder.new_function(mod, 'manager', C.int, [C.int])
|
||||
arg = cb.args[0]
|
||||
|
||||
worker_func = cb.get_function_named('worker')
|
||||
pthread_create = cb.get_function_named('pthread_create')
|
||||
pthread_join = cb.get_function_named('pthread_join')
|
||||
|
||||
|
||||
NULL = cb.constant_null(C.void_p)
|
||||
cast_to_null = lambda x: x.cast(C.void_p)
|
||||
|
||||
threads = cb.array(C.void_p, NUM_OF_THREAD)
|
||||
|
||||
for tid in range(NUM_OF_THREAD):
|
||||
pthread_create_args = [threads[tid].reference(),
|
||||
NULL,
|
||||
worker_func,
|
||||
arg.reference()]
|
||||
pthread_create(*map(cast_to_null, pthread_create_args))
|
||||
|
||||
worker_func(arg.reference())
|
||||
|
||||
for tid in range(NUM_OF_THREAD):
|
||||
pthread_join_args = threads[tid], NULL
|
||||
pthread_join(*map(cast_to_null, pthread_join_args))
|
||||
|
||||
|
||||
cb.ret(arg)
|
||||
cb.close()
|
||||
return cb.function
|
||||
|
||||
class TestAtomicCmpXchg(unittest.TestCase):
|
||||
def test_atomic_cmpxchg(self):
|
||||
mod = Module.new(__name__)
|
||||
# add pthread functions
|
||||
|
||||
mod.add_function(Type.function(C.int,
|
||||
[C.void_p, C.void_p, C.void_p, C.void_p]),
|
||||
'pthread_create')
|
||||
|
||||
mod.add_function(Type.function(C.int,
|
||||
[C.void_p, C.void_p]),
|
||||
'pthread_join')
|
||||
|
||||
lf_test_worker = gen_test_worker(mod)
|
||||
lf_test_pthread = gen_test_pthread(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
# optimize
|
||||
fpm = FunctionPassManager.new(mod)
|
||||
mpm = PassManager.new()
|
||||
pmb = PassManagerBuilder.new()
|
||||
pmb.vectorize = True
|
||||
pmb.opt_level = 3
|
||||
pmb.populate(fpm)
|
||||
pmb.populate(mpm)
|
||||
|
||||
fpm.run(lf_test_worker)
|
||||
fpm.run(lf_test_pthread)
|
||||
mpm.run(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
# run
|
||||
exe = CExecutor(mod)
|
||||
exe.engine.get_pointer_to_function(mod.get_function_named('worker'))
|
||||
func = exe.get_ctype_function(lf_test_pthread, 'int, int')
|
||||
|
||||
inarg = 1234
|
||||
gold = inarg + (NUM_OF_THREAD + 1) * REPEAT
|
||||
|
||||
for _ in range(1000): # run many many times to catch race condition
|
||||
res = func(inarg)
|
||||
self.assertEqual(res, gold,
|
||||
"Unexpected race condition: res = %d" % res)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
||||
|
|
@ -1,17 +0,0 @@
|
|||
from llvm.core import *
|
||||
from llvm_cbuilder import *
|
||||
from llvm_cbuilder import shortnames as C
|
||||
|
||||
import unittest
|
||||
|
||||
class TestCstrCollide(unittest.TestCase):
|
||||
def test_same_string(self):
|
||||
mod = Module.new(__name__)
|
||||
cb = CBuilder.new_function(mod, 'test_cstr_collide', C.void, [])
|
||||
|
||||
a = cb.constant_string("hello")
|
||||
b = cb.constant_string("hello")
|
||||
self.assertEqual(a.value, b.value)
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
|
@ -1,125 +0,0 @@
|
|||
from llvm.core import *
|
||||
from llvm.passes import *
|
||||
from llvm.ee import *
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
import unittest, logging
|
||||
|
||||
def is_prime(x):
|
||||
if x <= 2:
|
||||
return True
|
||||
if (x % 2) == 0:
|
||||
return False
|
||||
for y in range(2, int(1 + x**0.5)):
|
||||
if (x % y) == 0:
|
||||
return False
|
||||
return True
|
||||
|
||||
def gen_is_prime(mod):
|
||||
functype = Type.function(C.int, [C.int])
|
||||
func = mod.add_function(functype, 'isprime')
|
||||
|
||||
cb = CBuilder(func)
|
||||
|
||||
arg = cb.args[0]
|
||||
|
||||
two = cb.constant(C.int, 2)
|
||||
true = one = cb.constant(C.int, 1)
|
||||
false = zero = cb.constant(C.int, 0)
|
||||
|
||||
with cb.ifelse( arg <= two ) as ifelse:
|
||||
with ifelse.then():
|
||||
cb.ret(true)
|
||||
|
||||
with cb.ifelse( (arg % two) == zero ) as ifelse:
|
||||
with ifelse.then():
|
||||
cb.ret(false)
|
||||
|
||||
idx = cb.var(C.int, 3, name='idx')
|
||||
with cb.loop() as loop:
|
||||
with loop.condition() as setcond:
|
||||
setcond( idx < arg )
|
||||
|
||||
with loop.body():
|
||||
with cb.ifelse( (arg % idx) == zero ) as ifelse:
|
||||
with ifelse.then():
|
||||
cb.ret(false)
|
||||
# increment
|
||||
idx += two
|
||||
|
||||
cb.ret(true)
|
||||
cb.close()
|
||||
return func
|
||||
|
||||
|
||||
def gen_is_prime_fast(mod):
|
||||
functype = Type.function(C.int, [C.int])
|
||||
func = mod.add_function(functype, 'isprime_fast')
|
||||
|
||||
cb = CBuilder(func)
|
||||
|
||||
arg = cb.args[0]
|
||||
|
||||
two = cb.constant(C.int, 2)
|
||||
true = one = cb.constant(C.int, 1)
|
||||
false = zero = cb.constant(C.int, 0)
|
||||
|
||||
with cb.ifelse( arg <= two ) as ifelse:
|
||||
with ifelse.then():
|
||||
cb.ret(true)
|
||||
|
||||
with cb.ifelse( (arg % two) == zero ) as ifelse:
|
||||
with ifelse.then():
|
||||
cb.ret(false)
|
||||
|
||||
idx = cb.var(C.int, 3, name='idx')
|
||||
|
||||
sqrt = cb.get_intrinsic(INTR_SQRT, [C.float])
|
||||
|
||||
looplimit = one + sqrt(arg.cast(C.float)).cast(C.int)
|
||||
|
||||
|
||||
with cb.loop() as loop:
|
||||
with loop.condition() as setcond:
|
||||
setcond( idx < looplimit )
|
||||
|
||||
with loop.body():
|
||||
with cb.ifelse( (arg % idx) == zero ) as ifelse:
|
||||
with ifelse.then():
|
||||
cb.ret(false)
|
||||
# increment
|
||||
idx += two
|
||||
|
||||
|
||||
cb.ret(true)
|
||||
cb.close()
|
||||
return func
|
||||
|
||||
class TestIsPrime(unittest.TestCase):
|
||||
def test_isprime(self):
|
||||
mod = Module.new(__name__)
|
||||
lf_isprime = gen_is_prime(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
exe = CExecutor(mod)
|
||||
func = exe.get_ctype_function(lf_isprime, 'bool, int')
|
||||
for x in range(2, 1000):
|
||||
msg = "Failed at x = %d" % x
|
||||
self.assertEqual(func(x), is_prime(x), msg)
|
||||
|
||||
def test_isprime_fast(self):
|
||||
mod = Module.new(__name__)
|
||||
lf_isprime = gen_is_prime_fast(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
exe = CExecutor(mod)
|
||||
func = exe.get_ctype_function(lf_isprime, 'bool, int')
|
||||
for x in range(2, 1000):
|
||||
msg = "Failed at x = %d" % x
|
||||
self.assertEqual(func(x), is_prime(x), msg)
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
||||
|
|
@ -1,135 +0,0 @@
|
|||
from llvm.core import *
|
||||
from llvm.passes import *
|
||||
from llvm.ee import *
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
import unittest, logging
|
||||
|
||||
def loopbreak(d):
|
||||
z = 0
|
||||
for x in range(100):
|
||||
for y in range(100):
|
||||
z += x + y
|
||||
if z > 50:
|
||||
break
|
||||
z -= d
|
||||
return z
|
||||
|
||||
def gen_loopbreak(mod):
|
||||
functype = Type.function(C.int, [C.int])
|
||||
func = mod.add_function(functype, 'loopbreak')
|
||||
|
||||
cb = CBuilder(func)
|
||||
|
||||
d = cb.args[0]
|
||||
x = cb.var(C.int)
|
||||
y = cb.var(C.int)
|
||||
z = cb.var(C.int)
|
||||
|
||||
one = cb.constant(C.int, 1)
|
||||
zero = cb.constant(C.int, 0)
|
||||
limit = cb.constant(C.int, 100)
|
||||
fifty = cb.constant(C.int, 50)
|
||||
|
||||
z.assign(zero)
|
||||
x.assign(zero)
|
||||
with cb.loop() as outer:
|
||||
with outer.condition() as setcond:
|
||||
setcond( x < limit )
|
||||
|
||||
with outer.body():
|
||||
y.assign(zero)
|
||||
with cb.loop() as inner:
|
||||
with inner.condition() as setcond:
|
||||
setcond( y < limit )
|
||||
|
||||
with inner.body():
|
||||
z += x + y
|
||||
with cb.ifelse( z > fifty ) as ifelse:
|
||||
with ifelse.then():
|
||||
inner.break_loop()
|
||||
y += one
|
||||
z -= d
|
||||
x += one
|
||||
|
||||
cb.ret(z)
|
||||
cb.close()
|
||||
return func
|
||||
|
||||
def loopcontinue(d):
|
||||
z = 0
|
||||
for x in range(100):
|
||||
for y in range(100):
|
||||
z += x + y
|
||||
if z > 50:
|
||||
continue
|
||||
z += d
|
||||
return z
|
||||
|
||||
def gen_loopcontinue(mod):
|
||||
functype = Type.function(C.int, [C.int])
|
||||
func = mod.add_function(functype, 'loopcontinue')
|
||||
|
||||
cb = CBuilder(func)
|
||||
|
||||
d = cb.args[0]
|
||||
x = cb.var(C.int)
|
||||
y = cb.var(C.int)
|
||||
z = cb.var(C.int)
|
||||
|
||||
one = cb.constant(C.int, 1)
|
||||
zero = cb.constant(C.int, 0)
|
||||
limit = cb.constant(C.int, 100)
|
||||
fifty = cb.constant(C.int, 50)
|
||||
|
||||
z.assign(zero)
|
||||
x.assign(zero)
|
||||
with cb.loop() as outer:
|
||||
with outer.condition() as setcond:
|
||||
setcond( x < limit )
|
||||
|
||||
with outer.body():
|
||||
y.assign(zero)
|
||||
with cb.loop() as inner:
|
||||
with inner.condition() as setcond:
|
||||
setcond( y < limit )
|
||||
|
||||
with inner.body():
|
||||
z += x + y
|
||||
y += one
|
||||
with cb.ifelse( z > fifty ) as ifelse:
|
||||
with ifelse.then():
|
||||
inner.continue_loop()
|
||||
z += d
|
||||
x += one
|
||||
|
||||
cb.ret(z)
|
||||
cb.close()
|
||||
return func
|
||||
|
||||
class TestLoopControl(unittest.TestCase):
|
||||
def test_loopbreak(self):
|
||||
mod = Module.new(__name__)
|
||||
lfunc = gen_loopbreak(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
exe = CExecutor(mod)
|
||||
func = exe.get_ctype_function(lfunc, 'int, int')
|
||||
for x in range(100):
|
||||
self.assertEqual(func(x), loopbreak(x))
|
||||
|
||||
def test_loopcontinue(self):
|
||||
mod = Module.new(__name__)
|
||||
lfunc = gen_loopcontinue(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
exe = CExecutor(mod)
|
||||
func = exe.get_ctype_function(lfunc, 'int, int')
|
||||
for x in range(100):
|
||||
self.assertEqual(func(x), loopcontinue(x))
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
||||
|
|
@ -1,128 +0,0 @@
|
|||
from llvm.core import *
|
||||
from llvm.passes import *
|
||||
from llvm.ee import *
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
import unittest, logging
|
||||
|
||||
def nestedloop1(d):
|
||||
z = 0
|
||||
for x in range(100):
|
||||
for y in range(100):
|
||||
z += x * d + int(y / d)
|
||||
return z
|
||||
|
||||
def gen_nestedloop1(mod):
|
||||
functype = Type.function(C.int, [C.int])
|
||||
func = mod.add_function(functype, 'nestedloop1')
|
||||
|
||||
cb = CBuilder(func)
|
||||
|
||||
d = cb.args[0]
|
||||
x = cb.var(C.int)
|
||||
y = cb.var(C.int)
|
||||
z = cb.var(C.int)
|
||||
|
||||
one = cb.constant(C.int, 1)
|
||||
zero = cb.constant(C.int, 0)
|
||||
limit = cb.constant(C.int, 100)
|
||||
|
||||
z.assign(zero)
|
||||
x.assign(zero)
|
||||
with cb.loop() as outer:
|
||||
with outer.condition() as setcond:
|
||||
setcond( x < limit )
|
||||
|
||||
with outer.body():
|
||||
y.assign(zero)
|
||||
with cb.loop() as inner:
|
||||
with inner.condition() as setcond:
|
||||
setcond( y < limit )
|
||||
|
||||
with inner.body():
|
||||
z += x * d + y / d
|
||||
y += one
|
||||
x += one
|
||||
|
||||
cb.ret(z)
|
||||
cb.close()
|
||||
return func
|
||||
|
||||
|
||||
def nestedloop2(d):
|
||||
z = 0
|
||||
for x in range(1, 100):
|
||||
for y in range(1, 100):
|
||||
if x > y:
|
||||
z += int(x / y) * d
|
||||
else:
|
||||
z += int(y / x) * d
|
||||
return z
|
||||
|
||||
def gen_nestedloop2(mod):
|
||||
functype = Type.function(C.int, [C.int])
|
||||
func = mod.add_function(functype, 'nestedloop2')
|
||||
|
||||
cb = CBuilder(func)
|
||||
|
||||
d = cb.args[0]
|
||||
x = cb.var(C.int)
|
||||
y = cb.var(C.int)
|
||||
z = cb.var(C.int)
|
||||
|
||||
one = cb.constant(C.int, 1)
|
||||
zero = cb.constant(C.int, 0)
|
||||
limit = cb.constant(C.int, 100)
|
||||
|
||||
z.assign(zero)
|
||||
x.assign(one)
|
||||
with cb.loop() as outer:
|
||||
with outer.condition() as setcond:
|
||||
setcond( x < limit )
|
||||
|
||||
with outer.body():
|
||||
y.assign(one)
|
||||
with cb.loop() as inner:
|
||||
with inner.condition() as setcond:
|
||||
setcond( y < limit )
|
||||
|
||||
with inner.body():
|
||||
with cb.ifelse(x > y) as ifelse:
|
||||
with ifelse.then():
|
||||
z += x / y * d
|
||||
with ifelse.otherwise():
|
||||
z += y / x * d
|
||||
y += one
|
||||
x += one
|
||||
|
||||
cb.ret(z)
|
||||
cb.close()
|
||||
return func
|
||||
|
||||
|
||||
class TestNestedLoop(unittest.TestCase):
|
||||
def test_nestedloop1(self):
|
||||
mod = Module.new(__name__)
|
||||
lfunc = gen_nestedloop1(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
exe = CExecutor(mod)
|
||||
func = exe.get_ctype_function(lfunc, 'int, int')
|
||||
for x in range(1, 100):
|
||||
self.assertEqual(func(x), int(nestedloop1(x)))
|
||||
|
||||
def test_nestedloop2(self):
|
||||
mod = Module.new(__name__)
|
||||
lfunc = gen_nestedloop2(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
exe = CExecutor(mod)
|
||||
func = exe.get_ctype_function(lfunc, 'int, int')
|
||||
for x in range(1, 100):
|
||||
self.assertEqual(func(x), int(nestedloop2(x)))
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
||||
|
|
@ -1,60 +0,0 @@
|
|||
from llvm.core import *
|
||||
from llvm.passes import *
|
||||
from llvm.ee import *
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
import sys, unittest, logging
|
||||
from subprocess import Popen, PIPE
|
||||
|
||||
def gen_debugprint(mod):
|
||||
functype = Type.function(C.void, [])
|
||||
func = mod.add_function(functype, 'debugprint')
|
||||
|
||||
cb = CBuilder(func)
|
||||
fmt = cb.constant_string("Show %d %.3f %.3e\n")
|
||||
|
||||
an_int = cb.constant(C.int, 123)
|
||||
a_float = cb.constant(C.double, 1.234)
|
||||
a_double = cb.constant(C.double, 1e-31)
|
||||
cb.printf(fmt, an_int, a_float, a_double)
|
||||
|
||||
cb.debug('an_int =', an_int, 'a_float =', a_float, 'a_double =', a_double)
|
||||
|
||||
cb.ret()
|
||||
cb.close()
|
||||
return func
|
||||
|
||||
def main_debugprint():
|
||||
# generate code
|
||||
mod = Module.new(__name__)
|
||||
lfunc = gen_debugprint(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
# run
|
||||
exe = CExecutor(mod)
|
||||
func = exe.get_ctype_function(lfunc, 'void')
|
||||
func()
|
||||
|
||||
class TestPrint(unittest.TestCase):
|
||||
def test_debugprint(self):
|
||||
p = Popen(["python", "test_print.py", "-child"], stdout=PIPE)
|
||||
p.wait()
|
||||
|
||||
lines = p.stdout.read().decode().splitlines(False)
|
||||
|
||||
expect = [
|
||||
'Show 123 1.234 1.000e-31',
|
||||
'an_int = 123 a_float = 1.234000e+00 a_double = 1.000000e-31',
|
||||
]
|
||||
self.assertEqual(expect, lines)
|
||||
|
||||
p.stdout.close()
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
if sys.argv[1] == '-child':
|
||||
main_debugprint()
|
||||
except IndexError:
|
||||
unittest.main()
|
||||
|
||||
|
||||
|
|
@ -1,90 +0,0 @@
|
|||
from llvm.core import *
|
||||
from llvm.passes import *
|
||||
from llvm.ee import *
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
import unittest, logging
|
||||
|
||||
# logging.basicConfig(level=logging.DEBUG)
|
||||
|
||||
NUM_OF_THREAD = 4
|
||||
|
||||
def gen_test_worker(mod):
|
||||
cb = CBuilder.new_function(mod, 'worker', C.void, [C.pointer(C.int)])
|
||||
pval = cb.args[0]
|
||||
val = pval.load()
|
||||
one = cb.constant(val.type, 1)
|
||||
pval.store(val + one)
|
||||
cb.ret()
|
||||
cb.close()
|
||||
|
||||
def gen_test_pthread(mod):
|
||||
cb = CBuilder.new_function(mod, 'manager', C.int, [C.int])
|
||||
arg = cb.args[0]
|
||||
|
||||
worker_func = cb.get_function_named('worker')
|
||||
pthread_create = cb.get_function_named('pthread_create')
|
||||
pthread_join = cb.get_function_named('pthread_join')
|
||||
|
||||
|
||||
NULL = cb.constant_null(C.void_p)
|
||||
cast_to_null = lambda x: x.cast(C.void_p)
|
||||
|
||||
threads = cb.array(C.void_p, NUM_OF_THREAD)
|
||||
|
||||
for tid in range(NUM_OF_THREAD):
|
||||
pthread_create_args = [threads[tid].reference(),
|
||||
NULL,
|
||||
worker_func,
|
||||
arg.reference()]
|
||||
pthread_create(*map(cast_to_null, pthread_create_args))
|
||||
|
||||
worker_func(arg.reference())
|
||||
|
||||
for tid in range(NUM_OF_THREAD):
|
||||
pthread_join_args = threads[tid], NULL
|
||||
pthread_join(*map(cast_to_null, pthread_join_args))
|
||||
|
||||
cb.ret(arg)
|
||||
cb.close()
|
||||
return cb.function
|
||||
|
||||
class TestPThread(unittest.TestCase):
|
||||
def test_pthread(self):
|
||||
mod = Module.new(__name__)
|
||||
# add pthread functions
|
||||
|
||||
mod.add_function(Type.function(C.int,
|
||||
[C.void_p, C.void_p, C.void_p, C.void_p]),
|
||||
'pthread_create')
|
||||
|
||||
mod.add_function(Type.function(C.int,
|
||||
[C.void_p, C.void_p]),
|
||||
'pthread_join')
|
||||
|
||||
gen_test_worker(mod)
|
||||
lf_test_pthread = gen_test_pthread(mod)
|
||||
logging.debug(mod)
|
||||
mod.verify()
|
||||
|
||||
exe = CExecutor(mod)
|
||||
exe.engine.get_pointer_to_function(mod.get_function_named('worker'))
|
||||
func = exe.get_ctype_function(lf_test_pthread, 'int, int')
|
||||
|
||||
inarg = 1234
|
||||
gold = inarg + NUM_OF_THREAD + 1
|
||||
self.assertLessEqual(func(inarg), gold)
|
||||
# Cannot determine the exact return value due to untamed race condition
|
||||
|
||||
count_race = 0
|
||||
for _ in range(2**12):
|
||||
if func(inarg) != gold:
|
||||
count_race += 1
|
||||
|
||||
if count_race > 0:
|
||||
logging.info("Race condition occured %d times.", count_race)
|
||||
logging.info("Race condition is expected.")
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
|
||||
|
|
@ -1,52 +0,0 @@
|
|||
from llvm.core import *
|
||||
from llvm_cbuilder import *
|
||||
import llvm_cbuilder.shortnames as C
|
||||
import unittest, ctypes
|
||||
|
||||
class Vector2D(CStruct):
|
||||
_fields_ = [
|
||||
('x', C.float),
|
||||
('y', C.float),
|
||||
]
|
||||
|
||||
class Vector2DCtype(ctypes.Structure):
|
||||
_fields_ = [
|
||||
('x', ctypes.c_float),
|
||||
('y', ctypes.c_float),
|
||||
]
|
||||
|
||||
def gen_vector2d_dist(mod):
|
||||
functype = Type.function(C.float, [C.pointer(Vector2D.llvm_type())])
|
||||
func = mod.add_function(functype, 'vector2d_dist')
|
||||
|
||||
cb = CBuilder(func)
|
||||
vec = cb.var(Vector2D, cb.args[0].load())
|
||||
dist = vec.x * vec.x + vec.y * vec.y
|
||||
|
||||
cb.ret(dist)
|
||||
cb.close()
|
||||
return func
|
||||
|
||||
|
||||
class TestStruct(unittest.TestCase):
|
||||
def test_vector2d_dist(self):
|
||||
# prepare module
|
||||
mod = Module.new('mod')
|
||||
lfunc = gen_vector2d_dist(mod)
|
||||
mod.verify()
|
||||
# run
|
||||
exe = CExecutor(mod)
|
||||
func = exe.get_ctype_function(lfunc, ctypes.c_float, ctypes.POINTER(Vector2DCtype))
|
||||
|
||||
from random import random
|
||||
pydist = lambda x, y: x * x + y * y
|
||||
for _ in range(100):
|
||||
x, y = random(), random()
|
||||
vec = Vector2DCtype(x=x, y=y)
|
||||
ans = func(ctypes.pointer(vec))
|
||||
gold = pydist(x, y)
|
||||
|
||||
self.assertLess(abs(ans-gold)/gold, 1e-6)
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue