Fix up code-highlighting sections.
This commit is contained in:
parent
415c01f745
commit
ce8884aa33
18 changed files with 5611 additions and 6037 deletions
|
|
@ -13,34 +13,29 @@ References to functions already present in a module can be retrieved via
|
||||||
``Function.get``. All functions in a module can be enumerated by
|
``Function.get``. All functions in a module can be enumerated by
|
||||||
iterating over ``module_obj.functions``.
|
iterating over ``module_obj.functions``.
|
||||||
|
|
||||||
{% highlight python %} # create a type, representing functions that take
|
|
||||||
an integer and return # a floating point value. ft = Type.function(
|
|
||||||
Type.float(), [ Type.int() ] )
|
|
||||||
|
|
||||||
create a function of this type
|
.. code-block:: python
|
||||||
==============================
|
|
||||||
|
|
||||||
f1 = module\_obj.add\_function(ft, "func1")
|
# create a type, representing functions that take
|
||||||
|
an integer and return # a floating point value. ft = Type.function(
|
||||||
|
Type.float(), [ Type.int() ] )
|
||||||
|
|
||||||
or equivalently, like this:
|
# create a function of this type
|
||||||
===========================
|
f1 = module_obj.add_function(ft, "func1")
|
||||||
|
|
||||||
f2 = Function.new(module\_obj, ft, "func2")
|
# or equivalently, like this:
|
||||||
|
f2 = Function.new(module_obj, ft, "func2")
|
||||||
|
|
||||||
get a reference to an existing function
|
# get a reference to an existing function
|
||||||
=======================================
|
f3 = module_obj.get_function_named("func3")
|
||||||
|
|
||||||
f3 = module\_obj.get\_function\_named("func3")
|
# or like this:
|
||||||
|
f4 = Function.get(module_obj, "func4")
|
||||||
|
|
||||||
or like this:
|
# list all function names in a module
|
||||||
=============
|
for f in module_obj.functions: print f.name
|
||||||
|
|
||||||
f4 = Function.get(module\_obj, "func4")
|
|
||||||
|
|
||||||
list all function names in a module
|
|
||||||
===================================
|
|
||||||
|
|
||||||
for f in module\_obj.functions: print f.name {% endhighlight %}
|
|
||||||
|
|
||||||
Intrinsic
|
Intrinsic
|
||||||
=========
|
=========
|
||||||
|
|
@ -52,13 +47,16 @@ called with a module object, an intrinsic ID (which is a numeric
|
||||||
constant) and a list of the types of arguments (which LLVM uses to
|
constant) and a list of the types of arguments (which LLVM uses to
|
||||||
resolve overloaded intrinsic functions).
|
resolve overloaded intrinsic functions).
|
||||||
|
|
||||||
{% highlight python %} # get a reference to the llvm.bswap intrinsic
|
|
||||||
bswap = Function.intrinsic(mod, INTR\_BSWAP, [Type.int()])
|
|
||||||
|
|
||||||
call it
|
.. code-block:: python
|
||||||
=======
|
|
||||||
|
# get a reference to the llvm.bswap intrinsic
|
||||||
|
bswap = Function.intrinsic(mod, INTR_BSWAP, [Type.int()])
|
||||||
|
|
||||||
|
# call it
|
||||||
|
builder.call(bswap, [value])
|
||||||
|
|
||||||
|
|
||||||
builder.call(bswap, [value]) {% endhighlight %}
|
|
||||||
|
|
||||||
Here, the constant ``INTR_BSWAP``, available from ``llvm.core``,
|
Here, the constant ``INTR_BSWAP``, available from ``llvm.core``,
|
||||||
represents the LLVM intrinsic
|
represents the LLVM intrinsic
|
||||||
|
|
@ -111,13 +109,16 @@ The value objects corresponding to the arguments of a function can be
|
||||||
got using the read-only property ``args``. These can be iterated over,
|
got using the read-only property ``args``. These can be iterated over,
|
||||||
and also be indexed via integers. An example:
|
and also be indexed via integers. An example:
|
||||||
|
|
||||||
{% highlight python %} # list all argument names and types for arg in
|
|
||||||
fn.args: print arg.name, "of type", arg.type
|
|
||||||
|
|
||||||
change the name of the first argument
|
.. code-block:: python
|
||||||
=====================================
|
|
||||||
|
# list all argument names and types for arg in
|
||||||
|
fn.args: print arg.name, "of type", arg.type
|
||||||
|
|
||||||
|
# change the name of the first argument
|
||||||
|
fn.args[0].name = "objptr"
|
||||||
|
|
||||||
|
|
||||||
fn.args[0].name = "objptr" {% endhighlight %}
|
|
||||||
|
|
||||||
Basic blocks (see later) are contained within functions. When newly
|
Basic blocks (see later) are contained within functions. When newly
|
||||||
created, a function has no basic blocks. They have to be added
|
created, a function has no basic blocks. They have to be added
|
||||||
|
|
@ -130,71 +131,19 @@ blocks can be got via ``basic_block_count`` method. Note that
|
||||||
``get_entry_basic_block`` is slightly faster than ``basic_blocks[0]``
|
``get_entry_basic_block`` is slightly faster than ``basic_blocks[0]``
|
||||||
and so is ``basic_block_count``, over ``len(f.basic_blocks)``.
|
and so is ``basic_block_count``, over ``len(f.basic_blocks)``.
|
||||||
|
|
||||||
{% highlight python %} # add a basic block b1 =
|
|
||||||
fn.append\_basic\_block("entry")
|
|
||||||
|
|
||||||
get the first one
|
.. code-block:: python
|
||||||
=================
|
|
||||||
|
|
||||||
b2 = fn.get\_entry\_basic\_block() b2 = fn.basic\_mdblocks[0] # slower
|
# add a basic block b1 =
|
||||||
than previous method
|
fn.append_basic_block("entry")
|
||||||
|
|
||||||
print names of all basic blocks
|
# get the first one
|
||||||
===============================
|
b2 = fn.get_entry_basic_block() b2 = fn.basic_mdblocks[0] # slower
|
||||||
|
than previous method
|
||||||
|
|
||||||
for b in fn.basic\_blocks: print b.name
|
# print names of all basic blocks
|
||||||
|
for b in fn.basic_blocks: print b.name
|
||||||
|
|
||||||
get number of basic blocks
|
# get number of basic blocks
|
||||||
==========================
|
n = fn.basic_block_count n = len(fn.basic_blocks) # slower than
|
||||||
|
previous method
|
||||||
n = fn.basic\_block\_count n = len(fn.basic\_blocks) # slower than
|
|
||||||
previous method {% endhighlight %}
|
|
||||||
|
|
||||||
Functions can be deleted using the method ``delete``. This deletes them
|
|
||||||
from their containing module. All references to the function object
|
|
||||||
should be dropped after ``delete`` has been called.
|
|
||||||
|
|
||||||
Functions can be verified with the ``verify`` method. Note that this may
|
|
||||||
not work properly (aborts on errors).
|
|
||||||
|
|
||||||
Function Attributes # {#fnattr}
|
|
||||||
===============================
|
|
||||||
|
|
||||||
Function attributes, as documented
|
|
||||||
`here <http://www.llvm.org/docs/LangRef.html#fnattrs>`_, can be set on
|
|
||||||
functions using the methods ``add_attribute`` and ``remove_attribute``.
|
|
||||||
The following values may be used to refer to the LLVM attributes:
|
|
||||||
|
|
||||||
Value \| Equivalent LLVM Assembly Keyword \|
|
|
||||||
------\|----------------------------------\|
|
|
||||||
``ATTR_ALWAYS_INLINE``\ \|\ ``alwaysinline`` \|
|
|
||||||
``ATTR_INLINE_HINT``\ \|\ ``inlinehint`` \|
|
|
||||||
``ATTR_NO_INLINE``\ \|\ ``noinline`` \|
|
|
||||||
``ATTR_OPTIMIZE_FOR_SIZE``\ \|\ ``optsize`` \|
|
|
||||||
``ATTR_NO_RETURN``\ \|\ ``noreturn`` \|
|
|
||||||
``ATTR_NO_UNWIND``\ \|\ ``nounwind`` \|
|
|
||||||
``ATTR_READ_NONE``\ \|\ ``readnone`` \|
|
|
||||||
``ATTR_READONLY``\ \|\ ``readonly`` \|
|
|
||||||
``ATTR_STACK_PROTECT``\ \|\ ``ssp`` \|
|
|
||||||
``ATTR_STACK_PROTECT_REQ``\ \|\ ``sspreq`` \|
|
|
||||||
``ATTR_NO_REDZONE``\ \|\ ``noredzone`` \|
|
|
||||||
``ATTR_NO_IMPLICIT_FLOAT``\ \|\ ``noimplicitfloat`` \|
|
|
||||||
``ATTR_NAKED``\ \|\ ``naked`` \|
|
|
||||||
|
|
||||||
Here is how attributes can be set and removed:
|
|
||||||
|
|
||||||
{% highlight python %} # create a function ti = Type.int(32) tf =
|
|
||||||
Type.function(ti, [ti, ti]) m = Module.new('mod') f =
|
|
||||||
m.add\_function(tf, 'sum') print f # declare i32 @sum(i32, i32)
|
|
||||||
|
|
||||||
add a couple of attributes
|
|
||||||
==========================
|
|
||||||
|
|
||||||
f.add\_attribute(ATTR\_NO\_UNWIND) f.add\_attribute(ATTR\_READONLY)
|
|
||||||
print f # declare i32 @sum(i32, i32) nounwind readonly {% endhighlight
|
|
||||||
%}
|
|
||||||
|
|
||||||
**Related Links**
|
|
||||||
|
|
||||||
`llvm.core.Function <llvm.core.Function.html>`_,
|
|
||||||
`llvm.core.Argument <llvm.core.Argument.html>`_
|
|
||||||
|
|
|
||||||
|
|
@ -78,8 +78,9 @@ object files be built with the ``-fPIC`` option (generate position
|
||||||
independent code). Be sure to use the ``--enable-pic`` option while
|
independent code). Be sure to use the ``--enable-pic`` option while
|
||||||
configuring LLVM (default is no PIC), like this:
|
configuring LLVM (default is no PIC), like this:
|
||||||
|
|
||||||
{% highlight bash %} ~/llvm$ ./configure --enable-pic --enable-optimized
|
.. code-block:: bash
|
||||||
{% endhighlight %}
|
|
||||||
|
$ ~/llvm ./configure --enable-pic --enable-optimized
|
||||||
|
|
||||||
llvm-config
|
llvm-config
|
||||||
-----------
|
-----------
|
||||||
|
|
@ -103,51 +104,8 @@ LLVM's 'configure'.
|
||||||
|
|
||||||
Get llvmpy and install it:
|
Get llvmpy and install it:
|
||||||
|
|
||||||
{% highlight bash %} $ git clone git@github.com:numba/llvmpy.git $ cd
|
|
||||||
llvmpy $ python setup.py install {% endhighlight %}
|
|
||||||
|
|
||||||
If you need to tell the build script where ``llvm-config`` is, do it
|
.. code-block:: bash
|
||||||
this way:
|
|
||||||
|
|
||||||
{% highlight bash %} $ python setup.py install --user
|
$ git clone git@github.com:numba/llvmpy.git $ cd
|
||||||
--llvm-config=/home/mdevan/llvm/Release/bin/llvm-config {% endhighlight
|
llvmpy $ python setup.py install
|
||||||
%}
|
|
||||||
|
|
||||||
To build a debug version of llvmpy, that links against the debug
|
|
||||||
libraries of LLVM, use this:
|
|
||||||
|
|
||||||
{% highlight bash %} $ python setup.py build -g
|
|
||||||
--llvm-config=/home/mdevan/llvm/Debug/bin/llvm-config $ python setup.py
|
|
||||||
install --user --llvm-config=/home/mdevan/llvm/Debug/bin/llvm-config {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
Be warned that debug binaries will be huge (100MB+) ! They are required
|
|
||||||
only if you need to debug into LLVM also.
|
|
||||||
|
|
||||||
``setup.py`` is a standard Python distutils script. See the Python
|
|
||||||
documentation regarding `Installing Python
|
|
||||||
Modules <http://docs.python.org/inst/inst.html>`_ and `Distributing
|
|
||||||
Python Modules <http://docs.python.org/dist/dist.html>`_ for more
|
|
||||||
information on such scripts.
|
|
||||||
|
|
||||||
|
|
||||||
Uninstall
|
|
||||||
==============
|
|
||||||
|
|
||||||
If you'd installed llvmpy with the ``--user`` option, then llvmpy
|
|
||||||
would be present under ``~/.local/lib/python2.7/site-packages``.
|
|
||||||
Otherwise, it might be under ``/usr/lib/python2.7/site-packages`` or
|
|
||||||
``/usr/local/lib/python2.7/site-packages``. The directory would vary
|
|
||||||
with your Python version and OS flavour. Look around.
|
|
||||||
|
|
||||||
Once you've located the site-packages directory, the modules and the
|
|
||||||
"egg" can be removed like so:
|
|
||||||
|
|
||||||
{% highlight bash %} $ rm -rf /llvm /llvm\_py-.egg-info {% endhighlight
|
|
||||||
%}
|
|
||||||
|
|
||||||
See the `Python
|
|
||||||
documentation <http://docs.python.org/install/index.html>`_ for more
|
|
||||||
information.
|
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
|
||||||
|
|
@ -112,23 +112,36 @@ This gives the language a very nice and simple syntax. For example, the
|
||||||
following simple example computes `Fibonacci
|
following simple example computes `Fibonacci
|
||||||
numbers <http://en.wikipedia.org/wiki/Fibonacci_number>`_:
|
numbers <http://en.wikipedia.org/wiki/Fibonacci_number>`_:
|
||||||
|
|
||||||
{% highlight python %} # Compute the x'th fibonacci number. def fib(x)
|
|
||||||
if x < 3 then 1 else fib(x-1)+fib(x-2)
|
|
||||||
|
|
||||||
This expression will compute the 40th number.
|
.. code-block::
|
||||||
=============================================
|
|
||||||
|
# Compute the x'th fibonacci number.
|
||||||
|
def fib(x):
|
||||||
|
if x < 3:
|
||||||
|
return 1
|
||||||
|
else:
|
||||||
|
return fib(x-1)+fib(x-2)
|
||||||
|
|
||||||
|
# This expression will compute the 40th number.
|
||||||
|
fib(40)
|
||||||
|
|
||||||
|
|
||||||
fib(40) {% endhighlight %}
|
|
||||||
|
|
||||||
We also allow Kaleidoscope to call into standard library functions (the
|
We also allow Kaleidoscope to call into standard library functions (the
|
||||||
LLVM JIT makes this completely trivial). This means that you can use the
|
LLVM JIT makes this completely trivial). This means that you can use the
|
||||||
'extern' keyword to define a function before you use it (this is also
|
'extern' keyword to define a function before you use it (this is also
|
||||||
useful for mutually recursive functions). For example:
|
useful for mutually recursive functions). For example:
|
||||||
|
|
||||||
{% highlight python %} extern sin(arg); extern cos(arg); extern
|
|
||||||
atan2(arg1 arg2);
|
|
||||||
|
|
||||||
atan2(sin(0.4), cos(42)) {% endhighlight %}
|
.. code-block::
|
||||||
|
|
||||||
|
extern sin(arg);
|
||||||
|
extern cos(arg);
|
||||||
|
extern atan2(arg1 arg2);
|
||||||
|
|
||||||
|
atan2(sin(0.4), cos(42))
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
A more interesting example is included in Chapter 6 where we write a
|
A more interesting example is included in Chapter 6 where we write a
|
||||||
little Kaleidoscope application that
|
little Kaleidoscope application that
|
||||||
|
|
@ -150,23 +163,32 @@ traditional way to do this is to use a
|
||||||
the lexer includes a token type and potentially some metadata (e.g. the
|
the lexer includes a token type and potentially some metadata (e.g. the
|
||||||
numeric value of a number). First, we define the possibilities:
|
numeric value of a number). First, we define the possibilities:
|
||||||
|
|
||||||
{% highlight python %} # The lexer yields one of these types for each
|
|
||||||
token. class EOFToken(object): pass
|
|
||||||
|
|
||||||
class DefToken(object): pass
|
.. code-block:: python
|
||||||
|
|
||||||
class ExternToken(object): pass
|
# The lexer yields one of these types for each token.
|
||||||
|
class EOFToken(object): pass
|
||||||
|
|
||||||
class IdentifierToken(object): def **init**\ (self, name): self.name =
|
class DefToken(object): pass
|
||||||
name
|
|
||||||
|
|
||||||
class NumberToken(object): def **init**\ (self, value): self.value =
|
class ExternToken(object): pass
|
||||||
value
|
|
||||||
|
class IdentifierToken(object):
|
||||||
|
def __init__(self, name):
|
||||||
|
self.name = name
|
||||||
|
|
||||||
|
class NumberToken(object):
|
||||||
|
def __init__(self, value):
|
||||||
|
self.value = value
|
||||||
|
|
||||||
|
class CharacterToken(object):
|
||||||
|
def __init__(self, char):
|
||||||
|
self.char = char
|
||||||
|
def __eq__(self, other):
|
||||||
|
return isinstance(other, CharacterToken) and self.char == other.char
|
||||||
|
def __ne__(self, other):
|
||||||
|
return not self == other
|
||||||
|
|
||||||
class CharacterToken(object): def **init**\ (self, char): self.char =
|
|
||||||
char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
|
||||||
== other {% endhighlight %}
|
|
||||||
|
|
||||||
Each token yielded by our lexer will be of one of the above types. For
|
Each token yielded by our lexer will be of one of the above types. For
|
||||||
simple tokens that are always the same, like the "def" keyword, the
|
simple tokens that are always the same, like the "def" keyword, the
|
||||||
|
|
@ -193,82 +215,109 @@ digits. Identifiers (and keywords) are alphanumeric string starting with
|
||||||
a letter and comments are anything between a hash (``#``) and the end of
|
a letter and comments are anything between a hash (``#``) and the end of
|
||||||
the line.
|
the line.
|
||||||
|
|
||||||
{% highlight python %} import re
|
|
||||||
|
|
||||||
...
|
.. code-block:: python
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
import re
|
||||||
=============================================================
|
|
||||||
|
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
...
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
# Regular expressions that tokens and comments of our language.
|
||||||
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?')
|
||||||
|
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]\ *')
|
||||||
|
REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
Next, let's start defining the ``Tokenize`` function itself. The first
|
Next, let's start defining the ``Tokenize`` function itself. The first
|
||||||
thing we need to do is set up a loop that scans the string, while
|
thing we need to do is set up a loop that scans the string, while
|
||||||
ignoring whitespace between tokens:
|
ignoring whitespace between tokens:
|
||||||
|
|
||||||
{% highlight python %} def Tokenize(string): while string: # Skip
|
|
||||||
whitespace. if string[0].isspace(): string = string[1:] continue
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
def Tokenize(string):
|
||||||
|
while string: # Skip whitespace.
|
||||||
|
if string[0].isspace():
|
||||||
|
string = string[1:]
|
||||||
|
continue
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
...
|
...
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Next we want to find out what the next token is. For this we run the
|
Next we want to find out what the next token is. For this we run the
|
||||||
regexes we defined above on the remainder of the string. To simplify the
|
regexes we defined above on the remainder of the string. To simplify the
|
||||||
rest of the code, we run all three regexes each time. As mentioned
|
rest of the code, we run all three regexes each time. As mentioned
|
||||||
above, inefficiencies are ignored for the purpose of this tutorial:
|
above, inefficiencies are ignored for the purpose of this tutorial:
|
||||||
|
|
||||||
{% highlight python %} # Run regexes. comment\_match =
|
|
||||||
REGEX\_COMMENT.match(string) number\_match = REGEX\_NUMBER.match(string)
|
|
||||||
identifier\_match = REGEX\_IDENTIFIER.match(string) {% endhighlight %}
|
|
||||||
|
|
||||||
Now se check if any of the regexes matched. For comments, we simply
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Run regexes.
|
||||||
|
comment_match = REGEX_COMMENT.match(string)
|
||||||
|
number_match = REGEX_NUMBER.match(string)
|
||||||
|
identifier_match = REGEX_IDENTIFIER.match(string)
|
||||||
|
|
||||||
|
|
||||||
|
Now we check if any of the regexes matched. For comments, we simply
|
||||||
ignore the captured match:
|
ignore the captured match:
|
||||||
|
|
||||||
{% highlight python %} # Check if any of the regexes matched and yield
|
|
||||||
the appropriate result. if comment\_match: comment =
|
.. code-block:: python
|
||||||
comment\_match.group(0) string = string[len(comment):] {% endhighlight
|
|
||||||
python %}
|
# Check if any of the regexes matched and yield
|
||||||
|
# the appropriate result.
|
||||||
|
if comment_match:
|
||||||
|
comment = comment_match.group(0)
|
||||||
|
string = string[len(comment):]
|
||||||
|
|
||||||
For numbers, we yield the captured match, converted to a float and
|
For numbers, we yield the captured match, converted to a float and
|
||||||
tagged with the appropriate token type:
|
tagged with the appropriate token type:
|
||||||
|
|
||||||
{% highlight python %} elif number\_match: number =
|
.. code-block:: python
|
||||||
number\_match.group(0) yield NumberToken(float(number)) string =
|
|
||||||
string[len(number):] {% endhighlight %}
|
elif number_match:
|
||||||
|
number = number_match.group(0)
|
||||||
|
yield NumberToken(float(number))
|
||||||
|
string = string[len(number):]
|
||||||
|
|
||||||
The identifier case is a little more complex. We have to check for
|
The identifier case is a little more complex. We have to check for
|
||||||
keywords to decide whether we have captured an identifier or a keyword:
|
keywords to decide whether we have captured an identifier or a keyword:
|
||||||
|
|
||||||
{% highlight python %} elif identifier\_match: identifier =
|
.. code-block:: python
|
||||||
identifier\_match.group(0) # Check if we matched a keyword. if
|
|
||||||
identifier == 'def': yield DefToken() elif identifier == 'extern': yield
|
elif identifier_match:
|
||||||
ExternToken() else: yield IdentifierToken(identifier) string =
|
identifier = identifier_match.group(0)
|
||||||
string[len(identifier):] {% endhighlight %}
|
# Check if we matched a keyword.
|
||||||
|
if identifier == 'def':
|
||||||
|
yield DefToken()
|
||||||
|
elif identifier == 'extern':
|
||||||
|
yield ExternToken()
|
||||||
|
else:
|
||||||
|
yield IdentifierToken(identifier)
|
||||||
|
string = string[len(identifier):]
|
||||||
|
|
||||||
|
|
||||||
Finally, if we haven't recognized a comment, a number of an identifier,
|
Finally, if we haven't recognized a comment, a number of an identifier,
|
||||||
we yield the current character as an "unknown character" token. This is
|
we yield the current character as an "unknown character" token. This is
|
||||||
used, for example, for operators like ``+`` or ``*``:
|
used, for example, for operators like ``+`` or ``*``:
|
||||||
|
|
||||||
{% highlight python %} else: # Yield the unknown character. yield
|
|
||||||
CharacterToken(string[0]) string = string[1:] {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
else: # Yield the unknown character.
|
||||||
|
yield CharacterToken(string[0])
|
||||||
|
string = string[1:]
|
||||||
|
|
||||||
|
|
||||||
Once we're done with the loop, we return a final end-of-file token:
|
Once we're done with the loop, we return a final end-of-file token:
|
||||||
|
|
||||||
{% highlight python %} yield EOFToken() {% endhighlight %}
|
|
||||||
|
|
||||||
With this, we have the complete lexer for the basic Kaleidoscope
|
.. code-block:: python
|
||||||
language (the `full code listing <PythonLangImpl2.html#code>`_ for the
|
|
||||||
Lexer is available in the `next chapter <PythonLangImpl2.html>`_ of the
|
|
||||||
tutorial). Next we'll `build a simple parser that uses this to build an
|
|
||||||
Abstract Syntax Tree <PythonLangImpl2.html>`_. When we have that, we'll
|
|
||||||
include a driver so that you can use the lexer and parser together.
|
|
||||||
|
|
||||||
--------------
|
yield EOFToken()
|
||||||
|
|
||||||
**`Next: Implementing a Parser and AST <PythonLangImpl2.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -36,16 +36,19 @@ language, and the AST should closely model the language. In
|
||||||
Kaleidoscope, we have expressions, a prototype, and a function object.
|
Kaleidoscope, we have expressions, a prototype, and a function object.
|
||||||
We'll start with expressions first:
|
We'll start with expressions first:
|
||||||
|
|
||||||
{% highlight python %} # Base class for all expression nodes. class
|
|
||||||
ExpressionNode(object): pass
|
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
.. code-block:: python
|
||||||
=================================================
|
|
||||||
|
# Base class for all expression nodes. class
|
||||||
|
ExpressionNode(object): pass
|
||||||
|
|
||||||
|
# Expression class for numeric literals like "1.0".
|
||||||
|
class NumberExpressionNode(ExpressionNode): def **init**\ (self, value):
|
||||||
|
self.value = value
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode): def **init**\ (self, value):
|
|
||||||
self.value = value
|
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
The code above shows the definition of the base ExpressionNode class and
|
The code above shows the definition of the base ExpressionNode class and
|
||||||
one subclass which we use for numeric literals. The important thing to
|
one subclass which we use for numeric literals. The important thing to
|
||||||
|
|
@ -58,22 +61,23 @@ them. It would be very easy to add a virtual method to pretty print the
|
||||||
code, for example. Here are the other expression AST node definitions
|
code, for example. Here are the other expression AST node definitions
|
||||||
that we'll use in the basic form of the Kaleidoscope language:
|
that we'll use in the basic form of the Kaleidoscope language:
|
||||||
|
|
||||||
{% highlight python %} # Expression class for referencing a variable,
|
|
||||||
like "a". class VariableExpressionNode(ExpressionNode): def
|
|
||||||
**init**\ (self, name): self.name = name
|
|
||||||
|
|
||||||
Expression class for a binary operator.
|
.. code-block:: python
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode): def **init**\ (self,
|
# Expression class for referencing a variable,
|
||||||
operator, left, right): self.operator = operator self.left = left
|
like "a". class VariableExpressionNode(ExpressionNode): def
|
||||||
self.right = right
|
**init**\ (self, name): self.name = name
|
||||||
|
|
||||||
|
# Expression class for a binary operator.
|
||||||
|
class BinaryOperatorExpressionNode(ExpressionNode): def **init**\ (self,
|
||||||
|
operator, left, right): self.operator = operator self.left = left
|
||||||
|
self.right = right
|
||||||
|
|
||||||
|
# Expression class for function calls.
|
||||||
|
class CallExpressionNode(ExpressionNode): def **init**\ (self, callee,
|
||||||
|
args): self.callee = callee self.args = args
|
||||||
|
|
||||||
Expression class for function calls.
|
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode): def **init**\ (self, callee,
|
|
||||||
args): self.callee = callee self.args = args {% endhighlight %}
|
|
||||||
|
|
||||||
This is all (intentionally) rather straight-forward: variables capture
|
This is all (intentionally) rather straight-forward: variables capture
|
||||||
the variable name, binary operators capture their opcode (e.g. '+'), and
|
the variable name, binary operators capture their opcode (e.g. '+'), and
|
||||||
|
|
@ -89,17 +93,20 @@ Turing-complete; we'll fix that in a later installment. The two things
|
||||||
we need next are a way to talk about the interface to a function, and a
|
we need next are a way to talk about the interface to a function, and a
|
||||||
way to talk about functions themselves:
|
way to talk about functions themselves:
|
||||||
|
|
||||||
{% highlight python %} # This class represents the "prototype" for a
|
|
||||||
function, which captures its name, # and its argument names (thus
|
|
||||||
implicitly the number of arguments the function # takes). class
|
|
||||||
PrototypeNode(object): def **init**\ (self, name, args): self.name =
|
|
||||||
name self.args = args
|
|
||||||
|
|
||||||
This class represents a function definition itself.
|
.. code-block:: python
|
||||||
===================================================
|
|
||||||
|
# This class represents the "prototype" for a
|
||||||
|
function, which captures its name, # and its argument names (thus
|
||||||
|
implicitly the number of arguments the function # takes). class
|
||||||
|
PrototypeNode(object): def **init**\ (self, name, args): self.name =
|
||||||
|
name self.args = args
|
||||||
|
|
||||||
|
# This class represents a function definition itself.
|
||||||
|
class FunctionNode(object): def **init**\ (self, prototype, body):
|
||||||
|
self.prototype = prototype self.body = body
|
||||||
|
|
||||||
|
|
||||||
class FunctionNode(object): def **init**\ (self, prototype, body):
|
|
||||||
self.prototype = prototype self.body = body {% endhighlight %}
|
|
||||||
|
|
||||||
In Kaleidoscope, functions are typed with just a count of their
|
In Kaleidoscope, functions are typed with just a count of their
|
||||||
arguments. Since all values are double precision floating point, the
|
arguments. Since all values are double precision floating point, the
|
||||||
|
|
@ -120,22 +127,32 @@ build it. The idea here is that we want to parse something like
|
||||||
``x + y`` (which is returned as three tokens by the lexer) into an AST
|
``x + y`` (which is returned as three tokens by the lexer) into an AST
|
||||||
that could be generated with calls like this:
|
that could be generated with calls like this:
|
||||||
|
|
||||||
{% highlight python %} x = VariableExpressionNode('x') y =
|
|
||||||
VariableExpressionNode('y') result = BinaryOperatorExpressionNode('+',
|
.. code-block:: python
|
||||||
x, y) {% endhighlight %}
|
|
||||||
|
x = VariableExpressionNode('x') y =
|
||||||
|
VariableExpressionNode('y') result = BinaryOperatorExpressionNode('+',
|
||||||
|
x, y)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In order to do this, we'll start by defining a lightweight ``Parser``
|
In order to do this, we'll start by defining a lightweight ``Parser``
|
||||||
class with some basic helper routines:
|
class with some basic helper routines:
|
||||||
|
|
||||||
{% highlight python %} class Parser(object):
|
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
.. code-block:: python
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
|
||||||
|
class Parser(object):
|
||||||
|
|
||||||
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
|
and # updates Parser.current with its results. def Next(self):
|
||||||
|
self.current = self.tokens.next()
|
||||||
|
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
|
||||||
and # updates Parser.current with its results. def Next(self):
|
|
||||||
self.current = self.tokens.next() {% endhighlight %}
|
|
||||||
|
|
||||||
This implements a simple token buffer around the lexer. This allows us
|
This implements a simple token buffer around the lexer. This allows us
|
||||||
to look one token ahead at what the lexer is returning. Every function
|
to look one token ahead at what the lexer is returning. Every function
|
||||||
|
|
@ -157,9 +174,14 @@ We start with numeric literals, because they are the simplest to
|
||||||
process. For each production in our grammar, we'll define a function
|
process. For each production in our grammar, we'll define a function
|
||||||
which parses that production. For numeric literals, we have:
|
which parses that production. For numeric literals, we have:
|
||||||
|
|
||||||
{% highlight python %} # numberexpr ::= number def
|
|
||||||
ParseNumberExpr(self): result = NumberExpressionNode(self.current.value)
|
.. code-block:: python
|
||||||
self.Next() # consume the number. return result {% endhighlight %}
|
|
||||||
|
# numberexpr ::= number def
|
||||||
|
ParseNumberExpr(self): result = NumberExpressionNode(self.current.value)
|
||||||
|
self.Next() # consume the number. return result
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This method is very simple: it expects to be called when the current
|
This method is very simple: it expects to be called when the current
|
||||||
token is a ``NumberToken``. It takes the current number value, creates a
|
token is a ``NumberToken``. It takes the current number value, creates a
|
||||||
|
|
@ -173,10 +195,13 @@ not part of the grammar production) ready to go. This is a fairly
|
||||||
standard way to go for recursive descent parsers. For a better example,
|
standard way to go for recursive descent parsers. For a better example,
|
||||||
the parenthesis operator is defined like this:
|
the parenthesis operator is defined like this:
|
||||||
|
|
||||||
{% highlight python %} # parenexpr ::= '(' expression ')' def
|
|
||||||
ParseParenExpr(self): self.Next() # eat '('.
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
# parenexpr ::= '(' expression ')' def
|
||||||
|
ParseParenExpr(self): self.Next() # eat '('.
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
contents = self.ParseExpression()
|
contents = self.ParseExpression()
|
||||||
|
|
||||||
|
|
@ -186,7 +211,9 @@ ParseParenExpr(self): self.Next() # eat '('.
|
||||||
|
|
||||||
return contents
|
return contents
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This function illustrates an interesting aspect of the parser. The
|
This function illustrates an interesting aspect of the parser. The
|
||||||
function uses recursion by calling ``ParseExpression`` (we will soon see
|
function uses recursion by calling ``ParseExpression`` (we will soon see
|
||||||
|
|
@ -201,11 +228,14 @@ needed.
|
||||||
The next simple production is for handling variable references and
|
The next simple production is for handling variable references and
|
||||||
function calls:
|
function calls:
|
||||||
|
|
||||||
{% highlight python %} # identifierexpr ::= identifier \| identifier '('
|
|
||||||
expression\* ')' def ParseIdentifierExpr(self): identifier\_name =
|
|
||||||
self.current.name self.Next() # eat identifier.
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
# identifierexpr ::= identifier \| identifier '('
|
||||||
|
expression\* ')' def ParseIdentifierExpr(self): identifier_name =
|
||||||
|
self.current.name self.Next() # eat identifier.
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
if self.current != CharacterToken('('): # Simple variable reference.
|
if self.current != CharacterToken('('): # Simple variable reference.
|
||||||
return VariableExpressionNode(identifier_name);
|
return VariableExpressionNode(identifier_name);
|
||||||
|
|
@ -225,7 +255,9 @@ self.current.name self.Next() # eat identifier.
|
||||||
self.Next() # eat ')'.
|
self.Next() # eat ')'.
|
||||||
return CallExpressionNode(identifier_name, args)
|
return CallExpressionNode(identifier_name, args)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This routine follows the same style as the other routines. It expects to
|
This routine follows the same style as the other routines. It expects to
|
||||||
be called if the current token is an ``IdentifierToken``. It also has
|
be called if the current token is an ``IdentifierToken``. It also has
|
||||||
|
|
@ -243,13 +275,18 @@ that will become more clear `later in the
|
||||||
tutorial <PythonLangImpl6.html#unary>`_. In order to parse an arbitrary
|
tutorial <PythonLangImpl6.html#unary>`_. In order to parse an arbitrary
|
||||||
primary expression, we need to determine what sort of expression it is:
|
primary expression, we need to determine what sort of expression it is:
|
||||||
|
|
||||||
{% highlight python %} # primary ::= identifierexpr \| numberexpr \|
|
|
||||||
parenexpr def ParsePrimary(self): if isinstance(self.current,
|
.. code-block:: python
|
||||||
IdentifierToken): return self.ParseIdentifierExpr() elif
|
|
||||||
isinstance(self.current, NumberToken): return self.ParseNumberExpr();
|
# primary ::= identifierexpr \| numberexpr \|
|
||||||
elif self.current == CharacterToken('('): return self.ParseParenExpr()
|
parenexpr def ParsePrimary(self): if isinstance(self.current,
|
||||||
else: raise RuntimeError('Unknown token when expecting an expression.')
|
IdentifierToken): return self.ParseIdentifierExpr() elif
|
||||||
{% endhighlight %}
|
isinstance(self.current, NumberToken): return self.ParseNumberExpr();
|
||||||
|
elif self.current == CharacterToken('('): return self.ParseParenExpr()
|
||||||
|
else: raise RuntimeError('Unknown token when expecting an expression.')
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Now that you see the definition of this function, it is more obvious why
|
Now that you see the definition of this function, it is more obvious why
|
||||||
we can assume the state of ``Parser.current`` in the various functions.
|
we can assume the state of ``Parser.current`` in the various functions.
|
||||||
|
|
@ -278,19 +315,24 @@ recursion. To start with, we need a table of precedences. Remember the
|
||||||
``binop_precedence`` parameter we passed to the ``Parser`` constructor?
|
``binop_precedence`` parameter we passed to the ``Parser`` constructor?
|
||||||
Now is the time to use it:
|
Now is the time to use it:
|
||||||
|
|
||||||
{% highlight python %} def main(): # Install standard binary operators.
|
|
||||||
# 1 is lowest possible precedence. 40 is the highest.
|
|
||||||
operator\_precedence = { '<': 10, '+': 20, '-': 20, '\*': 40 }
|
|
||||||
|
|
||||||
# Run the main ``interpreter loop``. while True:
|
.. code-block:: python
|
||||||
|
|
||||||
::
|
def main(): # Install standard binary operators.
|
||||||
|
# 1 is lowest possible precedence. 40 is the highest.
|
||||||
|
operator_precedence = { '<': 10, '+': 20, '-': 20, '\*': 40 }
|
||||||
|
|
||||||
|
# Run the main ``interpreter loop``. while True:
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
...
|
...
|
||||||
|
|
||||||
parser = Parser(Tokenize(raw), operator_precedence)
|
parser = Parser(Tokenize(raw), operator_precedence)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
For the basic form of Kaleidoscope, we will only support 4 binary
|
For the basic form of Kaleidoscope, we will only support 4 binary
|
||||||
operators (this can obviously be extended by you, our brave and intrepid
|
operators (this can obviously be extended by you, our brave and intrepid
|
||||||
|
|
@ -302,11 +344,16 @@ hardcode the comparisons.
|
||||||
We also define a helper function to get the precedence of the current
|
We also define a helper function to get the precedence of the current
|
||||||
token, or -1 if the token is not a binary operator:
|
token, or -1 if the token is not a binary operator:
|
||||||
|
|
||||||
{% highlight python %} # Gets the precedence of the current token, or -1
|
|
||||||
if the token is not a binary # operator. def
|
.. code-block:: python
|
||||||
GetCurrentTokenPrecedence(self): if isinstance(self.current,
|
|
||||||
CharacterToken): return self.binop\_precedence.get(self.current.char,
|
# Gets the precedence of the current token, or -1
|
||||||
-1) else: return -1 {% endhighlight %}
|
if the token is not a binary # operator. def
|
||||||
|
GetCurrentTokenPrecedence(self): if isinstance(self.current,
|
||||||
|
CharacterToken): return self.binop_precedence.get(self.current.char,
|
||||||
|
-1) else: return -1
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
With the helper above defined, we can now start parsing binary
|
With the helper above defined, we can now start parsing binary
|
||||||
expressions. The basic idea of operator precedence parsing is to break
|
expressions. The basic idea of operator precedence parsing is to break
|
||||||
|
|
@ -322,9 +369,14 @@ doesn't need to worry about nested subexpressions like (c+d) at all.
|
||||||
To start, an expression is a primary expression potentially followed by
|
To start, an expression is a primary expression potentially followed by
|
||||||
a sequence of ``[binop,primaryexpr]`` pairs:
|
a sequence of ``[binop,primaryexpr]`` pairs:
|
||||||
|
|
||||||
{% highlight python %} # expression ::= primary binoprhs def
|
|
||||||
ParseExpression(self): left = self.ParsePrimary() return
|
.. code-block:: python
|
||||||
self.ParseBinOpRHS(left, 0) {% endhighlight %}
|
|
||||||
|
# expression ::= primary binoprhs def
|
||||||
|
ParseExpression(self): left = self.ParsePrimary() return
|
||||||
|
self.ParseBinOpRHS(left, 0)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
``ParseBinOpRHS`` is the function that parses the sequence of pairs for
|
``ParseBinOpRHS`` is the function that parses the sequence of pairs for
|
||||||
us. It takes a precedence and a pointer to an expression for the part
|
us. It takes a precedence and a pointer to an expression for the part
|
||||||
|
|
@ -341,19 +393,24 @@ is passed in a precedence of 40, it will not consume any tokens (because
|
||||||
the precedence of '+' is only 20). With this in mind, ``ParseBinOpRHS``
|
the precedence of '+' is only 20). With this in mind, ``ParseBinOpRHS``
|
||||||
starts with:
|
starts with:
|
||||||
|
|
||||||
{% highlight python %} # binoprhs ::= (operator primary)\* def
|
|
||||||
ParseBinOpRHS(self, left, left\_precedence): # If this is a binary
|
|
||||||
operator, find its precedence. while True: precedence =
|
|
||||||
self.GetCurrentTokenPrecedence()
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
# binoprhs ::= (operator primary)\* def
|
||||||
|
ParseBinOpRHS(self, left, left_precedence): # If this is a binary
|
||||||
|
operator, find its precedence. while True: precedence =
|
||||||
|
self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# If this is a binary operator that binds at least as tightly as the
|
# If this is a binary operator that binds at least as tightly as the
|
||||||
# current one, consume it; otherwise we are done.
|
# current one, consume it; otherwise we are done.
|
||||||
if precedence < left_precedence:
|
if precedence < left_precedence:
|
||||||
return left
|
return left
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code gets the precedence of the current token and checks to see if
|
This code gets the precedence of the current token and checks to see if
|
||||||
if is too low. Because we defined invalid tokens to have a precedence of
|
if is too low. Because we defined invalid tokens to have a precedence of
|
||||||
|
|
@ -362,15 +419,20 @@ stream runs out of binary operators. If this check succeeds, we know
|
||||||
that the token is a binary operator and that it will be included in this
|
that the token is a binary operator and that it will be included in this
|
||||||
expression:
|
expression:
|
||||||
|
|
||||||
{% highlight python %} binary\_operator = self.current.char self.Next()
|
|
||||||
# eat the operator.
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
binary_operator = self.current.char self.Next()
|
||||||
|
# eat the operator.
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# Parse the primary expression after the binary operator.
|
# Parse the primary expression after the binary operator.
|
||||||
right = self.ParsePrimary()
|
right = self.ParsePrimary()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
As such, this code eats (and remembers) the binary operator and then
|
As such, this code eats (and remembers) the binary operator and then
|
||||||
parses the primary expression that follows. This builds up the whole
|
parses the primary expression that follows. This builds up the whole
|
||||||
|
|
@ -383,10 +445,15 @@ In particular, we could have ``(a+b) binop unparsed`` or
|
||||||
``binop`` to determine its precedence and compare it to BinOp's
|
``binop`` to determine its precedence and compare it to BinOp's
|
||||||
precedence (which is '+' in this case):
|
precedence (which is '+' in this case):
|
||||||
|
|
||||||
{% highlight python %} # If binary\_operator binds less tightly with
|
|
||||||
right than the operator after # right, let the pending operator take
|
.. code-block:: python
|
||||||
right as its left. next\_precedence = self.GetCurrentTokenPrecedence()
|
|
||||||
if precedence < next\_precedence: {% endhighlight %}
|
# If binary_operator binds less tightly with
|
||||||
|
right than the operator after # right, let the pending operator take
|
||||||
|
right as its left. next_precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
if precedence < next_precedence:
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
If the precedence of the binop to the right of ``RHS`` is lower or equal
|
If the precedence of the binop to the right of ``RHS`` is lower or equal
|
||||||
to the precedence of our current operator, then we know that the
|
to the precedence of our current operator, then we know that the
|
||||||
|
|
@ -395,15 +462,20 @@ current operator is ``+`` and the next operator is ``+``, we know that
|
||||||
they have the same precedence. In this case we'll create the AST node
|
they have the same precedence. In this case we'll create the AST node
|
||||||
for ``a+b``, and then continue parsing:
|
for ``a+b``, and then continue parsing:
|
||||||
|
|
||||||
{% highlight python %} if precedence < next\_precedence: ... if body
|
|
||||||
omitted ...
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
if precedence < next_precedence: ... if body
|
||||||
|
omitted ...
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# Merge left/right.
|
# Merge left/right.
|
||||||
left = BinaryOperatorExpressionNode(binary_operator, left, right);
|
left = BinaryOperatorExpressionNode(binary_operator, left, right);
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In our example above, this will turn ``a+b+`` into ``(a+b)`` and execute
|
In our example above, this will turn ``a+b+`` into ``(a+b)`` and execute
|
||||||
the next iteration of the loop, with ``+`` as the current token. The
|
the next iteration of the loop, with ``+`` as the current token. The
|
||||||
|
|
@ -420,18 +492,23 @@ all of ``( c + d ) * e * f`` as the RHS expression variable. The code to
|
||||||
do this is surprisingly simple (code from the above two blocks
|
do this is surprisingly simple (code from the above two blocks
|
||||||
duplicated for context):
|
duplicated for context):
|
||||||
|
|
||||||
{% highlight python %} # If binary\_operator binds less tightly with
|
|
||||||
right than the operator after # right, let the pending operator take
|
|
||||||
right as its left. next\_precedence = self.GetCurrentTokenPrecedence()
|
|
||||||
if precedence < next\_precedence: right = self.ParseBinOpRHS(right,
|
|
||||||
precedence + 1)
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
# If binary_operator binds less tightly with
|
||||||
|
right than the operator after # right, let the pending operator take
|
||||||
|
right as its left. next_precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
if precedence < next_precedence: right = self.ParseBinOpRHS(right,
|
||||||
|
precedence + 1)
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# Merge left/right.
|
# Merge left/right.
|
||||||
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
At this point, we know that the binary operator to the RHS of our
|
At this point, we know that the binary operator to the RHS of our
|
||||||
primary has higher precedence than the binop we are currently parsing.
|
primary has higher precedence than the binop we are currently parsing.
|
||||||
|
|
@ -466,11 +543,14 @@ well as function body definitions. The code to do this is
|
||||||
straight-forward and not very interesting (once you've survived
|
straight-forward and not very interesting (once you've survived
|
||||||
expressions):
|
expressions):
|
||||||
|
|
||||||
{% highlight python %} # prototype ::= id '(' id\* ')' def
|
|
||||||
ParsePrototype(self): if not isinstance(self.current, IdentifierToken):
|
|
||||||
raise RuntimeError('Expected function name in prototype.')
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
# prototype ::= id '(' id\* ')' def
|
||||||
|
ParsePrototype(self): if not isinstance(self.current, IdentifierToken):
|
||||||
|
raise RuntimeError('Expected function name in prototype.')
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
function_name = self.current.name
|
function_name = self.current.name
|
||||||
self.Next() # eat function name.
|
self.Next() # eat function name.
|
||||||
|
|
@ -492,31 +572,48 @@ raise RuntimeError('Expected function name in prototype.')
|
||||||
|
|
||||||
return PrototypeNode(function_name, arg_names)
|
return PrototypeNode(function_name, arg_names)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Given this, a function definition is very simple, just a prototype plus
|
Given this, a function definition is very simple, just a prototype plus
|
||||||
an expression to implement the body:
|
an expression to implement the body:
|
||||||
|
|
||||||
{% highlight python %} # definition ::= 'def' prototype expression def
|
|
||||||
ParseDefinition(self): self.Next() # eat def. proto =
|
.. code-block:: python
|
||||||
self.ParsePrototype() body = self.ParseExpression() return
|
|
||||||
FunctionNode(proto, body) {% endhighlight %}
|
# definition ::= 'def' prototype expression def
|
||||||
|
ParseDefinition(self): self.Next() # eat def. proto =
|
||||||
|
self.ParsePrototype() body = self.ParseExpression() return
|
||||||
|
FunctionNode(proto, body)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In addition, we support 'extern' to declare functions like 'sin' and
|
In addition, we support 'extern' to declare functions like 'sin' and
|
||||||
'cos' as well as to support forward declaration of user functions. These
|
'cos' as well as to support forward declaration of user functions. These
|
||||||
'extern's are just prototypes with no body:
|
'extern's are just prototypes with no body:
|
||||||
|
|
||||||
{% highlight python %} # external ::= 'extern' prototype def
|
|
||||||
ParseExtern(self): self.Next() # eat extern. return
|
.. code-block:: python
|
||||||
self.ParsePrototype() {% endhighlight %}
|
|
||||||
|
# external ::= 'extern' prototype def
|
||||||
|
ParseExtern(self): self.Next() # eat extern. return
|
||||||
|
self.ParsePrototype()
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Finally, we'll also let the user type in arbitrary top-level expressions
|
Finally, we'll also let the user type in arbitrary top-level expressions
|
||||||
and evaluate them on the fly. We will handle this by defining anonymous
|
and evaluate them on the fly. We will handle this by defining anonymous
|
||||||
nullary (zero argument) functions for them:
|
nullary (zero argument) functions for them:
|
||||||
|
|
||||||
{% highlight python %} # toplevelexpr ::= expression def
|
|
||||||
ParseTopLevelExpr(self): proto = PrototypeNode('', []) return
|
.. code-block:: python
|
||||||
FunctionNode(proto, self.ParseExpression()) {% endhighlight %}
|
|
||||||
|
# toplevelexpr ::= expression def
|
||||||
|
ParseTopLevelExpr(self): proto = PrototypeNode('', []) return
|
||||||
|
FunctionNode(proto, self.ParseExpression())
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Now that we have all the pieces, let's build a little driver that will
|
Now that we have all the pieces, let's build a little driver that will
|
||||||
let us actually *execute* this code we've built!
|
let us actually *execute* this code we've built!
|
||||||
|
|
@ -530,10 +627,13 @@ The driver for this simply invokes all of the parsing pieces with a
|
||||||
top-level dispatch loop. There isn't much interesting here, so I'll just
|
top-level dispatch loop. There isn't much interesting here, so I'll just
|
||||||
include the top-level loop. See `below <#code>`_ for full code.
|
include the top-level loop. See `below <#code>`_ for full code.
|
||||||
|
|
||||||
{% highlight python %} # Run the main "interpreter loop". while True:
|
|
||||||
print 'ready>', try: raw = raw\_input() except KeyboardInterrupt: return
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Run the main "interpreter loop". while True:
|
||||||
|
print 'ready>', try: raw = raw_input() except KeyboardInterrupt: return
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
parser = Parser(Tokenize(raw), operator_precedence)
|
parser = Parser(Tokenize(raw), operator_precedence)
|
||||||
while True:
|
while True:
|
||||||
|
|
@ -547,7 +647,9 @@ print 'ready>', try: raw = raw\_input() except KeyboardInterrupt: return
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Here we create a new ``Parser`` for each line read, and try to parse out
|
Here we create a new ``Parser`` for each line read, and try to parse out
|
||||||
all the expressions, declarations and definitions in the line. We also
|
all the expressions, declarations and definitions in the line. We also
|
||||||
|
|
@ -564,12 +666,17 @@ lexer, parser, and AST builder. With this done, the executable will
|
||||||
validate Kaleidoscope code and tell us if it is grammatically invalid.
|
validate Kaleidoscope code and tell us if it is grammatically invalid.
|
||||||
For example, here is a sample interaction:
|
For example, here is a sample interaction:
|
||||||
|
|
||||||
{% highlight python %} $ python kaleidoscope.py ready> def foo(x y)
|
|
||||||
x+foo(y, 4.0) Parsed a function definition. ready> def foo(x y) x+y y
|
.. code-block:: python
|
||||||
Parsed a function definition. Parsed a top-level expression. ready> def
|
|
||||||
foo(x y) x+y ) Parsed a function definition. Error: Unknown token when
|
$ python kaleidoscope.py ready> def foo(x y)
|
||||||
expecting an expression. ready> extern sin(a); Parsed an extern. ready>
|
x+foo(y, 4.0) Parsed a function definition. ready> def foo(x y) x+y y
|
||||||
^C $ {% endhighlight %}
|
Parsed a function definition. Parsed a top-level expression. ready> def
|
||||||
|
foo(x y) x+y ) Parsed a function definition. Error: Unknown token when
|
||||||
|
expecting an expression. ready> extern sin(a); Parsed an extern. ready>
|
||||||
|
^C $
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
There is a lot of room for extension here. You can define new AST nodes,
|
There is a lot of room for extension here. You can define new AST nodes,
|
||||||
extend the language in many ways, etc. In the `next
|
extend the language in many ways, etc. In the `next
|
||||||
|
|
@ -585,43 +692,42 @@ Here is the complete code listing for this and the previous chapter.
|
||||||
Note that it is fully self-contained: you don't need LLVM or any
|
Note that it is fully self-contained: you don't need LLVM or any
|
||||||
external libraries at all for this.
|
external libraries at all for this.
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
|
||||||
import re
|
.. code-block:: python
|
||||||
|
|
||||||
Lexer
|
#!/usr/bin/env python
|
||||||
-----
|
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
import re
|
||||||
===================================================
|
|
||||||
|
|
||||||
class EOFToken(object): pass
|
Lexer
|
||||||
|
-----
|
||||||
|
|
||||||
class DefToken(object): pass
|
# The lexer yields one of these types for each token.
|
||||||
|
class EOFToken(object): pass
|
||||||
|
|
||||||
class ExternToken(object): pass
|
class DefToken(object): pass
|
||||||
|
|
||||||
class IdentifierToken(object): def **init**\ (self, name): self.name =
|
class ExternToken(object): pass
|
||||||
name
|
|
||||||
|
|
||||||
class NumberToken(object): def **init**\ (self, value): self.value =
|
class IdentifierToken(object): def **init**\ (self, name): self.name =
|
||||||
value
|
name
|
||||||
|
|
||||||
class CharacterToken(object): def **init**\ (self, char): self.char =
|
class NumberToken(object): def **init**\ (self, value): self.value =
|
||||||
char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
value
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
|
||||||
== other
|
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
class CharacterToken(object): def **init**\ (self, char): self.char =
|
||||||
=============================================================
|
char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
|
== other
|
||||||
|
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
# Regular expressions that tokens and comments of our language.
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
string[0].isspace(): string = string[1:] continue
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
# Run regexes.
|
# Run regexes.
|
||||||
comment_match = REGEX_COMMENT.match(string)
|
comment_match = REGEX_COMMENT.match(string)
|
||||||
|
|
@ -651,82 +757,64 @@ string[0].isspace(): string = string[1:] continue
|
||||||
yield CharacterToken(string[0])
|
yield CharacterToken(string[0])
|
||||||
string = string[1:]
|
string = string[1:]
|
||||||
|
|
||||||
yield EOFToken()
|
yield EOFToken()
|
||||||
|
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
# Expression class for numeric literals like "1.0".
|
||||||
|
class NumberExpressionNode(ExpressionNode): def **init**\ (self, value):
|
||||||
|
self.value = value
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
# Expression class for referencing a variable, like "a".
|
||||||
=================================================
|
class VariableExpressionNode(ExpressionNode): def **init**\ (self,
|
||||||
|
name): self.name = name
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode): def **init**\ (self, value):
|
# Expression class for a binary operator.
|
||||||
self.value = value
|
class BinaryOperatorExpressionNode(ExpressionNode): def **init**\ (self,
|
||||||
|
operator, left, right): self.operator = operator self.left = left
|
||||||
|
self.right = right
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
# Expression class for function calls.
|
||||||
======================================================
|
class CallExpressionNode(ExpressionNode): def **init**\ (self, callee,
|
||||||
|
args): self.callee = callee self.args = args
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode): def **init**\ (self,
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
name): self.name = name
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes).
|
||||||
|
class PrototypeNode(object): def **init**\ (self, name, args): self.name
|
||||||
|
= name self.args = args
|
||||||
|
|
||||||
Expression class for a binary operator.
|
# This class represents a function definition itself.
|
||||||
=======================================
|
class FunctionNode(object): def **init**\ (self, prototype, body):
|
||||||
|
self.prototype = prototype self.body = body
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode): def **init**\ (self,
|
Parser
|
||||||
operator, left, right): self.operator = operator self.left = left
|
------
|
||||||
self.right = right
|
|
||||||
|
|
||||||
Expression class for function calls.
|
class Parser(object):
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode): def **init**\ (self, callee,
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
args): self.callee = callee self.args = args
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
==============================================================================
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
|
and # updates Parser.current with its results. def Next(self):
|
||||||
|
self.current = self.tokens.next()
|
||||||
|
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
============================================================================
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
|
isinstance(self.current, CharacterToken): return
|
||||||
|
self.binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
takes).
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
=======
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
class PrototypeNode(object): def **init**\ (self, name, args): self.name
|
::
|
||||||
= name self.args = args
|
|
||||||
|
|
||||||
This class represents a function definition itself.
|
|
||||||
===================================================
|
|
||||||
|
|
||||||
class FunctionNode(object): def **init**\ (self, prototype, body):
|
|
||||||
self.prototype = prototype self.body = body
|
|
||||||
|
|
||||||
Parser
|
|
||||||
------
|
|
||||||
|
|
||||||
class Parser(object):
|
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
|
||||||
and # updates Parser.current with its results. def Next(self):
|
|
||||||
self.current = self.tokens.next()
|
|
||||||
|
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
|
||||||
isinstance(self.current, CharacterToken): return
|
|
||||||
self.binop\_precedence.get(self.current.char, -1) else: return -1
|
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
|
||||||
self.Next() # eat identifier.
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
if self.current != CharacterToken('('): # Simple variable reference.
|
if self.current != CharacterToken('('): # Simple variable reference.
|
||||||
return VariableExpressionNode(identifier_name)
|
return VariableExpressionNode(identifier_name)
|
||||||
|
|
@ -746,14 +834,14 @@ self.Next() # eat identifier.
|
||||||
self.Next() # eat ')'.
|
self.Next() # eat ')'.
|
||||||
return CallExpressionNode(identifier_name, args)
|
return CallExpressionNode(identifier_name, args)
|
||||||
|
|
||||||
# numberexpr ::= number def ParseNumberExpr(self): result =
|
# numberexpr ::= number def ParseNumberExpr(self): result =
|
||||||
NumberExpressionNode(self.current.value) self.Next() # consume the
|
NumberExpressionNode(self.current.value) self.Next() # consume the
|
||||||
number. return result
|
number. return result
|
||||||
|
|
||||||
# parenexpr ::= '(' expression ')' def ParseParenExpr(self): self.Next()
|
# parenexpr ::= '(' expression ')' def ParseParenExpr(self): self.Next()
|
||||||
# eat '('.
|
# eat '('.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
contents = self.ParseExpression()
|
contents = self.ParseExpression()
|
||||||
|
|
||||||
|
|
@ -763,18 +851,18 @@ number. return result
|
||||||
|
|
||||||
return contents
|
return contents
|
||||||
|
|
||||||
# primary ::= identifierexpr \| numberexpr \| parenexpr def
|
# primary ::= identifierexpr \| numberexpr \| parenexpr def
|
||||||
ParsePrimary(self): if isinstance(self.current, IdentifierToken): return
|
ParsePrimary(self): if isinstance(self.current, IdentifierToken): return
|
||||||
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
||||||
return self.ParseNumberExpr() elif self.current == CharacterToken('('):
|
return self.ParseNumberExpr() elif self.current == CharacterToken('('):
|
||||||
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
||||||
when expecting an expression.')
|
when expecting an expression.')
|
||||||
|
|
||||||
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
# If this is a binary operator that binds at least as tightly as the
|
# If this is a binary operator that binds at least as tightly as the
|
||||||
# current one, consume it; otherwise we are done.
|
# current one, consume it; otherwise we are done.
|
||||||
|
|
@ -796,14 +884,14 @@ while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
# Merge left/right.
|
# Merge left/right.
|
||||||
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
||||||
|
|
||||||
# expression ::= primary binoprhs def ParseExpression(self): left =
|
# expression ::= primary binoprhs def ParseExpression(self): left =
|
||||||
self.ParsePrimary() return self.ParseBinOpRHS(left, 0)
|
self.ParsePrimary() return self.ParseBinOpRHS(left, 0)
|
||||||
|
|
||||||
# prototype ::= id '(' id\* ')' def ParsePrototype(self): if not
|
# prototype ::= id '(' id\* ')' def ParsePrototype(self): if not
|
||||||
isinstance(self.current, IdentifierToken): raise RuntimeError('Expected
|
isinstance(self.current, IdentifierToken): raise RuntimeError('Expected
|
||||||
function name in prototype.')
|
function name in prototype.')
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
function_name = self.current.name
|
function_name = self.current.name
|
||||||
self.Next() # eat function name.
|
self.Next() # eat function name.
|
||||||
|
|
@ -825,40 +913,40 @@ function name in prototype.')
|
||||||
|
|
||||||
return PrototypeNode(function_name, arg_names)
|
return PrototypeNode(function_name, arg_names)
|
||||||
|
|
||||||
# definition ::= 'def' prototype expression def ParseDefinition(self):
|
# definition ::= 'def' prototype expression def ParseDefinition(self):
|
||||||
self.Next() # eat def. proto = self.ParsePrototype() body =
|
self.Next() # eat def. proto = self.ParsePrototype() body =
|
||||||
self.ParseExpression() return FunctionNode(proto, body)
|
self.ParseExpression() return FunctionNode(proto, body)
|
||||||
|
|
||||||
# toplevelexpr ::= expression def ParseTopLevelExpr(self): proto =
|
# toplevelexpr ::= expression def ParseTopLevelExpr(self): proto =
|
||||||
PrototypeNode('', []) return FunctionNode(proto, self.ParseExpression())
|
PrototypeNode('', []) return FunctionNode(proto, self.ParseExpression())
|
||||||
|
|
||||||
# external ::= 'extern' prototype def ParseExtern(self): self.Next() #
|
# external ::= 'extern' prototype def ParseExtern(self): self.Next() #
|
||||||
eat extern. return self.ParsePrototype()
|
eat extern. return self.ParsePrototype()
|
||||||
|
|
||||||
# Top-Level parsing def HandleDefinition(self):
|
# Top-Level parsing def HandleDefinition(self):
|
||||||
self.Handle(self.ParseDefinition, 'Parsed a function definition.')
|
self.Handle(self.ParseDefinition, 'Parsed a function definition.')
|
||||||
|
|
||||||
def HandleExtern(self): self.Handle(self.ParseExtern, 'Parsed an
|
def HandleExtern(self): self.Handle(self.ParseExtern, 'Parsed an
|
||||||
extern.')
|
extern.')
|
||||||
|
|
||||||
def HandleTopLevelExpression(self): self.Handle(self.ParseTopLevelExpr,
|
def HandleTopLevelExpression(self): self.Handle(self.ParseTopLevelExpr,
|
||||||
'Parsed a top-level expression.')
|
'Parsed a top-level expression.')
|
||||||
|
|
||||||
def Handle(self, function, message): try: function() print message
|
def Handle(self, function, message): try: function() print message
|
||||||
except Exception, e: print 'Error:', e try: self.Next() # Skip for error
|
except Exception, e: print 'Error:', e try: self.Next() # Skip for error
|
||||||
recovery. except: pass
|
recovery. except: pass
|
||||||
|
|
||||||
Main driver code.
|
Main driver code.
|
||||||
-----------------
|
-----------------
|
||||||
|
|
||||||
def main(): # Install standard binary operators. # 1 is lowest possible
|
def main(): # Install standard binary operators. # 1 is lowest possible
|
||||||
precedence. 40 is the highest. operator\_precedence = { '<': 10, '+':
|
precedence. 40 is the highest. operator_precedence = { '<': 10, '+':
|
||||||
20, '-': 20, '\*': 40 }
|
20, '-': 20, '\*': 40 }
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: return
|
= raw_input() except KeyboardInterrupt: return
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
parser = Parser(Tokenize(raw), operator_precedence)
|
parser = Parser(Tokenize(raw), operator_precedence)
|
||||||
while True:
|
while True:
|
||||||
|
|
@ -872,9 +960,4 @@ precedence. 40 is the highest. operator\_precedence = { '<': 10, '+':
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Implementing Code Generation to LLVM
|
|
||||||
IR <PythonLangImpl3.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -31,23 +31,26 @@ Code Generation Setup # {#basics}
|
||||||
In order to generate LLVM IR, we want some simple setup to get started.
|
In order to generate LLVM IR, we want some simple setup to get started.
|
||||||
First we define code generation methods in each AST node class:
|
First we define code generation methods in each AST node class:
|
||||||
|
|
||||||
{% highlight python %} # Expression class for numeric literals like
|
|
||||||
"1.0". class NumberExpressionNode(ExpressionNode):
|
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
.. code-block:: python
|
||||||
|
|
||||||
def CodeGen(self): ...
|
# Expression class for numeric literals like
|
||||||
|
"1.0". class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
def **init**\ (self, value): self.value = value
|
||||||
======================================================
|
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
def CodeGen(self): ...
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
# Expression class for referencing a variable, like "a".
|
||||||
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
|
def CodeGen(self): ...
|
||||||
|
|
||||||
|
...
|
||||||
|
|
||||||
def CodeGen(self): ...
|
|
||||||
|
|
||||||
... {% endhighlight %}
|
|
||||||
|
|
||||||
The ``CodeGen`` method says to emit IR for that AST node along with all
|
The ``CodeGen`` method says to emit IR for that AST node along with all
|
||||||
the things it depends on, and they all return an LLVM Value object.
|
the things it depends on, and they all return an LLVM Value object.
|
||||||
|
|
@ -64,21 +67,20 @@ Assignment <http://en.wikipedia.org/wiki/Static_single_assignment_form>`_
|
||||||
We will also need to define some global variables which we will be used
|
We will also need to define some global variables which we will be used
|
||||||
during code generation:
|
during code generation:
|
||||||
|
|
||||||
{% highlight python %} # The LLVM module, which holds all the IR code.
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
.. code-block:: python
|
||||||
=========================================================================
|
|
||||||
|
|
||||||
g\_llvm\_builder = None
|
# The LLVM module, which holds all the IR code.
|
||||||
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
==============================================================================
|
g_llvm_builder = None
|
||||||
|
|
||||||
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
|
|
||||||
and what their LLVM representation is.
|
|
||||||
======================================
|
|
||||||
|
|
||||||
g\_named\_values = {} {% endhighlight %}
|
|
||||||
|
|
||||||
``g_llvm_module`` is the LLVM construct that contains all of the
|
``g_llvm_module`` is the LLVM construct that contains all of the
|
||||||
functions and global variables in a chunk of code. In many ways, it is
|
functions and global variables in a chunk of code. In many ways, it is
|
||||||
|
|
@ -112,8 +114,13 @@ Generating LLVM code for expression nodes is very straightforward: less
|
||||||
than 35 lines of commented code for all four of our expression nodes.
|
than 35 lines of commented code for all four of our expression nodes.
|
||||||
First we'll do numeric literals:
|
First we'll do numeric literals:
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): return
|
|
||||||
Constant.real(Type.double(), self.value) {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): return
|
||||||
|
Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In llvmpy, floating point numeric constants are represented with the
|
In llvmpy, floating point numeric constants are represented with the
|
||||||
``llvm.core.ConstantFP`` class. To create one, we can use the static
|
``llvm.core.ConstantFP`` class. To create one, we can use the static
|
||||||
|
|
@ -123,9 +130,14 @@ LLVM IR constants are all uniqued together and shared. For this reason,
|
||||||
we create the constant through a factory method instead of instantiating
|
we create the constant through a factory method instead of instantiating
|
||||||
one directly.
|
one directly.
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): if self.name in
|
|
||||||
g\_named\_values: return g\_named\_values[self.name] else: raise
|
.. code-block:: python
|
||||||
RuntimeError('Unknown variable name: ' + self.name) {% endhighlight %}
|
|
||||||
|
def CodeGen(self): if self.name in
|
||||||
|
g_named_values: return g_named_values[self.name] else: raise
|
||||||
|
RuntimeError('Unknown variable name: ' + self.name)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
References to variables are also quite simple using LLVM. In the simple
|
References to variables are also quite simple using LLVM. In the simple
|
||||||
version of Kaleidoscope, we assume that the variable has already been
|
version of Kaleidoscope, we assume that the variable has already been
|
||||||
|
|
@ -137,10 +149,13 @@ the value for it. In future chapters, we'll add support for `loop
|
||||||
induction variables <PythonLangImpl5.html#for>`_ in the symbol table,
|
induction variables <PythonLangImpl5.html#for>`_ in the symbol table,
|
||||||
and for `local variables <PythonLangImpl7.html#localvars>`_.
|
and for `local variables <PythonLangImpl7.html#localvars>`_.
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): left = self.left.CodeGen()
|
|
||||||
right = self.right.CodeGen()
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): left = self.left.CodeGen()
|
||||||
|
right = self.right.CodeGen()
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
if self.operator == '+':
|
if self.operator == '+':
|
||||||
return g_llvm_builder.fadd(left, right, 'addtmp')
|
return g_llvm_builder.fadd(left, right, 'addtmp')
|
||||||
|
|
@ -155,7 +170,9 @@ right = self.right.CodeGen()
|
||||||
else:
|
else:
|
||||||
raise RuntimeError('Unknown binary operator.')
|
raise RuntimeError('Unknown binary operator.')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Binary operators start to get more interesting. The basic idea here is
|
Binary operators start to get more interesting. The basic idea here is
|
||||||
that we recursively emit code for the left-hand side of the expression,
|
that we recursively emit code for the left-hand side of the expression,
|
||||||
|
|
@ -193,11 +210,14 @@ treating the input as an unsigned value. In contrast, if we used the
|
||||||
the Kaleidoscope ``<`` operator would return 0.0 and -1.0, depending on
|
the Kaleidoscope ``<`` operator would return 0.0 and -1.0, depending on
|
||||||
the input value.
|
the input value.
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): # Look up the name in the
|
|
||||||
global module table. callee =
|
|
||||||
g\_llvm\_module.get\_function\_named(self.callee)
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): # Look up the name in the
|
||||||
|
global module table. callee =
|
||||||
|
g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# Check for argument mismatch error.
|
# Check for argument mismatch error.
|
||||||
if len(callee.args) != len(self.args):
|
if len(callee.args) != len(self.args):
|
||||||
|
|
@ -207,7 +227,9 @@ g\_llvm\_module.get\_function\_named(self.callee)
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Code generation for function calls is quite straightforward with LLVM.
|
Code generation for function calls is quite straightforward with LLVM.
|
||||||
The code above initially does a function name lookup in the LLVM
|
The code above initially does a function name lookup in the LLVM
|
||||||
|
|
@ -242,15 +264,20 @@ let's talk about code generation for prototypes: they are used both for
|
||||||
function bodies and external function declarations. The code starts
|
function bodies and external function declarations. The code starts
|
||||||
with:
|
with:
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): # Make the function type, eg.
|
|
||||||
double(double,double). funct\_type = Type.function( Type.double(),
|
|
||||||
[Type.double()] \* len(self.args), False)
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): # Make the function type, eg.
|
||||||
|
double(double,double). funct_type = Type.function( Type.double(),
|
||||||
|
[Type.double()] \* len(self.args), False)
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
function = Function.new(g_llvm_module, funct_type, self.name)
|
function = Function.new(g_llvm_module, funct_type, self.name)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The call to ``Type.function`` creates the ``FunctionType`` that should
|
The call to ``Type.function`` creates the ``FunctionType`` that should
|
||||||
be used for a given Prototype. Since all function arguments in
|
be used for a given Prototype. Since all function arguments in
|
||||||
|
|
@ -272,11 +299,16 @@ the name the user specified: since ``g_llvm_module`` is specified, this
|
||||||
name is registered in ``g_llvm_module``'s symbol table, which is used by
|
name is registered in ``g_llvm_module``'s symbol table, which is used by
|
||||||
the function call code above.
|
the function call code above.
|
||||||
|
|
||||||
{% highlight python %} # If the name conflicted, there was already
|
|
||||||
something with the same name. # If it has a body, don't allow
|
.. code-block:: python
|
||||||
redefinition or reextern. if function.name != self.name:
|
|
||||||
function.delete() function =
|
# If the name conflicted, there was already
|
||||||
g\_llvm\_module.get\_function\_named(self.name) {% endhighlight %}
|
something with the same name. # If it has a body, don't allow
|
||||||
|
redefinition or reextern. if function.name != self.name:
|
||||||
|
function.delete() function =
|
||||||
|
g_llvm_module.get_function_named(self.name)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The Module symbol table works just like the Function symbol table when
|
The Module symbol table works just like the Function symbol table when
|
||||||
it comes to name conflicts: if a new function is created with a name was
|
it comes to name conflicts: if a new function is created with a name was
|
||||||
|
|
@ -298,18 +330,23 @@ function we just created (by calling ``delete``) and then calling
|
||||||
``get_function_named`` to get the existing function with the specified
|
``get_function_named`` to get the existing function with the specified
|
||||||
name.
|
name.
|
||||||
|
|
||||||
{% highlight python %} # If the function already has a body, reject
|
|
||||||
this. if not function.is\_declaration: raise RuntimeError('Redefinition
|
|
||||||
of function.')
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
# If the function already has a body, reject
|
||||||
|
this. if not function.is_declaration: raise RuntimeError('Redefinition
|
||||||
|
of function.')
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# If F took a different number of args, reject.
|
# If F took a different number of args, reject.
|
||||||
if len(callee.args) != len(self.args):
|
if len(callee.args) != len(self.args):
|
||||||
raise RuntimeError('Redeclaration of a function with different number '
|
raise RuntimeError('Redeclaration of a function with different number '
|
||||||
'of args.')
|
'of args.')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In order to verify the logic above, we first check to see if the
|
In order to verify the logic above, we first check to see if the
|
||||||
pre-existing function is a forward declaration. Since we don't allow
|
pre-existing function is a forward declaration. Since we don't allow
|
||||||
|
|
@ -318,16 +355,21 @@ case. If the previous reference to a function was an 'extern', we simply
|
||||||
verify that the number of arguments for that definition and this one
|
verify that the number of arguments for that definition and this one
|
||||||
match up. If not, we emit an error.
|
match up. If not, we emit an error.
|
||||||
|
|
||||||
{% highlight python %} # Set names for all arguments and add them to the
|
|
||||||
variables symbol table. for arg, arg\_name in zip(function.args,
|
|
||||||
self.args): arg.name = arg\_name # Add arguments to variable symbol
|
|
||||||
table. g\_named\_values[arg\_name] = arg
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Set names for all arguments and add them to the
|
||||||
|
variables symbol table. for arg, arg_name in zip(function.args,
|
||||||
|
self.args): arg.name = arg_name # Add arguments to variable symbol
|
||||||
|
table. g_named_values[arg_name] = arg
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The last bit of code for prototypes loops over all of the arguments in
|
The last bit of code for prototypes loops over all of the arguments in
|
||||||
the function, setting the name of the LLVM Argument objects to match,
|
the function, setting the name of the LLVM Argument objects to match,
|
||||||
|
|
@ -338,15 +380,20 @@ would be very straight-forward with the mechanics we have already used
|
||||||
above. Once this is all set up, it returns the Function object to the
|
above. Once this is all set up, it returns the Function object to the
|
||||||
caller.
|
caller.
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): # Clear scope.
|
|
||||||
g\_named\_values.clear()
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): # Clear scope.
|
||||||
|
g_named_values.clear()
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# Create a function object.
|
# Create a function object.
|
||||||
function = self.prototype.CodeGen()
|
function = self.prototype.CodeGen()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Code generation for function definitions starts out simply enough: we
|
Code generation for function definitions starts out simply enough: we
|
||||||
just clear out the ``g_named_values`` dictionary to make sure that there
|
just clear out the ``g_named_values`` dictionary to make sure that there
|
||||||
|
|
@ -354,32 +401,37 @@ isn't anything in it from the last function we compiled and codegen the
|
||||||
prototype. Code generation of the prototype ensures that there is an
|
prototype. Code generation of the prototype ensures that there is an
|
||||||
LLVM Function object that is ready to go for us.
|
LLVM Function object that is ready to go for us.
|
||||||
|
|
||||||
{% highlight python %} # Create a new basic block to start insertion
|
|
||||||
into. block = function.append\_basic\_block('entry') global
|
|
||||||
g\_llvm\_builder g\_llvm\_builder = Builder.new(block) {% endhighlight
|
|
||||||
%}
|
|
||||||
|
|
||||||
Now we get to the point where ``g_llvm_builder`` is set up. The first
|
.. code-block:: python
|
||||||
line creates a new `basic
|
|
||||||
block <http://en.wikipedia.org/wiki/Basic_block>`_ (named "entry"),
|
|
||||||
which is inserted into the function. The second line declares that the
|
|
||||||
global ``g_llvm_builder`` object is to be changed. The last line creates
|
|
||||||
a new builder that is set up to insert new instructions into the basic
|
|
||||||
block we just created. Basic blocks in LLVM are an important part of
|
|
||||||
functions that define the `Control Flow
|
|
||||||
Graph <http://en.wikipedia.org/wiki/Control_flow_graph>`_. Since we
|
|
||||||
don't have any control flow, our functions will only contain one block
|
|
||||||
at this point. We'll fix this in `Chapter 5 <PythonLangImpl5.html>`_ :).
|
|
||||||
|
|
||||||
{% highlight python %} # Finish off the function. try: return\_value =
|
# Create a new basic block to start insertion
|
||||||
self.body.CodeGen() g\_llvm\_builder.ret(return\_value)
|
into. block = function.append_basic_block('entry') global
|
||||||
|
g_llvm_builder g_llvm_builder = Builder.new(block) {% endhighlight
|
||||||
|
%}
|
||||||
|
|
||||||
::
|
Now we get to the point where ``g_llvm_builder`` is set up. The first
|
||||||
|
line creates a new `basic
|
||||||
|
block <http://en.wikipedia.org/wiki/Basic_block>`_ (named "entry"),
|
||||||
|
which is inserted into the function. The second line declares that the
|
||||||
|
global ``g_llvm_builder`` object is to be changed. The last line creates
|
||||||
|
a new builder that is set up to insert new instructions into the basic
|
||||||
|
block we just created. Basic blocks in LLVM are an important part of
|
||||||
|
functions that define the `Control Flow
|
||||||
|
Graph <http://en.wikipedia.org/wiki/Control_flow_graph>`_. Since we
|
||||||
|
don't have any control flow, our functions will only contain one block
|
||||||
|
at this point. We'll fix this in `Chapter 5 <PythonLangImpl5.html>`_ :).
|
||||||
|
|
||||||
|
{% highlight python %} # Finish off the function. try: return_value =
|
||||||
|
self.body.CodeGen() g_llvm_builder.ret(return_value)
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# Validate the generated code, checking for consistency.
|
# Validate the generated code, checking for consistency.
|
||||||
function.verify()
|
function.verify()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Once the insertion point is set up, we call the ``CodeGen`` method for
|
Once the insertion point is set up, we call the ``CodeGen`` method for
|
||||||
the root expression of the function. If no error happens, this emits
|
the root expression of the function. If no error happens, this emits
|
||||||
|
|
@ -392,13 +444,18 @@ checks on the generated code, to determine if our compiler is doing
|
||||||
everything right. Using this is important: it can catch a lot of bugs.
|
everything right. Using this is important: it can catch a lot of bugs.
|
||||||
Once the function is finished and validated, we return it.
|
Once the function is finished and validated, we return it.
|
||||||
|
|
||||||
{% highlight python %} except: function.delete() raise
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
except: function.delete() raise
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The only piece left here is handling of the error case. For simplicity,
|
The only piece left here is handling of the error case. For simplicity,
|
||||||
we handle this by merely deleting the function we produced with the
|
we handle this by merely deleting the function we produced with the
|
||||||
|
|
@ -411,9 +468,14 @@ can return a previously defined forward declaration, our code can
|
||||||
actually delete a forward declaration. There are a number of ways to fix
|
actually delete a forward declaration. There are a number of ways to fix
|
||||||
this bug; see what you can come up with! Here is a testcase:
|
this bug; see what you can come up with! Here is a testcase:
|
||||||
|
|
||||||
{% highlight python %} extern foo(a b) # ok, defines foo. def foo(a b) c
|
|
||||||
# error, 'c' is invalid. def bar() foo(1, 2) # error, unknown function
|
.. code-block:: python
|
||||||
"foo" {% endhighlight %}
|
|
||||||
|
extern foo(a b) # ok, defines foo. def foo(a b) c
|
||||||
|
# error, 'c' is invalid. def bar() foo(1, 2) # error, unknown function
|
||||||
|
"foo"
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
--------------
|
--------------
|
||||||
|
|
||||||
|
|
@ -426,8 +488,13 @@ CodeGen into the ``Handle*`` functions, and then dumps out the LLVM IR.
|
||||||
This gives a nice way to look at the LLVM IR for simple functions. For
|
This gives a nice way to look at the LLVM IR for simple functions. For
|
||||||
example:
|
example:
|
||||||
|
|
||||||
{% highlight bash %} ready> 4+5 Read a top-level expression: define
|
|
||||||
double @0() { entry: ret double 9.000000e+00 } {% endhighlight %}
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> 4+5 Read a top-level expression: define
|
||||||
|
double @0() { entry: ret double 9.000000e+00 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Note how the parser turns the top-level expression into anonymous
|
Note how the parser turns the top-level expression into anonymous
|
||||||
functions for us. This will be handy when we add JIT support in the next
|
functions for us. This will be handy when we add JIT support in the next
|
||||||
|
|
@ -435,56 +502,76 @@ chapter. Also note that the code is very literally transcribed, no
|
||||||
optimizations are being performed except simple constant folding done by
|
optimizations are being performed except simple constant folding done by
|
||||||
the Builder. We will add optimizations explicitly in the next chapter.
|
the Builder. We will add optimizations explicitly in the next chapter.
|
||||||
|
|
||||||
{% highlight bash %} ready> def foo(a b) a\ *a + 2*\ a\ *b + b*\ b Read
|
|
||||||
a function definition: define double @foo(double %a, double %b) { entry:
|
.. code-block:: bash
|
||||||
%multmp = fmul double %a, %a ; [#uses=1] %multmp1 = fmul double
|
|
||||||
2.000000e+00, %a ; [#uses=1] %multmp2 = fmul double %multmp1, %b ;
|
ready> def foo(a b) a\ *a + 2*\ a\ *b + b*\ b Read
|
||||||
[#uses=1] %addtmp = fadd double %multmp, %multmp2 ; [#uses=1] %multmp3 =
|
a function definition: define double @foo(double %a, double %b) { entry:
|
||||||
fmul double %b, %b ; [#uses=1] %addtmp4 = fadd double %addtmp, %multmp3
|
%multmp = fmul double %a, %a ; [#uses=1] %multmp1 = fmul double
|
||||||
; [#uses=1] ret double %addtmp4 } {% endhighlight %}
|
2.000000e+00, %a ; [#uses=1] %multmp2 = fmul double %multmp1, %b ;
|
||||||
|
[#uses=1] %addtmp = fadd double %multmp, %multmp2 ; [#uses=1] %multmp3 =
|
||||||
|
fmul double %b, %b ; [#uses=1] %addtmp4 = fadd double %addtmp, %multmp3
|
||||||
|
; [#uses=1] ret double %addtmp4 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This shows some simple arithmetic. Notice the striking similarity to the
|
This shows some simple arithmetic. Notice the striking similarity to the
|
||||||
LLVM builder calls that we use to create the instructions.
|
LLVM builder calls that we use to create the instructions.
|
||||||
|
|
||||||
{% highlight bash %} ready> def bar(a) foo(a, 4.0) + bar(31337) Read a
|
|
||||||
function definition: define double @bar(double %a) { entry: %calltmp =
|
.. code-block:: bash
|
||||||
call double @foo(double %a, double 4.000000e+00) ; [#uses=1] %calltmp1 =
|
|
||||||
call double @bar(double 3.133700e+04) ; [#uses=1] %addtmp = fadd double
|
ready> def bar(a) foo(a, 4.0) + bar(31337) Read a
|
||||||
%calltmp, %calltmp1 ; [#uses=1] ret double %addtmp } {% endhighlight %}
|
function definition: define double @bar(double %a) { entry: %calltmp =
|
||||||
|
call double @foo(double %a, double 4.000000e+00) ; [#uses=1] %calltmp1 =
|
||||||
|
call double @bar(double 3.133700e+04) ; [#uses=1] %addtmp = fadd double
|
||||||
|
%calltmp, %calltmp1 ; [#uses=1] ret double %addtmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This shows some function calls. Note that this function will take a long
|
This shows some function calls. Note that this function will take a long
|
||||||
time to execute if you call it. In the future we'll add conditional
|
time to execute if you call it. In the future we'll add conditional
|
||||||
control flow to actually make recursion useful :).
|
control flow to actually make recursion useful :).
|
||||||
|
|
||||||
{% highlight bash %} ready> extern cos(x) Read extern: declare double
|
|
||||||
@cos(double)
|
|
||||||
|
|
||||||
ready> cos(1.234) Read a top-level expression: define double @1() {
|
.. code-block:: bash
|
||||||
entry: %calltmp = call double @cos(double 1.234000e+00) ; [#uses=1] ret
|
|
||||||
double %calltmp } {% endhighlight %}
|
ready> extern cos(x) Read extern: declare double
|
||||||
|
@cos(double)
|
||||||
|
|
||||||
|
ready> cos(1.234) Read a top-level expression: define double @1() {
|
||||||
|
entry: %calltmp = call double @cos(double 1.234000e+00) ; [#uses=1] ret
|
||||||
|
double %calltmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This shows an extern for the libm "cos" function, and a call to it.
|
This shows an extern for the libm "cos" function, and a call to it.
|
||||||
|
|
||||||
{% highlight bash %} ready> ^C ; ModuleID = 'my cool jit'
|
|
||||||
|
|
||||||
define double @0() { entry: ret double 9.000000e+00 }
|
.. code-block:: bash
|
||||||
|
|
||||||
define double @foo(double %a, double %b) { entry: %multmp = fmul double
|
ready> ^C ; ModuleID = 'my cool jit'
|
||||||
%a, %a ; [#uses=1] %multmp1 = fmul double 2.000000e+00, %a ; [#uses=1]
|
|
||||||
%multmp2 = fmul double %multmp1, %b ; [#uses=1] %addtmp = fadd double
|
|
||||||
%multmp, %multmp2 ; [#uses=1] %multmp3 = fmul double %b, %b ; [#uses=1]
|
|
||||||
%addtmp4 = fadd double %addtmp, %multmp3 ; [#uses=1] ret double %addtmp4
|
|
||||||
}
|
|
||||||
|
|
||||||
define double @bar(double %a) { entry: %calltmp = call double
|
define double @0() { entry: ret double 9.000000e+00 }
|
||||||
@foo(double %a, double 4.000000e+00) ; [#uses=1] %calltmp1 = call double
|
|
||||||
@bar(double 3.133700e+04) ; [#uses=1] %addtmp = fadd double %calltmp,
|
define double @foo(double %a, double %b) { entry: %multmp = fmul double
|
||||||
%calltmp1 ; [#uses=1] ret double %addtmp }
|
%a, %a ; [#uses=1] %multmp1 = fmul double 2.000000e+00, %a ; [#uses=1]
|
||||||
|
%multmp2 = fmul double %multmp1, %b ; [#uses=1] %addtmp = fadd double
|
||||||
|
%multmp, %multmp2 ; [#uses=1] %multmp3 = fmul double %b, %b ; [#uses=1]
|
||||||
|
%addtmp4 = fadd double %addtmp, %multmp3 ; [#uses=1] ret double %addtmp4
|
||||||
|
}
|
||||||
|
|
||||||
|
define double @bar(double %a) { entry: %calltmp = call double
|
||||||
|
@foo(double %a, double 4.000000e+00) ; [#uses=1] %calltmp1 = call double
|
||||||
|
@bar(double 3.133700e+04) ; [#uses=1] %addtmp = fadd double %calltmp,
|
||||||
|
%calltmp1 ; [#uses=1] ret double %addtmp }
|
||||||
|
|
||||||
|
declare double @cos(double)
|
||||||
|
|
||||||
|
define double @1() { entry: %calltmp = call double @cos(double
|
||||||
|
1.234000e+00) ; [#uses=1] ret double %calltmp }
|
||||||
|
|
||||||
declare double @cos(double)
|
|
||||||
|
|
||||||
define double @1() { entry: %calltmp = call double @cos(double
|
|
||||||
1.234000e+00) ; [#uses=1] ret double %calltmp } {% endhighlight %}
|
|
||||||
|
|
||||||
When you quit the current demo, it dumps out the IR for the entire
|
When you quit the current demo, it dumps out the IR for the entire
|
||||||
module generated. Here you can see the big picture with all the
|
module generated. Here you can see the big picture with all the
|
||||||
|
|
@ -505,65 +592,56 @@ the LLVM code generator. Because this uses the llvmpy libraries, you
|
||||||
need to `download <../download.html>`_ and
|
need to `download <../download.html>`_ and
|
||||||
`install <../userguide.html#install>`_ them.
|
`install <../userguide.html#install>`_ them.
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
|
||||||
import re from llvm.core import Module, Constant, Type, Function,
|
.. code-block:: python
|
||||||
Builder, FCMP\_ULT
|
|
||||||
|
|
||||||
Globals
|
#!/usr/bin/env python
|
||||||
-------
|
|
||||||
|
|
||||||
The LLVM module, which holds all the IR code.
|
import re from llvm.core import Module, Constant, Type, Function,
|
||||||
=============================================
|
Builder, FCMP_ULT
|
||||||
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
Globals
|
||||||
|
-------
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
# The LLVM module, which holds all the IR code.
|
||||||
=========================================================================
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
g\_llvm\_builder = None
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
|
g_llvm_builder = None
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
==============================================================================
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
|
|
||||||
and what their LLVM representation is.
|
Lexer
|
||||||
======================================
|
-----
|
||||||
|
|
||||||
g\_named\_values = {}
|
# The lexer yields one of these types for each token.
|
||||||
|
class EOFToken(object): pass
|
||||||
|
|
||||||
Lexer
|
class DefToken(object): pass
|
||||||
-----
|
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
class ExternToken(object): pass
|
||||||
===================================================
|
|
||||||
|
|
||||||
class EOFToken(object): pass
|
class IdentifierToken(object): def **init**\ (self, name): self.name =
|
||||||
|
name
|
||||||
|
|
||||||
class DefToken(object): pass
|
class NumberToken(object): def **init**\ (self, value): self.value =
|
||||||
|
value
|
||||||
|
|
||||||
class ExternToken(object): pass
|
class CharacterToken(object): def **init**\ (self, char): self.char =
|
||||||
|
char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
|
== other
|
||||||
|
|
||||||
class IdentifierToken(object): def **init**\ (self, name): self.name =
|
# Regular expressions that tokens and comments of our language.
|
||||||
name
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
class NumberToken(object): def **init**\ (self, value): self.value =
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
value
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
||||||
class CharacterToken(object): def **init**\ (self, char): self.char =
|
::
|
||||||
char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
|
||||||
== other
|
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
|
||||||
=============================================================
|
|
||||||
|
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
|
||||||
string[0].isspace(): string = string[1:] continue
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
# Run regexes.
|
# Run regexes.
|
||||||
comment_match = REGEX_COMMENT.match(string)
|
comment_match = REGEX_COMMENT.match(string)
|
||||||
|
|
@ -593,48 +671,40 @@ string[0].isspace(): string = string[1:] continue
|
||||||
yield CharacterToken(string[0])
|
yield CharacterToken(string[0])
|
||||||
string = string[1:]
|
string = string[1:]
|
||||||
|
|
||||||
yield EOFToken()
|
yield EOFToken()
|
||||||
|
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
# Expression class for numeric literals like "1.0".
|
||||||
|
class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
def **init**\ (self, value): self.value = value
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode):
|
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
# Expression class for referencing a variable, like "a".
|
||||||
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
def CodeGen(self): if self.name in g_named_values: return
|
||||||
======================================================
|
g_named_values[self.name] else: raise RuntimeError('Unknown variable
|
||||||
|
name: ' + self.name)
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
# Expression class for a binary operator.
|
||||||
|
class BinaryOperatorExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
def **init**\ (self, operator, left, right): self.operator = operator
|
||||||
|
self.left = left self.right = right
|
||||||
|
|
||||||
def CodeGen(self): if self.name in g\_named\_values: return
|
def CodeGen(self): left = self.left.CodeGen() right =
|
||||||
g\_named\_values[self.name] else: raise RuntimeError('Unknown variable
|
self.right.CodeGen()
|
||||||
name: ' + self.name)
|
|
||||||
|
|
||||||
Expression class for a binary operator.
|
::
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode):
|
|
||||||
|
|
||||||
def **init**\ (self, operator, left, right): self.operator = operator
|
|
||||||
self.left = left self.right = right
|
|
||||||
|
|
||||||
def CodeGen(self): left = self.left.CodeGen() right =
|
|
||||||
self.right.CodeGen()
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
if self.operator == '+':
|
if self.operator == '+':
|
||||||
return g_llvm_builder.fadd(left, right, 'addtmp')
|
return g_llvm_builder.fadd(left, right, 'addtmp')
|
||||||
|
|
@ -649,18 +719,16 @@ self.right.CodeGen()
|
||||||
else:
|
else:
|
||||||
raise RuntimeError('Unknown binary operator.')
|
raise RuntimeError('Unknown binary operator.')
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
class CallExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode):
|
def **init**\ (self, callee, args): self.callee = callee self.args =
|
||||||
|
args
|
||||||
|
|
||||||
def **init**\ (self, callee, args): self.callee = callee self.args =
|
def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
args
|
= g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
def CodeGen(self): # Look up the name in the global module table. callee
|
::
|
||||||
= g\_llvm\_module.get\_function\_named(self.callee)
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
# Check for argument mismatch error.
|
# Check for argument mismatch error.
|
||||||
if len(callee.args) != len(self.args):
|
if len(callee.args) != len(self.args):
|
||||||
|
|
@ -670,24 +738,18 @@ def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
==============================================================================
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes).
|
||||||
|
class PrototypeNode(object):
|
||||||
|
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
def **init**\ (self, name, args): self.name = name self.args = args
|
||||||
============================================================================
|
|
||||||
|
|
||||||
takes).
|
def CodeGen(self): # Make the function type, eg. double(double,double).
|
||||||
=======
|
funct_type = Type.function( Type.double(), [Type.double()] \*
|
||||||
|
len(self.args), False)
|
||||||
|
|
||||||
class PrototypeNode(object):
|
::
|
||||||
|
|
||||||
def **init**\ (self, name, args): self.name = name self.args = args
|
|
||||||
|
|
||||||
def CodeGen(self): # Make the function type, eg. double(double,double).
|
|
||||||
funct\_type = Type.function( Type.double(), [Type.double()] \*
|
|
||||||
len(self.args), False)
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
function = Function.new(g_llvm_module, funct_type, self.name)
|
function = Function.new(g_llvm_module, funct_type, self.name)
|
||||||
|
|
||||||
|
|
@ -714,17 +776,15 @@ len(self.args), False)
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
This class represents a function definition itself.
|
# This class represents a function definition itself.
|
||||||
===================================================
|
class FunctionNode(object):
|
||||||
|
|
||||||
class FunctionNode(object):
|
def **init**\ (self, prototype, body): self.prototype = prototype
|
||||||
|
self.body = body
|
||||||
|
|
||||||
def **init**\ (self, prototype, body): self.prototype = prototype
|
def CodeGen(self): # Clear scope. g_named_values.clear()
|
||||||
self.body = body
|
|
||||||
|
|
||||||
def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
::
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
# Create a function object.
|
# Create a function object.
|
||||||
function = self.prototype.CodeGen()
|
function = self.prototype.CodeGen()
|
||||||
|
|
@ -747,29 +807,29 @@ def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
Parser
|
Parser
|
||||||
------
|
------
|
||||||
|
|
||||||
class Parser(object):
|
class Parser(object):
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
and # updates Parser.current with its results. def Next(self):
|
and # updates Parser.current with its results. def Next(self):
|
||||||
self.current = self.tokens.next()
|
self.current = self.tokens.next()
|
||||||
|
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
isinstance(self.current, CharacterToken): return
|
isinstance(self.current, CharacterToken): return
|
||||||
self.binop\_precedence.get(self.current.char, -1) else: return -1
|
self.binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
self.Next() # eat identifier.
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
if self.current != CharacterToken('('): # Simple variable reference.
|
if self.current != CharacterToken('('): # Simple variable reference.
|
||||||
return VariableExpressionNode(identifier_name)
|
return VariableExpressionNode(identifier_name)
|
||||||
|
|
@ -789,14 +849,14 @@ self.Next() # eat identifier.
|
||||||
self.Next() # eat ')'.
|
self.Next() # eat ')'.
|
||||||
return CallExpressionNode(identifier_name, args)
|
return CallExpressionNode(identifier_name, args)
|
||||||
|
|
||||||
# numberexpr ::= number def ParseNumberExpr(self): result =
|
# numberexpr ::= number def ParseNumberExpr(self): result =
|
||||||
NumberExpressionNode(self.current.value) self.Next() # consume the
|
NumberExpressionNode(self.current.value) self.Next() # consume the
|
||||||
number. return result
|
number. return result
|
||||||
|
|
||||||
# parenexpr ::= '(' expression ')' def ParseParenExpr(self): self.Next()
|
# parenexpr ::= '(' expression ')' def ParseParenExpr(self): self.Next()
|
||||||
# eat '('.
|
# eat '('.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
contents = self.ParseExpression()
|
contents = self.ParseExpression()
|
||||||
|
|
||||||
|
|
@ -806,18 +866,18 @@ number. return result
|
||||||
|
|
||||||
return contents
|
return contents
|
||||||
|
|
||||||
# primary ::= identifierexpr \| numberexpr \| parenexpr def
|
# primary ::= identifierexpr \| numberexpr \| parenexpr def
|
||||||
ParsePrimary(self): if isinstance(self.current, IdentifierToken): return
|
ParsePrimary(self): if isinstance(self.current, IdentifierToken): return
|
||||||
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
||||||
return self.ParseNumberExpr() elif self.current == CharacterToken('('):
|
return self.ParseNumberExpr() elif self.current == CharacterToken('('):
|
||||||
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
||||||
when expecting an expression.')
|
when expecting an expression.')
|
||||||
|
|
||||||
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
# If this is a binary operator that binds at least as tightly as the
|
# If this is a binary operator that binds at least as tightly as the
|
||||||
# current one, consume it; otherwise we are done.
|
# current one, consume it; otherwise we are done.
|
||||||
|
|
@ -839,14 +899,14 @@ while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
# Merge left/right.
|
# Merge left/right.
|
||||||
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
||||||
|
|
||||||
# expression ::= primary binoprhs def ParseExpression(self): left =
|
# expression ::= primary binoprhs def ParseExpression(self): left =
|
||||||
self.ParsePrimary() return self.ParseBinOpRHS(left, 0)
|
self.ParsePrimary() return self.ParseBinOpRHS(left, 0)
|
||||||
|
|
||||||
# prototype ::= id '(' id\* ')' def ParsePrototype(self): if not
|
# prototype ::= id '(' id\* ')' def ParsePrototype(self): if not
|
||||||
isinstance(self.current, IdentifierToken): raise RuntimeError('Expected
|
isinstance(self.current, IdentifierToken): raise RuntimeError('Expected
|
||||||
function name in prototype.')
|
function name in prototype.')
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
function_name = self.current.name
|
function_name = self.current.name
|
||||||
self.Next() # eat function name.
|
self.Next() # eat function name.
|
||||||
|
|
@ -868,39 +928,39 @@ function name in prototype.')
|
||||||
|
|
||||||
return PrototypeNode(function_name, arg_names)
|
return PrototypeNode(function_name, arg_names)
|
||||||
|
|
||||||
# definition ::= 'def' prototype expression def ParseDefinition(self):
|
# definition ::= 'def' prototype expression def ParseDefinition(self):
|
||||||
self.Next() # eat def. proto = self.ParsePrototype() body =
|
self.Next() # eat def. proto = self.ParsePrototype() body =
|
||||||
self.ParseExpression() return FunctionNode(proto, body)
|
self.ParseExpression() return FunctionNode(proto, body)
|
||||||
|
|
||||||
# toplevelexpr ::= expression def ParseTopLevelExpr(self): proto =
|
# toplevelexpr ::= expression def ParseTopLevelExpr(self): proto =
|
||||||
PrototypeNode('', []) return FunctionNode(proto, self.ParseExpression())
|
PrototypeNode('', []) return FunctionNode(proto, self.ParseExpression())
|
||||||
|
|
||||||
# external ::= 'extern' prototype def ParseExtern(self): self.Next() #
|
# external ::= 'extern' prototype def ParseExtern(self): self.Next() #
|
||||||
eat extern. return self.ParsePrototype()
|
eat extern. return self.ParsePrototype()
|
||||||
|
|
||||||
# Top-Level parsing def HandleDefinition(self):
|
# Top-Level parsing def HandleDefinition(self):
|
||||||
self.Handle(self.ParseDefinition, 'Read a function definition:')
|
self.Handle(self.ParseDefinition, 'Read a function definition:')
|
||||||
|
|
||||||
def HandleExtern(self): self.Handle(self.ParseExtern, 'Read an extern:')
|
def HandleExtern(self): self.Handle(self.ParseExtern, 'Read an extern:')
|
||||||
|
|
||||||
def HandleTopLevelExpression(self): self.Handle(self.ParseTopLevelExpr,
|
def HandleTopLevelExpression(self): self.Handle(self.ParseTopLevelExpr,
|
||||||
'Read a top-level expression:')
|
'Read a top-level expression:')
|
||||||
|
|
||||||
def Handle(self, function, message): try: print message,
|
def Handle(self, function, message): try: print message,
|
||||||
function().CodeGen() except Exception, e: print 'Error:', e try:
|
function().CodeGen() except Exception, e: print 'Error:', e try:
|
||||||
self.Next() # Skip for error recovery. except: pass
|
self.Next() # Skip for error recovery. except: pass
|
||||||
|
|
||||||
Main driver code.
|
Main driver code.
|
||||||
-----------------
|
-----------------
|
||||||
|
|
||||||
def main(): # Install standard binary operators. # 1 is lowest possible
|
def main(): # Install standard binary operators. # 1 is lowest possible
|
||||||
precedence. 40 is the highest. operator\_precedence = { '<': 10, '+':
|
precedence. 40 is the highest. operator_precedence = { '<': 10, '+':
|
||||||
20, '-': 20, '\*': 40 }
|
20, '-': 20, '\*': 40 }
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: break
|
= raw_input() except KeyboardInterrupt: break
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
parser = Parser(Tokenize(raw), operator_precedence)
|
parser = Parser(Tokenize(raw), operator_precedence)
|
||||||
while True:
|
while True:
|
||||||
|
|
@ -914,10 +974,6 @@ precedence. 40 is the highest. operator\_precedence = { '<': 10, '+':
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
# Print out all of the generated code. print '', g\_llvm\_module
|
# Print out all of the generated code. print '', g_llvm_module
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Adding JIT and Optimizer Support <PythonLangImpl4.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -25,17 +25,27 @@ Our demonstration for Chapter 3 is elegant and easy to extend.
|
||||||
Unfortunately, it does not produce wonderful code. The LLVM Builder,
|
Unfortunately, it does not produce wonderful code. The LLVM Builder,
|
||||||
however, does give us obvious optimizations when compiling simple code:
|
however, does give us obvious optimizations when compiling simple code:
|
||||||
|
|
||||||
{% highlight bash %} ready> def test(x) 1+2+x Read function definition:
|
|
||||||
define double @test(double %x) { entry: %addtmp = fadd double
|
.. code-block:: bash
|
||||||
3.000000e+00, %x ret double %addtmp } {% endhighlight %}
|
|
||||||
|
ready> def test(x) 1+2+x Read function definition:
|
||||||
|
define double @test(double %x) { entry: %addtmp = fadd double
|
||||||
|
3.000000e+00, %x ret double %addtmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code is not a literal transcription of the AST built by parsing the
|
This code is not a literal transcription of the AST built by parsing the
|
||||||
input. That would be:
|
input. That would be:
|
||||||
|
|
||||||
{% highlight bash %} ready> def test(x) 1+2+x Read function definition:
|
|
||||||
define double @test(double %x) { entry: %addtmp = fadd double
|
.. code-block:: bash
|
||||||
2.000000e+00, 1.000000e+00 %addtmp1 = fadd double %addtmp, %x ret double
|
|
||||||
%addtmp1 } {% endhighlight %}
|
ready> def test(x) 1+2+x Read function definition:
|
||||||
|
define double @test(double %x) { entry: %addtmp = fadd double
|
||||||
|
2.000000e+00, 1.000000e+00 %addtmp1 = fadd double %addtmp, %x ret double
|
||||||
|
%addtmp1 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Constant folding, as seen above, in particular, is a very common and
|
Constant folding, as seen above, in particular, is a very common and
|
||||||
very important optimization: so much so that many language implementors
|
very important optimization: so much so that many language implementors
|
||||||
|
|
@ -58,11 +68,16 @@ On the other hand, the ``Builder`` is limited by the fact that it does
|
||||||
all of its analysis inline with the code as it is built. If you take a
|
all of its analysis inline with the code as it is built. If you take a
|
||||||
slightly more complex example:
|
slightly more complex example:
|
||||||
|
|
||||||
{% highlight bash %} ready> def test(x) (1+2+x)\*(x+(1+2)) Read a
|
|
||||||
function definition: define double @test(double %x) { entry: %addtmp =
|
.. code-block:: bash
|
||||||
fadd double 3.000000e+00, %x ; [#uses=1] %addtmp1 = fadd double %x,
|
|
||||||
3.000000e+00 ; [#uses=1] %multmp = fmul double %addtmp, %addtmp1 ;
|
ready> def test(x) (1+2+x)\*(x+(1+2)) Read a
|
||||||
[#uses=1] ret double %multmp } {% endhighlight %}
|
function definition: define double @test(double %x) { entry: %addtmp =
|
||||||
|
fadd double 3.000000e+00, %x ; [#uses=1] %addtmp1 = fadd double %x,
|
||||||
|
3.000000e+00 ; [#uses=1] %multmp = fmul double %addtmp, %addtmp1 ;
|
||||||
|
[#uses=1] ret double %multmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In this case, the LHS and RHS of the multiplication are the same value.
|
In this case, the LHS and RHS of the multiplication are the same value.
|
||||||
We'd really like to see this generate"``tmp = x+3; result = tmp*tmp;``
|
We'd really like to see this generate"``tmp = x+3; result = tmp*tmp;``
|
||||||
|
|
@ -112,27 +127,30 @@ to hold and organize the LLVM optimizations that we want to run. Once we
|
||||||
have that, we can add a set of optimizations to run. The code looks like
|
have that, we can add a set of optimizations to run. The code looks like
|
||||||
this:
|
this:
|
||||||
|
|
||||||
{% highlight python %} # The function optimization passes manager.
|
|
||||||
g\_llvm\_pass\_manager = FunctionPassManager.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
The LLVM execution engine.
|
.. code-block:: python
|
||||||
==========================
|
|
||||||
|
|
||||||
g\_llvm\_executor = ExecutionEngine.new(g\_llvm\_module)
|
# The function optimization passes manager.
|
||||||
|
g_llvm_pass_manager = FunctionPassManager.new(g_llvm_module)
|
||||||
|
|
||||||
...
|
# The LLVM execution engine.
|
||||||
|
g_llvm_executor = ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
|
...
|
||||||
|
|
||||||
|
def main(): # Set up the optimizer pipeline. Start with registering info
|
||||||
|
about how the # target lays out data structures.
|
||||||
|
g_llvm_pass_manager.add(g_llvm_executor.target_data) # Do simple
|
||||||
|
"peephole" optimizations and bit-twiddling optzns.
|
||||||
|
g_llvm_pass_manager.add(PASS_INSTRUCTION_COMBINING) # Reassociate
|
||||||
|
expressions. g_llvm_pass_manager.add(PASS_REASSOCIATE) # Eliminate
|
||||||
|
Common SubExpressions. g_llvm_pass_manager.add(PASS_GVN) # Simplify
|
||||||
|
the control flow graph (deleting unreachable blocks, etc).
|
||||||
|
g_llvm_pass_manager.add(PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
|
g_llvm_pass_manager.initialize()
|
||||||
|
|
||||||
def main(): # Set up the optimizer pipeline. Start with registering info
|
|
||||||
about how the # target lays out data structures.
|
|
||||||
g\_llvm\_pass\_manager.add(g\_llvm\_executor.target\_data) # Do simple
|
|
||||||
"peephole" optimizations and bit-twiddling optzns.
|
|
||||||
g\_llvm\_pass\_manager.add(PASS\_INSTRUCTION\_COMBINING) # Reassociate
|
|
||||||
expressions. g\_llvm\_pass\_manager.add(PASS\_REASSOCIATE) # Eliminate
|
|
||||||
Common SubExpressions. g\_llvm\_pass\_manager.add(PASS\_GVN) # Simplify
|
|
||||||
the control flow graph (deleting unreachable blocks, etc).
|
|
||||||
g\_llvm\_pass\_manager.add(PASS\_CFG\_SIMPLIFICATION)
|
|
||||||
|
|
||||||
g\_llvm\_pass\_manager.initialize() {% endhighlight %}
|
|
||||||
|
|
||||||
This code defines a ``FunctionPassManager``, ``g_llvm_pass_manager``.
|
This code defines a ``FunctionPassManager``, ``g_llvm_pass_manager``.
|
||||||
Once it is set up, we use a series of "add" calls to add a bunch of LLVM
|
Once it is set up, we use a series of "add" calls to add a bunch of LLVM
|
||||||
|
|
@ -149,10 +167,13 @@ Once the pass manager is set up, we need to make use of it. We do this
|
||||||
by running it after our newly created function is constructed (in
|
by running it after our newly created function is constructed (in
|
||||||
``FunctionNode.CodeGen``), but before it is returned to the client:
|
``FunctionNode.CodeGen``), but before it is returned to the client:
|
||||||
|
|
||||||
{% highlight python %} return\_value = self.body.CodeGen()
|
|
||||||
g\_llvm\_builder.ret(return\_value)
|
|
||||||
|
|
||||||
::
|
.. code-block:: python
|
||||||
|
|
||||||
|
return_value = self.body.CodeGen()
|
||||||
|
g_llvm_builder.ret(return_value)
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
# Validate the generated code, checking for consistency.
|
# Validate the generated code, checking for consistency.
|
||||||
function.verify()
|
function.verify()
|
||||||
|
|
@ -160,17 +181,24 @@ g\_llvm\_builder.ret(return\_value)
|
||||||
# Optimize the function.
|
# Optimize the function.
|
||||||
g_llvm_pass_manager.run(function)
|
g_llvm_pass_manager.run(function)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
As you can see, this is pretty straightforward. The
|
As you can see, this is pretty straightforward. The
|
||||||
``FunctionPassManager`` optimizes and updates the LLVM Function in
|
``FunctionPassManager`` optimizes and updates the LLVM Function in
|
||||||
place, improving (hopefully) its body. With this in place, we can try
|
place, improving (hopefully) its body. With this in place, we can try
|
||||||
our test above again:
|
our test above again:
|
||||||
|
|
||||||
{% highlight bash %} ready> def test(x) (1+2+x)\*(x+(1+2)) Read a
|
|
||||||
function definition: define double @test(double %x) { entry: %addtmp =
|
.. code-block:: bash
|
||||||
fadd double %x, 3.000000e+00 ; [#uses=2] %multmp = fmul double %addtmp,
|
|
||||||
%addtmp ; [#uses=1] ret double %multmp } {% endhighlight %}
|
ready> def test(x) (1+2+x)\*(x+(1+2)) Read a
|
||||||
|
function definition: define double @test(double %x) { entry: %addtmp =
|
||||||
|
fadd double %x, 3.000000e+00 ; [#uses=2] %multmp = fmul double %addtmp,
|
||||||
|
%addtmp ; [#uses=1] ret double %multmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
As expected, we now get our nicely optimized code, saving a floating
|
As expected, we now get our nicely optimized code, saving a floating
|
||||||
point add instruction from every execution of this function.
|
point add instruction from every execution of this function.
|
||||||
|
|
@ -208,8 +236,13 @@ be able to call it from the command line.
|
||||||
In order to do this, we first declare and initialize the JIT. This is
|
In order to do this, we first declare and initialize the JIT. This is
|
||||||
done by adding and initializing a global variable:
|
done by adding and initializing a global variable:
|
||||||
|
|
||||||
{% highlight python %} # The LLVM execution engine. g\_llvm\_executor =
|
|
||||||
ExecutionEngine.new(g\_llvm\_module) {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
# The LLVM execution engine. g_llvm_executor =
|
||||||
|
ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This creates an abstract "Execution Engine" which can be either a JIT
|
This creates an abstract "Execution Engine" which can be either a JIT
|
||||||
compiler or the LLVM interpreter. LLVM will automatically pick a JIT
|
compiler or the LLVM interpreter. LLVM will automatically pick a JIT
|
||||||
|
|
@ -222,38 +255,48 @@ compiled function and get its return value. In our case, this means that
|
||||||
we can change the code that parses a top-level expression to look like
|
we can change the code that parses a top-level expression to look like
|
||||||
this:
|
this:
|
||||||
|
|
||||||
{% highlight python %} def HandleTopLevelExpression(self): try: function
|
|
||||||
= self.ParseTopLevelExpr().CodeGen() result =
|
|
||||||
g\_llvm\_executor.run\_function(function, []) print 'Evaluated to:',
|
|
||||||
result.as\_real(Type.double()) except Exception, e: print 'Error:', e
|
|
||||||
try: self.Next() # Skip for error recovery. except: pass {% endhighlight
|
|
||||||
%}
|
|
||||||
|
|
||||||
Recall that we compile top-level expressions into a self-contained LLVM
|
.. code-block:: python
|
||||||
function that takes no arguments and returns the computed double.
|
|
||||||
|
|
||||||
With just these two changes, lets see how Kaleidoscope works now!
|
def HandleTopLevelExpression(self): try: function
|
||||||
|
= self.ParseTopLevelExpr().CodeGen() result =
|
||||||
|
g_llvm_executor.run_function(function, []) print 'Evaluated to:',
|
||||||
|
result.as_real(Type.double()) except Exception, e: print 'Error:', e
|
||||||
|
try: self.Next() # Skip for error recovery. except: pass {% endhighlight
|
||||||
|
%}
|
||||||
|
|
||||||
|
Recall that we compile top-level expressions into a self-contained LLVM
|
||||||
|
function that takes no arguments and returns the computed double.
|
||||||
|
|
||||||
|
With just these two changes, lets see how Kaleidoscope works now!
|
||||||
|
|
||||||
|
{% highlight python %} ready> 4+5 Read a top level expression: define
|
||||||
|
double @0() { entry: ret double 9.000000e+00 }
|
||||||
|
|
||||||
|
Evaluated to: 9.0
|
||||||
|
|
||||||
{% highlight python %} ready> 4+5 Read a top level expression: define
|
|
||||||
double @0() { entry: ret double 9.000000e+00 }
|
|
||||||
|
|
||||||
Evaluated to: 9.0 {% endhighlight %}
|
|
||||||
|
|
||||||
Well this looks like it is basically working. The dump of the function
|
Well this looks like it is basically working. The dump of the function
|
||||||
shows the "no argument function that always returns double" that we
|
shows the "no argument function that always returns double" that we
|
||||||
synthesize for each top-level expression that is typed in. This
|
synthesize for each top-level expression that is typed in. This
|
||||||
demonstrates very basic functionality, but can we do more?
|
demonstrates very basic functionality, but can we do more?
|
||||||
|
|
||||||
{% highlight python %} ready> def testfunc(x y) x + y\*2 Read a function
|
|
||||||
definition: define double @testfunc(double %x, double %y) { entry:
|
|
||||||
%multmp = fmul double %y, 2.000000e+00 ; [#uses=1] %addtmp = fadd double
|
|
||||||
%multmp, %x ; [#uses=1] ret double %addtmp }
|
|
||||||
|
|
||||||
ready> testfunc(4, 10) Read a top level expression: define double @0() {
|
.. code-block:: python
|
||||||
entry: %calltmp = call double @testfunc(double 4.000000e+00, double
|
|
||||||
1.000000e+01) ; [#uses=1] ret double %calltmp }
|
ready> def testfunc(x y) x + y\*2 Read a function
|
||||||
|
definition: define double @testfunc(double %x, double %y) { entry:
|
||||||
|
%multmp = fmul double %y, 2.000000e+00 ; [#uses=1] %addtmp = fadd double
|
||||||
|
%multmp, %x ; [#uses=1] ret double %addtmp }
|
||||||
|
|
||||||
|
ready> testfunc(4, 10) Read a top level expression: define double @0() {
|
||||||
|
entry: %calltmp = call double @testfunc(double 4.000000e+00, double
|
||||||
|
1.000000e+01) ; [#uses=1] ret double %calltmp }
|
||||||
|
|
||||||
|
*Evaluated to: 24.0*
|
||||||
|
|
||||||
|
|
||||||
*Evaluated to: 24.0* {% endhighlight %}
|
|
||||||
|
|
||||||
This illustrates that we can now call user code, but there is something
|
This illustrates that we can now call user code, but there is something
|
||||||
a bit subtle going on here. Note that we only invoke the JIT on the
|
a bit subtle going on here. Note that we only invoke the JIT on the
|
||||||
|
|
@ -269,23 +312,28 @@ etc. However, even with this simple code, we get some surprisingly
|
||||||
powerful capabilities - check this out (I removed the dump of the
|
powerful capabilities - check this out (I removed the dump of the
|
||||||
anonymous functions, you should get the idea by now :) :
|
anonymous functions, you should get the idea by now :) :
|
||||||
|
|
||||||
{% highlight bash %} ready> extern sin(x) Read an extern: declare double
|
|
||||||
@sin(double)
|
|
||||||
|
|
||||||
ready> extern cos(x) Read an extern: declare double @cos(double)
|
.. code-block:: bash
|
||||||
|
|
||||||
ready> sin(1.0) *Evaluated to: 0.841470984808*
|
ready> extern sin(x) Read an extern: declare double
|
||||||
|
@sin(double)
|
||||||
|
|
||||||
|
ready> extern cos(x) Read an extern: declare double @cos(double)
|
||||||
|
|
||||||
|
ready> sin(1.0) *Evaluated to: 0.841470984808*
|
||||||
|
|
||||||
|
ready> def foo(x) sin(x)\ *sin(x) + cos(x)*\ cos(x) Read a function
|
||||||
|
definition: define double @foo(double %x) { entry: %calltmp = call
|
||||||
|
double @sin(double %x) ; [#uses=1] %calltmp1 = call double @sin(double
|
||||||
|
%x) ; [#uses=1] %multmp = fmul double %calltmp, %calltmp1 ; [#uses=1]
|
||||||
|
%calltmp2 = call double @cos(double %x) ; [#uses=1] %calltmp3 = call
|
||||||
|
double @cos(double %x) ; [#uses=1] %multmp4 = fmul double %calltmp2,
|
||||||
|
%calltmp3 ; [#uses=1] %addtmp = fadd double %multmp, %multmp4 ;
|
||||||
|
[#uses=1] ret double %addtmp }
|
||||||
|
|
||||||
|
ready> foo(4.0) *Evaluated to: 1.000000*
|
||||||
|
|
||||||
ready> def foo(x) sin(x)\ *sin(x) + cos(x)*\ cos(x) Read a function
|
|
||||||
definition: define double @foo(double %x) { entry: %calltmp = call
|
|
||||||
double @sin(double %x) ; [#uses=1] %calltmp1 = call double @sin(double
|
|
||||||
%x) ; [#uses=1] %multmp = fmul double %calltmp, %calltmp1 ; [#uses=1]
|
|
||||||
%calltmp2 = call double @cos(double %x) ; [#uses=1] %calltmp3 = call
|
|
||||||
double @cos(double %x) ; [#uses=1] %multmp4 = fmul double %calltmp2,
|
|
||||||
%calltmp3 ; [#uses=1] %addtmp = fadd double %multmp, %multmp4 ;
|
|
||||||
[#uses=1] ret double %addtmp }
|
|
||||||
|
|
||||||
ready> foo(4.0) *Evaluated to: 1.000000* {% endhighlight %}
|
|
||||||
|
|
||||||
Whoa, how does the JIT know about sin and cos? The answer is
|
Whoa, how does the JIT know about sin and cos? The answer is
|
||||||
surprisingly simple: in this example, the JIT started execution of a
|
surprisingly simple: in this example, the JIT started execution of a
|
||||||
|
|
@ -301,27 +349,32 @@ One interesting application of this is that we can now extend the
|
||||||
language by writing arbitrary C++ code to implement operations. For
|
language by writing arbitrary C++ code to implement operations. For
|
||||||
example, we can create a C file with the following simple function:
|
example, we can create a C file with the following simple function:
|
||||||
|
|
||||||
{% highlight c %} #include
|
|
||||||
|
|
||||||
double putchard(double x) { putchar((char)x); return 0; } {%
|
.. code-block:: c
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
We can then compile this into a shared library with GCC:
|
#include
|
||||||
|
|
||||||
{% highlight bash %} gcc -shared -fPIC -o putchard.so putchard.c {%
|
double putchard(double x) { putchar((char)x); return 0; } {%
|
||||||
endhighlight %}
|
endhighlight %}
|
||||||
|
|
||||||
Now we can load this library into the Python process using
|
We can then compile this into a shared library with GCC:
|
||||||
``llvm.core.load_library_permanently`` and access it from Kaleidoscope
|
|
||||||
to produce simple output to the console:
|
{% highlight bash %} gcc -shared -fPIC -o putchard.so putchard.c {%
|
||||||
|
endhighlight %}
|
||||||
|
|
||||||
|
Now we can load this library into the Python process using
|
||||||
|
``llvm.core.load_library_permanently`` and access it from Kaleidoscope
|
||||||
|
to produce simple output to the console:
|
||||||
|
|
||||||
|
{% highlight python %} >>> import llvm.core >>>
|
||||||
|
llvm.core.load_library_permanently('/home/max/llvmpy-tutorial/putchard.so')
|
||||||
|
>>> import kaleidoscope >>> kaleidoscope.main() ready> extern
|
||||||
|
putchard(x) Read an extern: declare double @putchard(double)
|
||||||
|
|
||||||
|
ready> putchard(65) + putchard(66) + putchard(67) + putchard(10) *ABC*
|
||||||
|
Evaluated to: 0.0
|
||||||
|
|
||||||
{% highlight python %} >>> import llvm.core >>>
|
|
||||||
llvm.core.load\_library\_permanently('/home/max/llvmpy-tutorial/putchard.so')
|
|
||||||
>>> import kaleidoscope >>> kaleidoscope.main() ready> extern
|
|
||||||
putchard(x) Read an extern: declare double @putchard(double)
|
|
||||||
|
|
||||||
ready> putchard(65) + putchard(66) + putchard(67) + putchard(10) *ABC*
|
|
||||||
Evaluated to: 0.0 {% endhighlight %}
|
|
||||||
|
|
||||||
Similar code could be used to implement file I/O, console input, and
|
Similar code could be used to implement file I/O, console input, and
|
||||||
many other capabilities in Kaleidoscope.
|
many other capabilities in Kaleidoscope.
|
||||||
|
|
@ -341,78 +394,65 @@ Full Code Listing # {#code}
|
||||||
Here is the complete code listing for our running example, enhanced with
|
Here is the complete code listing for our running example, enhanced with
|
||||||
the LLVM JIT and optimizer:
|
the LLVM JIT and optimizer:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
|
||||||
import re from llvm.core import Module, Constant, Type, Function,
|
.. code-block:: python
|
||||||
Builder, FCMP\_ULT from llvm.ee import ExecutionEngine, TargetData from
|
|
||||||
llvm.passes import FunctionPassManager from llvm.passes import
|
|
||||||
(PASS\_INSTRUCTION\_COMBINING, PASS\_REASSOCIATE, PASS\_GVN,
|
|
||||||
PASS\_CFG\_SIMPLIFICATION)
|
|
||||||
|
|
||||||
Globals
|
#!/usr/bin/env python
|
||||||
-------
|
|
||||||
|
|
||||||
The LLVM module, which holds all the IR code.
|
import re from llvm.core import Module, Constant, Type, Function,
|
||||||
=============================================
|
Builder, FCMP_ULT from llvm.ee import ExecutionEngine, TargetData from
|
||||||
|
llvm.passes import FunctionPassManager from llvm.passes import
|
||||||
|
(PASS_INSTRUCTION_COMBINING, PASS_REASSOCIATE, PASS_GVN,
|
||||||
|
PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
Globals
|
||||||
|
-------
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
# The LLVM module, which holds all the IR code.
|
||||||
=========================================================================
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
g\_llvm\_builder = None
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
|
g_llvm_builder = None
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
==============================================================================
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
|
|
||||||
and what their LLVM representation is.
|
# The function optimization passes manager.
|
||||||
======================================
|
g_llvm_pass_manager = FunctionPassManager.new(g_llvm_module)
|
||||||
|
|
||||||
g\_named\_values = {}
|
# The LLVM execution engine.
|
||||||
|
g_llvm_executor = ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
The function optimization passes manager.
|
Lexer
|
||||||
=========================================
|
-----
|
||||||
|
|
||||||
g\_llvm\_pass\_manager = FunctionPassManager.new(g\_llvm\_module)
|
# The lexer yields one of these types for each token.
|
||||||
|
class EOFToken(object): pass
|
||||||
|
|
||||||
The LLVM execution engine.
|
class DefToken(object): pass
|
||||||
==========================
|
|
||||||
|
|
||||||
g\_llvm\_executor = ExecutionEngine.new(g\_llvm\_module)
|
class ExternToken(object): pass
|
||||||
|
|
||||||
Lexer
|
class IdentifierToken(object): def **init**\ (self, name): self.name =
|
||||||
-----
|
name
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
class NumberToken(object): def **init**\ (self, value): self.value =
|
||||||
===================================================
|
value
|
||||||
|
|
||||||
class EOFToken(object): pass
|
class CharacterToken(object): def **init**\ (self, char): self.char =
|
||||||
|
char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
|
== other
|
||||||
|
|
||||||
class DefToken(object): pass
|
# Regular expressions that tokens and comments of our language.
|
||||||
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
class ExternToken(object): pass
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
||||||
class IdentifierToken(object): def **init**\ (self, name): self.name =
|
::
|
||||||
name
|
|
||||||
|
|
||||||
class NumberToken(object): def **init**\ (self, value): self.value =
|
|
||||||
value
|
|
||||||
|
|
||||||
class CharacterToken(object): def **init**\ (self, char): self.char =
|
|
||||||
char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
|
||||||
== other
|
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
|
||||||
=============================================================
|
|
||||||
|
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
|
||||||
string[0].isspace(): string = string[1:] continue
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
# Run regexes.
|
# Run regexes.
|
||||||
comment_match = REGEX_COMMENT.match(string)
|
comment_match = REGEX_COMMENT.match(string)
|
||||||
|
|
@ -442,48 +482,40 @@ string[0].isspace(): string = string[1:] continue
|
||||||
yield CharacterToken(string[0])
|
yield CharacterToken(string[0])
|
||||||
string = string[1:]
|
string = string[1:]
|
||||||
|
|
||||||
yield EOFToken()
|
yield EOFToken()
|
||||||
|
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
# Expression class for numeric literals like "1.0".
|
||||||
|
class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
def **init**\ (self, value): self.value = value
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode):
|
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
# Expression class for referencing a variable, like "a".
|
||||||
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
def CodeGen(self): if self.name in g_named_values: return
|
||||||
======================================================
|
g_named_values[self.name] else: raise RuntimeError('Unknown variable
|
||||||
|
name: ' + self.name)
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
# Expression class for a binary operator.
|
||||||
|
class BinaryOperatorExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
def **init**\ (self, operator, left, right): self.operator = operator
|
||||||
|
self.left = left self.right = right
|
||||||
|
|
||||||
def CodeGen(self): if self.name in g\_named\_values: return
|
def CodeGen(self): left = self.left.CodeGen() right =
|
||||||
g\_named\_values[self.name] else: raise RuntimeError('Unknown variable
|
self.right.CodeGen()
|
||||||
name: ' + self.name)
|
|
||||||
|
|
||||||
Expression class for a binary operator.
|
::
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode):
|
|
||||||
|
|
||||||
def **init**\ (self, operator, left, right): self.operator = operator
|
|
||||||
self.left = left self.right = right
|
|
||||||
|
|
||||||
def CodeGen(self): left = self.left.CodeGen() right =
|
|
||||||
self.right.CodeGen()
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
if self.operator == '+':
|
if self.operator == '+':
|
||||||
return g_llvm_builder.fadd(left, right, 'addtmp')
|
return g_llvm_builder.fadd(left, right, 'addtmp')
|
||||||
|
|
@ -498,18 +530,16 @@ self.right.CodeGen()
|
||||||
else:
|
else:
|
||||||
raise RuntimeError('Unknown binary operator.')
|
raise RuntimeError('Unknown binary operator.')
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
class CallExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode):
|
def **init**\ (self, callee, args): self.callee = callee self.args =
|
||||||
|
args
|
||||||
|
|
||||||
def **init**\ (self, callee, args): self.callee = callee self.args =
|
def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
args
|
= g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
def CodeGen(self): # Look up the name in the global module table. callee
|
::
|
||||||
= g\_llvm\_module.get\_function\_named(self.callee)
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
# Check for argument mismatch error.
|
# Check for argument mismatch error.
|
||||||
if len(callee.args) != len(self.args):
|
if len(callee.args) != len(self.args):
|
||||||
|
|
@ -519,24 +549,18 @@ def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
==============================================================================
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes).
|
||||||
|
class PrototypeNode(object):
|
||||||
|
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
def **init**\ (self, name, args): self.name = name self.args = args
|
||||||
============================================================================
|
|
||||||
|
|
||||||
takes).
|
def CodeGen(self): # Make the function type, eg. double(double,double).
|
||||||
=======
|
funct_type = Type.function( Type.double(), [Type.double()] \*
|
||||||
|
len(self.args), False)
|
||||||
|
|
||||||
class PrototypeNode(object):
|
::
|
||||||
|
|
||||||
def **init**\ (self, name, args): self.name = name self.args = args
|
|
||||||
|
|
||||||
def CodeGen(self): # Make the function type, eg. double(double,double).
|
|
||||||
funct\_type = Type.function( Type.double(), [Type.double()] \*
|
|
||||||
len(self.args), False)
|
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
function = Function.new(g_llvm_module, funct_type, self.name)
|
function = Function.new(g_llvm_module, funct_type, self.name)
|
||||||
|
|
||||||
|
|
@ -563,17 +587,15 @@ len(self.args), False)
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
This class represents a function definition itself.
|
# This class represents a function definition itself.
|
||||||
===================================================
|
class FunctionNode(object):
|
||||||
|
|
||||||
class FunctionNode(object):
|
def **init**\ (self, prototype, body): self.prototype = prototype
|
||||||
|
self.body = body
|
||||||
|
|
||||||
def **init**\ (self, prototype, body): self.prototype = prototype
|
def CodeGen(self): # Clear scope. g_named_values.clear()
|
||||||
self.body = body
|
|
||||||
|
|
||||||
def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
::
|
||||||
|
|
||||||
::
|
|
||||||
|
|
||||||
# Create a function object.
|
# Create a function object.
|
||||||
function = self.prototype.CodeGen()
|
function = self.prototype.CodeGen()
|
||||||
|
|
@ -599,29 +621,29 @@ def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
Parser
|
Parser
|
||||||
------
|
------
|
||||||
|
|
||||||
class Parser(object):
|
class Parser(object):
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
and # updates Parser.current with its results. def Next(self):
|
and # updates Parser.current with its results. def Next(self):
|
||||||
self.current = self.tokens.next()
|
self.current = self.tokens.next()
|
||||||
|
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
isinstance(self.current, CharacterToken): return
|
isinstance(self.current, CharacterToken): return
|
||||||
self.binop\_precedence.get(self.current.char, -1) else: return -1
|
self.binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
self.Next() # eat identifier.
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
if self.current != CharacterToken('('): # Simple variable reference.
|
if self.current != CharacterToken('('): # Simple variable reference.
|
||||||
return VariableExpressionNode(identifier_name)
|
return VariableExpressionNode(identifier_name)
|
||||||
|
|
@ -641,14 +663,14 @@ self.Next() # eat identifier.
|
||||||
self.Next() # eat ')'.
|
self.Next() # eat ')'.
|
||||||
return CallExpressionNode(identifier_name, args)
|
return CallExpressionNode(identifier_name, args)
|
||||||
|
|
||||||
# numberexpr ::= number def ParseNumberExpr(self): result =
|
# numberexpr ::= number def ParseNumberExpr(self): result =
|
||||||
NumberExpressionNode(self.current.value) self.Next() # consume the
|
NumberExpressionNode(self.current.value) self.Next() # consume the
|
||||||
number. return result
|
number. return result
|
||||||
|
|
||||||
# parenexpr ::= '(' expression ')' def ParseParenExpr(self): self.Next()
|
# parenexpr ::= '(' expression ')' def ParseParenExpr(self): self.Next()
|
||||||
# eat '('.
|
# eat '('.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
contents = self.ParseExpression()
|
contents = self.ParseExpression()
|
||||||
|
|
||||||
|
|
@ -658,18 +680,18 @@ number. return result
|
||||||
|
|
||||||
return contents
|
return contents
|
||||||
|
|
||||||
# primary ::= identifierexpr \| numberexpr \| parenexpr def
|
# primary ::= identifierexpr \| numberexpr \| parenexpr def
|
||||||
ParsePrimary(self): if isinstance(self.current, IdentifierToken): return
|
ParsePrimary(self): if isinstance(self.current, IdentifierToken): return
|
||||||
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
||||||
return self.ParseNumberExpr() elif self.current == CharacterToken('('):
|
return self.ParseNumberExpr() elif self.current == CharacterToken('('):
|
||||||
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
||||||
when expecting an expression.')
|
when expecting an expression.')
|
||||||
|
|
||||||
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
# If this is a binary operator that binds at least as tightly as the
|
# If this is a binary operator that binds at least as tightly as the
|
||||||
# current one, consume it; otherwise we are done.
|
# current one, consume it; otherwise we are done.
|
||||||
|
|
@ -691,14 +713,14 @@ while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
# Merge left/right.
|
# Merge left/right.
|
||||||
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
||||||
|
|
||||||
# expression ::= primary binoprhs def ParseExpression(self): left =
|
# expression ::= primary binoprhs def ParseExpression(self): left =
|
||||||
self.ParsePrimary() return self.ParseBinOpRHS(left, 0)
|
self.ParsePrimary() return self.ParseBinOpRHS(left, 0)
|
||||||
|
|
||||||
# prototype ::= id '(' id\* ')' def ParsePrototype(self): if not
|
# prototype ::= id '(' id\* ')' def ParsePrototype(self): if not
|
||||||
isinstance(self.current, IdentifierToken): raise RuntimeError('Expected
|
isinstance(self.current, IdentifierToken): raise RuntimeError('Expected
|
||||||
function name in prototype.')
|
function name in prototype.')
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
function_name = self.current.name
|
function_name = self.current.name
|
||||||
self.Next() # eat function name.
|
self.Next() # eat function name.
|
||||||
|
|
@ -720,54 +742,54 @@ function name in prototype.')
|
||||||
|
|
||||||
return PrototypeNode(function_name, arg_names)
|
return PrototypeNode(function_name, arg_names)
|
||||||
|
|
||||||
# definition ::= 'def' prototype expression def ParseDefinition(self):
|
# definition ::= 'def' prototype expression def ParseDefinition(self):
|
||||||
self.Next() # eat def. proto = self.ParsePrototype() body =
|
self.Next() # eat def. proto = self.ParsePrototype() body =
|
||||||
self.ParseExpression() return FunctionNode(proto, body)
|
self.ParseExpression() return FunctionNode(proto, body)
|
||||||
|
|
||||||
# toplevelexpr ::= expression def ParseTopLevelExpr(self): proto =
|
# toplevelexpr ::= expression def ParseTopLevelExpr(self): proto =
|
||||||
PrototypeNode('', []) return FunctionNode(proto, self.ParseExpression())
|
PrototypeNode('', []) return FunctionNode(proto, self.ParseExpression())
|
||||||
|
|
||||||
# external ::= 'extern' prototype def ParseExtern(self): self.Next() #
|
# external ::= 'extern' prototype def ParseExtern(self): self.Next() #
|
||||||
eat extern. return self.ParsePrototype()
|
eat extern. return self.ParsePrototype()
|
||||||
|
|
||||||
# Top-Level parsing def HandleDefinition(self):
|
# Top-Level parsing def HandleDefinition(self):
|
||||||
self.Handle(self.ParseDefinition, 'Read a function definition:')
|
self.Handle(self.ParseDefinition, 'Read a function definition:')
|
||||||
|
|
||||||
def HandleExtern(self): self.Handle(self.ParseExtern, 'Read an extern:')
|
def HandleExtern(self): self.Handle(self.ParseExtern, 'Read an extern:')
|
||||||
|
|
||||||
def HandleTopLevelExpression(self): try: function =
|
def HandleTopLevelExpression(self): try: function =
|
||||||
self.ParseTopLevelExpr().CodeGen() result =
|
self.ParseTopLevelExpr().CodeGen() result =
|
||||||
g\_llvm\_executor.run\_function(function, []) print 'Evaluated to:',
|
g_llvm_executor.run_function(function, []) print 'Evaluated to:',
|
||||||
result.as\_real(Type.double()) except Exception, e: print 'Error:', e
|
result.as_real(Type.double()) except Exception, e: print 'Error:', e
|
||||||
try: self.Next() # Skip for error recovery. except: pass
|
try: self.Next() # Skip for error recovery. except: pass
|
||||||
|
|
||||||
def Handle(self, function, message): try: print message,
|
def Handle(self, function, message): try: print message,
|
||||||
function().CodeGen() except Exception, e: print 'Error:', e try:
|
function().CodeGen() except Exception, e: print 'Error:', e try:
|
||||||
self.Next() # Skip for error recovery. except: pass
|
self.Next() # Skip for error recovery. except: pass
|
||||||
|
|
||||||
Main driver code.
|
Main driver code.
|
||||||
-----------------
|
-----------------
|
||||||
|
|
||||||
def main(): # Set up the optimizer pipeline. Start with registering info
|
def main(): # Set up the optimizer pipeline. Start with registering info
|
||||||
about how the # target lays out data structures.
|
about how the # target lays out data structures.
|
||||||
g\_llvm\_pass\_manager.add(g\_llvm\_executor.target\_data) # Do simple
|
g_llvm_pass_manager.add(g_llvm_executor.target_data) # Do simple
|
||||||
"peephole" optimizations and bit-twiddling optzns.
|
"peephole" optimizations and bit-twiddling optzns.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_INSTRUCTION\_COMBINING) # Reassociate
|
g_llvm_pass_manager.add(PASS_INSTRUCTION_COMBINING) # Reassociate
|
||||||
expressions. g\_llvm\_pass\_manager.add(PASS\_REASSOCIATE) # Eliminate
|
expressions. g_llvm_pass_manager.add(PASS_REASSOCIATE) # Eliminate
|
||||||
Common SubExpressions. g\_llvm\_pass\_manager.add(PASS\_GVN) # Simplify
|
Common SubExpressions. g_llvm_pass_manager.add(PASS_GVN) # Simplify
|
||||||
the control flow graph (deleting unreachable blocks, etc).
|
the control flow graph (deleting unreachable blocks, etc).
|
||||||
g\_llvm\_pass\_manager.add(PASS\_CFG\_SIMPLIFICATION)
|
g_llvm_pass_manager.add(PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
g\_llvm\_pass\_manager.initialize()
|
g_llvm_pass_manager.initialize()
|
||||||
|
|
||||||
# Install standard binary operators. # 1 is lowest possible precedence.
|
# Install standard binary operators. # 1 is lowest possible precedence.
|
||||||
40 is the highest. operator\_precedence = { '<': 10, '+': 20, '-': 20,
|
40 is the highest. operator_precedence = { '<': 10, '+': 20, '-': 20,
|
||||||
'\*': 40 }
|
'\*': 40 }
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: break
|
= raw_input() except KeyboardInterrupt: break
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
parser = Parser(Tokenize(raw), operator_precedence)
|
parser = Parser(Tokenize(raw), operator_precedence)
|
||||||
while True:
|
while True:
|
||||||
|
|
@ -781,10 +803,6 @@ g\_llvm\_pass\_manager.initialize()
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
# Print out all of the generated code. print '', g\_llvm\_module
|
# Print out all of the generated code. print '', g_llvm_module
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Extending the language: control flow <PythonLangImpl5.html>`_**
|
|
||||||
|
|
|
||||||
File diff suppressed because it is too large
Load diff
File diff suppressed because it is too large
Load diff
File diff suppressed because it is too large
Load diff
|
|
@ -1,277 +0,0 @@
|
||||||
*****************************************************************
|
|
||||||
Chapter 8: Conclusion and other useful LLVM tidbits
|
|
||||||
*****************************************************************
|
|
||||||
|
|
||||||
Written by `Chris Lattner <mailto:sabre@nondot.org>`_
|
|
||||||
|
|
||||||
|
|
||||||
Tutorial Conclusion # {#conclusion}
|
|
||||||
===================================
|
|
||||||
|
|
||||||
Welcome to the the final chapter of the `Implementing a language with
|
|
||||||
LLVM <http://www.llvm.org/docs/tutorial/index.html>`_ tutorial. In the
|
|
||||||
course of this tutorial, we have grown our little Kaleidoscope language
|
|
||||||
from being a useless toy, to being a semi-interesting (but probably
|
|
||||||
still useless) toy. :)
|
|
||||||
|
|
||||||
It is interesting to see how far we've come, and how little code it has
|
|
||||||
taken. We built the entire lexer, parser, AST, code generator, and an
|
|
||||||
interactive run-loop (with a JIT!) by-hand in under 540 lines of
|
|
||||||
(non-comment/non-blank) code.
|
|
||||||
|
|
||||||
Our little language supports a couple of interesting features: it
|
|
||||||
supports user defined binary and unary operators, it uses JIT
|
|
||||||
compilation for immediate evaluation, and it supports a few control flow
|
|
||||||
constructs with SSA construction.
|
|
||||||
|
|
||||||
Part of the idea of this tutorial was to show you how easy and fun it
|
|
||||||
can be to define, build, and play with languages. Building a compiler
|
|
||||||
need not be a scary or mystical process! Now that you've seen some of
|
|
||||||
the basics, I strongly encourage you to take the code and hack on it.
|
|
||||||
For example, try adding:
|
|
||||||
|
|
||||||
- **global variables** -- While global variables have questional value
|
|
||||||
in modern software engineering, they are often useful when putting
|
|
||||||
together quick little hacks like the Kaleidoscope compiler itself.
|
|
||||||
Fortunately, our current setup makes it very easy to add global
|
|
||||||
variables: just have value lookup check to see if an unresolved
|
|
||||||
variable is in the global variable symbol table before rejecting it.
|
|
||||||
To create a new global variable, make an instance of the LLVM
|
|
||||||
``GlobalVariable`` class.
|
|
||||||
|
|
||||||
- **typed variables** -- Kaleidoscope currently only supports variables
|
|
||||||
of type double. This gives the language a very nice elegance, because
|
|
||||||
only supporting one type means that you never have to specify types.
|
|
||||||
Different languages have different ways of handling this. The easiest
|
|
||||||
way is to require the user to specify types for every variable
|
|
||||||
definition, and record the type of the variable in the symbol table
|
|
||||||
along with its Value\*.
|
|
||||||
|
|
||||||
- **arrays, structs, vectors, etc** -- Once you add types, you can
|
|
||||||
start extending the type system in all sorts of interesting ways.
|
|
||||||
Simple arrays are very easy and are quite useful for many different
|
|
||||||
applications. Adding them is mostly an exercise in learning how the
|
|
||||||
LLVM
|
|
||||||
`getelementptr <http://www.llvm.org/docs/LangRef.html#i_getelementptr>`_
|
|
||||||
instruction works: it is so nifty/unconventional, it `has its own
|
|
||||||
FAQ <http://www.llvm.org/docs/GetElementPtr.html>`_! If you add
|
|
||||||
support for recursive types (e.g. linked lists), make sure to read
|
|
||||||
the `section in the LLVM Programmer's
|
|
||||||
Manual <http://www.llvm.org/docs/ProgrammersManual.html#TypeResolve>`_
|
|
||||||
that describes how to construct them.
|
|
||||||
|
|
||||||
- **standard runtime** -- Our current language allows the user to
|
|
||||||
access arbitrary external functions, and we use it for things like
|
|
||||||
"putchard". As you extend the language to add higher-level
|
|
||||||
constructs, often these constructs make the most sense if they are
|
|
||||||
lowered to calls into a language-supplied runtime. For example, if
|
|
||||||
you add hash tables to the language, it would probably make sense to
|
|
||||||
add the routines to a runtime, instead of inlining them all the way.
|
|
||||||
|
|
||||||
- **memory management** -- Currently we can only access the stack in
|
|
||||||
Kaleidoscope. It would also be useful to be able to allocate heap
|
|
||||||
memory, either with calls to the standard libc malloc/free interface
|
|
||||||
or with a garbage collector. If you would like to use garbage
|
|
||||||
collection, note that LLVM fully supports `Accurate Garbage
|
|
||||||
Collection <http://www.llvm.org/docs/GarbageCollection.html>`_
|
|
||||||
including algorithms that move objects and need to scan/update the
|
|
||||||
stack.
|
|
||||||
|
|
||||||
- **debugger support** -- LLVM supports generation of `DWARF Debug
|
|
||||||
info <http://www.llvm.org/docs/SourceLevelDebugging.html>`_ which is
|
|
||||||
understood by common debuggers like GDB. Adding support for debug
|
|
||||||
info is fairly straightforward. The best way to understand it is to
|
|
||||||
compile some C/C++ code with "``llvm-gcc -g -O0``\ " and taking a
|
|
||||||
look at what it produces.
|
|
||||||
|
|
||||||
- **exception handling support** - LLVM supports generation of `zero
|
|
||||||
cost exceptions <http://www.llvm.org/docs/ExceptionHandling.html>`_
|
|
||||||
which interoperate with code compiled in other languages. You could
|
|
||||||
also generate code by implicitly making every function return an
|
|
||||||
error value and checking it. You could also make explicit use of
|
|
||||||
setjmp/longjmp. There are many different ways to go here.
|
|
||||||
|
|
||||||
- **object orientation, generics, database access, complex numbers,
|
|
||||||
geometric programming, ...** -- Really, there is no end of crazy
|
|
||||||
features that you can add to the language.
|
|
||||||
|
|
||||||
- **unusual domains** -- We've been talking about applying LLVM to a
|
|
||||||
domain that many people are interested in: building a compiler for a
|
|
||||||
specific language. However, there are many other domains that can use
|
|
||||||
compiler technology that are not typically considered. For example,
|
|
||||||
LLVM has been used to implement OpenGL graphics acceleration,
|
|
||||||
translate C++ code to ActionScript, and many other cute and clever
|
|
||||||
things. Maybe you will be the first to JIT compile a regular
|
|
||||||
expression interpreter into native code with LLVM?
|
|
||||||
|
|
||||||
Have fun - try doing something crazy and unusual. Building a language
|
|
||||||
like everyone else always has, is much less fun than trying something a
|
|
||||||
little crazy or off the wall and seeing how it turns out. If you get
|
|
||||||
stuck or want to talk about it, feel free to email the `llvmdev mailing
|
|
||||||
list <http://lists.cs.uiuc.edu/mailman/listinfo/llvmdev>`_: it has lots
|
|
||||||
of people who are interested in languages and are often willing to help
|
|
||||||
out.
|
|
||||||
|
|
||||||
Before we end this tutorial, I want to talk about some "tips and tricks"
|
|
||||||
for generating LLVM IR. These are some of the more subtle things that
|
|
||||||
may not be obvious, but are very useful if you want to take advantage of
|
|
||||||
LLVM's capabilities.
|
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
Properties of the LLVM IR # {#llvmirproperties}
|
|
||||||
===============================================
|
|
||||||
|
|
||||||
We have a couple common questions about code in the LLVM IR form - let's
|
|
||||||
just get these out of the way right now, shall we?
|
|
||||||
|
|
||||||
Target Independence ## {#targetindep}
|
|
||||||
-------------------------------------
|
|
||||||
|
|
||||||
Kaleidoscope is an example of a "portable language": any program written
|
|
||||||
in Kaleidoscope will work the same way on any target that it runs on.
|
|
||||||
Many other languages have this property, e.g. LISP, Java, Haskell,
|
|
||||||
Javascript, Python, etc. (note that while these languages are portable,
|
|
||||||
not all their libraries are).
|
|
||||||
|
|
||||||
One nice aspect of LLVM is that it is often capable of preserving target
|
|
||||||
independence in the IR: you can take the LLVM IR for a
|
|
||||||
Kaleidoscope-compiled program and run it on any target that LLVM
|
|
||||||
supports, even emitting C code and compiling that on targets that LLVM
|
|
||||||
doesn't support natively. You can trivially tell that the Kaleidoscope
|
|
||||||
compiler generates target-independent code because it never queries for
|
|
||||||
any target-specific information when generating code.
|
|
||||||
|
|
||||||
The fact that LLVM provides a compact, target-independent,
|
|
||||||
representation for code gets a lot of people excited. Unfortunately,
|
|
||||||
these people are usually thinking about C or a language from the C
|
|
||||||
family when they are asking questions about language portability. I say
|
|
||||||
"unfortunately", because there is really no way to make (fully general)
|
|
||||||
C code portable, other than shipping the source code around (and of
|
|
||||||
course, C source code is not actually portable in general either - ever
|
|
||||||
port a really old application from 32- to 64-bits?).
|
|
||||||
|
|
||||||
The problem with C (again, in its full generality) is that it is heavily
|
|
||||||
laden with target specific assumptions. As one simple example, the
|
|
||||||
preprocessor often destructively removes target-independence from the
|
|
||||||
code when it processes the input text:
|
|
||||||
|
|
||||||
{% highlight c %} #ifdef **i386** int X = 1; #else int X = 42; #endif {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
While it is possible to engineer more and more complex solutions to
|
|
||||||
problems like this, it cannot be solved in full generality in a way that
|
|
||||||
is better than shipping the actual source code.
|
|
||||||
|
|
||||||
That said, there are interesting subsets of C that can be made portable.
|
|
||||||
If you are willing to fix primitive types to a fixed size (say int =
|
|
||||||
32-bits, and long = 64-bits), don't care about ABI compatibility with
|
|
||||||
existing binaries, and are willing to give up some other minor features,
|
|
||||||
you can have portable code. This can make sense for specialized domains
|
|
||||||
such as an in-kernel language.
|
|
||||||
|
|
||||||
Safety Guarantees ## {#safety}
|
|
||||||
------------------------------
|
|
||||||
|
|
||||||
Many of the languages above are also "safe" languages: it is impossible
|
|
||||||
for a program written in Java to corrupt its address space and crash the
|
|
||||||
process (assuming the JVM has no bugs). Safety is an interesting
|
|
||||||
property that requires a combination of language design, runtime
|
|
||||||
support, and often operating system support.
|
|
||||||
|
|
||||||
It is certainly possible to implement a safe language in LLVM, but LLVM
|
|
||||||
IR does not itself guarantee safety. The LLVM IR allows unsafe pointer
|
|
||||||
casts, use after free bugs, buffer over-runs, and a variety of other
|
|
||||||
problems. Safety needs to be implemented as a layer on top of LLVM and,
|
|
||||||
conveniently, several groups have investigated this. Ask on the `llvmdev
|
|
||||||
mailing list <http://lists.cs.uiuc.edu/mailman/listinfo/llvmdev>`_ if
|
|
||||||
you are interested in more details.
|
|
||||||
|
|
||||||
Language-Specific Optimizations ## {#langspecific}
|
|
||||||
--------------------------------------------------
|
|
||||||
|
|
||||||
One thing about LLVM that turns off many people is that it does not
|
|
||||||
solve all the world's problems in one system (sorry 'world hunger',
|
|
||||||
someone else will have to solve you some other day). One specific
|
|
||||||
complaint is that people perceive LLVM as being incapable of performing
|
|
||||||
high-level language-specific optimization: LLVM "loses too much
|
|
||||||
information".
|
|
||||||
|
|
||||||
Unfortunately, this is really not the place to give you a full and
|
|
||||||
unified version of "Chris Lattner's theory of compiler design". Instead,
|
|
||||||
I'll make a few observations:
|
|
||||||
|
|
||||||
First, you're right that LLVM does lose information. For example, as of
|
|
||||||
this writing, there is no way to distinguish in the LLVM IR whether an
|
|
||||||
SSA-value came from a C "int" or a C "long" on an ILP32 machine (other
|
|
||||||
than debug info). Both get compiled down to an 'i32' value and the
|
|
||||||
information about what it came from is lost. The more general issue
|
|
||||||
here, is that the LLVM type system uses "structural equivalence" instead
|
|
||||||
of "name equivalence". Another place this surprises people is if you
|
|
||||||
have two types in a high-level language that have the same structure
|
|
||||||
(e.g. two different structs that have a single int field): these types
|
|
||||||
will compile down into a single LLVM type and it will be impossible to
|
|
||||||
tell what it came from.
|
|
||||||
|
|
||||||
Second, while LLVM does lose information, LLVM is not a fixed target: we
|
|
||||||
continue to enhance and improve it in many different ways. In addition
|
|
||||||
to adding new features (LLVM did not always support exceptions or debug
|
|
||||||
info), we also extend the IR to capture important information for
|
|
||||||
optimization (e.g. whether an argument is sign or zero extended,
|
|
||||||
information about pointers aliasing, etc). Many of the enhancements are
|
|
||||||
user-driven: people want LLVM to include some specific feature, so they
|
|
||||||
go ahead and extend it.
|
|
||||||
|
|
||||||
Third, it is *possible and easy* to add language-specific optimizations,
|
|
||||||
and you have a number of choices in how to do it. As one trivial
|
|
||||||
example, it is easy to add language-specific optimization passes that
|
|
||||||
"know" things about code compiled for a language. In the case of the C
|
|
||||||
family, there is an optimization pass that "knows" about the standard C
|
|
||||||
library functions. If you call "exit(0)" in main(), it knows that it is
|
|
||||||
safe to optimize that into "return 0;" because C specifies what the
|
|
||||||
'exit' function does.
|
|
||||||
|
|
||||||
In addition to simple library knowledge, it is possible to embed a
|
|
||||||
variety of other language-specific information into the LLVM IR. If you
|
|
||||||
have a specific need and run into a wall, please bring the topic up on
|
|
||||||
the llvmdev list. At the very worst, you can always treat LLVM as if it
|
|
||||||
were a "dumb code generator" and implement the high-level optimizations
|
|
||||||
you desire in your front-end, on the language-specific AST.
|
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
Tips and Tricks # {#tipsandtricks}
|
|
||||||
==================================
|
|
||||||
|
|
||||||
There is a variety of useful tips and tricks that you come to know after
|
|
||||||
working on/with LLVM that aren't obvious at first glance. Instead of
|
|
||||||
letting everyone rediscover them, this section talks about some of these
|
|
||||||
issues.
|
|
||||||
|
|
||||||
Implementing portable offsetof/sizeof ## {#offsetofsizeof}
|
|
||||||
----------------------------------------------------------
|
|
||||||
|
|
||||||
One interesting thing that comes up, if you are trying to keep the code
|
|
||||||
generated by your compiler "target independent", is that you often need
|
|
||||||
to know the size of some LLVM type or the offset of some field in an
|
|
||||||
llvm structure. For example, you might need to pass the size of a type
|
|
||||||
into a function that allocates memory.
|
|
||||||
|
|
||||||
Unfortunately, this can vary widely across targets: for example the
|
|
||||||
width of a pointer is trivially target-specific. However, there is a
|
|
||||||
`clever way to use the getelementptr
|
|
||||||
instruction <http://nondot.org/sabre/LLVMNotes/SizeOf-OffsetOf-VariableSizedStructs.txt>`_
|
|
||||||
that allows you to compute this in a portable way.
|
|
||||||
|
|
||||||
Garbage Collected Stack Frames ## {#gcstack}
|
|
||||||
--------------------------------------------
|
|
||||||
|
|
||||||
Some languages want to explicitly manage their stack frames, often so
|
|
||||||
that they are garbage collected or to allow easy implementation of
|
|
||||||
closures. There are often better ways to implement these features than
|
|
||||||
explicit stack frames, but `LLVM does support
|
|
||||||
them <http://nondot.org/sabre/LLVMNotes/ExplicitlyManagedStackFrames.txt>`_,
|
|
||||||
if you want. It requires your front-end to convert the code into
|
|
||||||
`Continuation Passing
|
|
||||||
Style <http://en.wikipedia.org/wiki/Continuation-passing_style>`_ and
|
|
||||||
the use of tail calls (which LLVM also supports).
|
|
||||||
|
|
@ -11,344 +11,343 @@ created from Python constants. A constant expression is also a constant
|
||||||
etc) can be specified, to yield a new ``Constant`` object. Let's see
|
etc) can be specified, to yield a new ``Constant`` object. Let's see
|
||||||
some examples:
|
some examples:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
|
||||||
ti = Type.int() # a 32-bit int type
|
.. code-block:: python
|
||||||
|
|
||||||
k1 = Constant.int(ti, 42) # "int k1 = 42;" k2 = k1.add( Constant.int(
|
#!/usr/bin/env python
|
||||||
ti, 10 ) ) # "int k2 = k1 + 10;"
|
|
||||||
|
|
||||||
tr = Type.float()
|
ti = Type.int() # a 32-bit int type
|
||||||
|
|
||||||
r1 = Constant.real(tr, "3.141592") # create from a string r2 =
|
k1 = Constant.int(ti, 42) # "int k1 = 42;" k2 = k1.add( Constant.int(
|
||||||
Constant.real(tr, 1.61803399) # create from a Python float {%
|
ti, 10 ) ) # "int k2 = k1 + 10;"
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
llvm.core.Constant
|
tr = Type.float()
|
||||||
==================
|
|
||||||
|
|
||||||
- This will become a table of contents (this text will be scraped).
|
r1 = Constant.real(tr, "3.141592") # create from a string r2 =
|
||||||
|
Constant.real(tr, 1.61803399) # create from a Python float {%
|
||||||
|
endhighlight %}
|
||||||
|
|
||||||
|
# llvm.core.Constant
|
||||||
|
- This will become a table of contents (this text will be scraped).
|
||||||
{:toc}
|
{:toc}
|
||||||
|
|
||||||
Static factory methods
|
Static factory methods
|
||||||
----------------------
|
----------------------
|
||||||
|
|
||||||
``null(ty)``
|
``null(ty)``
|
||||||
~~~~~~~~~~~~
|
~~~~~~~~~~~~
|
||||||
|
|
||||||
A null value (all zeros) of type ``ty``
|
A null value (all zeros) of type ``ty``
|
||||||
|
|
||||||
``all_ones(ty)``
|
``all_ones(ty)``
|
||||||
~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
All 1's value of type ``ty``
|
All 1's value of type ``ty``
|
||||||
|
|
||||||
``undef(ty)``
|
``undef(ty)``
|
||||||
~~~~~~~~~~~~~
|
~~~~~~~~~~~~~
|
||||||
|
|
||||||
An undefined value of type ``ty``
|
An undefined value of type ``ty``
|
||||||
|
|
||||||
``int(ty, value)``
|
``int(ty, value)``
|
||||||
~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Integer of type ``ty``, with value ``value`` (a Python int or long)
|
Integer of type ``ty``, with value ``value`` (a Python int or long)
|
||||||
|
|
||||||
``int_signextend(ty, value)``
|
``int_signextend(ty, value)``
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Integer of signed type ``ty`` (use for signed types)
|
Integer of signed type ``ty`` (use for signed types)
|
||||||
|
|
||||||
``real(ty, value)``
|
``real(ty, value)``
|
||||||
~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Floating point value of type ``ty``, with value ``value`` (a Python
|
Floating point value of type ``ty``, with value ``value`` (a Python
|
||||||
float)
|
float)
|
||||||
|
|
||||||
``stringz(value)``
|
``stringz(value)``
|
||||||
~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
A null-terminated string. ``value`` is a Python string
|
A null-terminated string. ``value`` is a Python string
|
||||||
|
|
||||||
``string(value)``
|
``string(value)``
|
||||||
~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
As ``string(ty)``, but not null terminated
|
As ``string(ty)``, but not null terminated
|
||||||
|
|
||||||
``array(ty, consts)``
|
``array(ty, consts)``
|
||||||
~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Array of type ``ty``, initialized with ``consts`` (an iterable yielding
|
Array of type ``ty``, initialized with ``consts`` (an iterable yielding
|
||||||
``Constant`` objects of the appropriate type)
|
``Constant`` objects of the appropriate type)
|
||||||
|
|
||||||
``struct(ty, consts)``
|
``struct(ty, consts)``
|
||||||
~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Struct (unpacked) of type ``ty``, initialized with ``consts`` (an
|
Struct (unpacked) of type ``ty``, initialized with ``consts`` (an
|
||||||
iterable yielding ``Constant`` objects of the appropriate type)
|
iterable yielding ``Constant`` objects of the appropriate type)
|
||||||
|
|
||||||
``packed_struct(ty, consts)``
|
``packed_struct(ty, consts)``
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
As ``struct(ty, consts)`` but packed
|
As ``struct(ty, consts)`` but packed
|
||||||
|
|
||||||
``vector(consts)``
|
``vector(consts)``
|
||||||
~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Vector, initialized with ``consts`` (an iterable yielding ``Constant``
|
Vector, initialized with ``consts`` (an iterable yielding ``Constant``
|
||||||
objects of the appropriate type)
|
objects of the appropriate type)
|
||||||
|
|
||||||
``sizeof(ty)``
|
``sizeof(ty)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Constant value representing the sizeof the type ``ty``
|
Constant value representing the sizeof the type ``ty``
|
||||||
|
|
||||||
Methods
|
Methods
|
||||||
-------
|
-------
|
||||||
|
|
||||||
The following operations on constants are supported. For more details on
|
The following operations on constants are supported. For more details on
|
||||||
any operation, consult the `Constant
|
any operation, consult the `Constant
|
||||||
Expressions <http://www.llvm.org/docs/LangRef.html#constantexprs>`_
|
Expressions <http://www.llvm.org/docs/LangRef.html#constantexprs>`_
|
||||||
section of the LLVM Language Reference.
|
section of the LLVM Language Reference.
|
||||||
|
|
||||||
``k.neg()``
|
``k.neg()``
|
||||||
~~~~~~~~~~~
|
~~~~~~~~~~~
|
||||||
|
|
||||||
negation, same as ``0 - k``
|
negation, same as ``0 - k``
|
||||||
|
|
||||||
``k.not_()``
|
``k.not_()``
|
||||||
~~~~~~~~~~~~
|
~~~~~~~~~~~~
|
||||||
|
|
||||||
1's complement of ``k``. Note trailing underscore.
|
1's complement of ``k``. Note trailing underscore.
|
||||||
|
|
||||||
``k.add(k2)``
|
``k.add(k2)``
|
||||||
~~~~~~~~~~~~~
|
~~~~~~~~~~~~~
|
||||||
|
|
||||||
``k + k2``, where ``k`` and ``k2`` are integers.
|
``k + k2``, where ``k`` and ``k2`` are integers.
|
||||||
|
|
||||||
``k.fadd(k2)``
|
``k.fadd(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
``k + k2``, where ``k`` and ``k2`` are floating-point.
|
``k + k2``, where ``k`` and ``k2`` are floating-point.
|
||||||
|
|
||||||
``k.sub(k2)``
|
``k.sub(k2)``
|
||||||
~~~~~~~~~~~~~
|
~~~~~~~~~~~~~
|
||||||
|
|
||||||
``k - k2``, where ``k`` and ``k2`` are integers.
|
``k - k2``, where ``k`` and ``k2`` are integers.
|
||||||
|
|
||||||
``k.fsub(k2)``
|
``k.fsub(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
``k - k2``, where ``k`` and ``k2`` are floating-point.
|
``k - k2``, where ``k`` and ``k2`` are floating-point.
|
||||||
|
|
||||||
``k.mul(k2)``
|
``k.mul(k2)``
|
||||||
~~~~~~~~~~~~~
|
~~~~~~~~~~~~~
|
||||||
|
|
||||||
``k * k2``, where ``k`` and ``k2`` are integers.
|
``k * k2``, where ``k`` and ``k2`` are integers.
|
||||||
|
|
||||||
``k.fmul(k2)``
|
``k.fmul(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
``k * k2``, where ``k`` and ``k2`` are floating-point.
|
``k * k2``, where ``k`` and ``k2`` are floating-point.
|
||||||
|
|
||||||
``k.udiv(k2)``
|
``k.udiv(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Quotient of unsigned division of ``k`` with ``k2``
|
Quotient of unsigned division of ``k`` with ``k2``
|
||||||
|
|
||||||
``k.sdiv(k2)``
|
``k.sdiv(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Quotient of signed division of ``k`` with ``k2``
|
Quotient of signed division of ``k`` with ``k2``
|
||||||
|
|
||||||
``k.fdiv(k2)``
|
``k.fdiv(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Quotient of floating point division of ``k`` with ``k2``
|
Quotient of floating point division of ``k`` with ``k2``
|
||||||
|
|
||||||
``k.urem(k2)``
|
``k.urem(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Reminder of unsigned division of ``k`` with ``k2``
|
Reminder of unsigned division of ``k`` with ``k2``
|
||||||
|
|
||||||
``k.srem(k2)``
|
``k.srem(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Reminder of signed division of ``k`` with ``k2``
|
Reminder of signed division of ``k`` with ``k2``
|
||||||
|
|
||||||
``k.frem(k2)``
|
``k.frem(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Reminder of floating point division of ``k`` with ``k2``
|
Reminder of floating point division of ``k`` with ``k2``
|
||||||
|
|
||||||
``k.and_(k2)``
|
``k.and_(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Bitwise and of ``k`` and ``k2``. Note trailing underscore.
|
Bitwise and of ``k`` and ``k2``. Note trailing underscore.
|
||||||
|
|
||||||
``k.or_(k2)``
|
``k.or_(k2)``
|
||||||
~~~~~~~~~~~~~
|
~~~~~~~~~~~~~
|
||||||
|
|
||||||
Bitwise or of ``k`` and ``k2``. Note trailing underscore.
|
Bitwise or of ``k`` and ``k2``. Note trailing underscore.
|
||||||
|
|
||||||
``k.xor(k2)``
|
``k.xor(k2)``
|
||||||
~~~~~~~~~~~~~
|
~~~~~~~~~~~~~
|
||||||
|
|
||||||
Bitwise exclusive-or of ``k`` and ``k2``.
|
Bitwise exclusive-or of ``k`` and ``k2``.
|
||||||
|
|
||||||
``k.icmp(icmp, k2)``
|
``k.icmp(icmp, k2)``
|
||||||
~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Compare ``k`` with ``k2`` using the predicate ``icmp``. See
|
Compare ``k`` with ``k2`` using the predicate ``icmp``. See
|
||||||
`here <comparision.html#icmp>`_ for list of predicates for integer
|
`here <comparision.html#icmp>`_ for list of predicates for integer
|
||||||
operands.
|
operands.
|
||||||
|
|
||||||
``k.fcmp(fcmp, k2)``
|
``k.fcmp(fcmp, k2)``
|
||||||
~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Compare ``k`` with ``k2`` using the predicate ``fcmp``. See
|
Compare ``k`` with ``k2`` using the predicate ``fcmp``. See
|
||||||
`here <comparision.html#fcmp>`_ for list of predicates for real
|
`here <comparision.html#fcmp>`_ for list of predicates for real
|
||||||
operands.
|
operands.
|
||||||
|
|
||||||
``k.shl(k2)``
|
``k.shl(k2)``
|
||||||
~~~~~~~~~~~~~
|
~~~~~~~~~~~~~
|
||||||
|
|
||||||
Shift ``k`` left by ``k2`` bits.
|
Shift ``k`` left by ``k2`` bits.
|
||||||
|
|
||||||
``k.lshr(k2)``
|
``k.lshr(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Shift ``k`` logically right by ``k2`` bits (new bits are 0s).
|
Shift ``k`` logically right by ``k2`` bits (new bits are 0s).
|
||||||
|
|
||||||
``k.ashr(k2)``
|
``k.ashr(k2)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Shift ``k`` arithmetically right by ``k2`` bits (new bits are same as
|
Shift ``k`` arithmetically right by ``k2`` bits (new bits are same as
|
||||||
previous sign bit).
|
previous sign bit).
|
||||||
|
|
||||||
``k.gep(indices)``
|
``k.gep(indices)``
|
||||||
~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
GEP, see `LLVM docs <http://www.llvm.org/docs/GetElementPtr.html>`_.
|
GEP, see `LLVM docs <http://www.llvm.org/docs/GetElementPtr.html>`_.
|
||||||
|
|
||||||
``k.trunc(ty)``
|
``k.trunc(ty)``
|
||||||
~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Truncate ``k`` to a type ``ty`` of lower bitwidth.
|
Truncate ``k`` to a type ``ty`` of lower bitwidth.
|
||||||
|
|
||||||
``k.sext(ty)``
|
``k.sext(ty)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Sign extend ``k`` to a type ``ty`` of higher bitwidth, while extending
|
Sign extend ``k`` to a type ``ty`` of higher bitwidth, while extending
|
||||||
the sign bit.
|
the sign bit.
|
||||||
|
|
||||||
``k.zext(ty)``
|
``k.zext(ty)``
|
||||||
~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Sign extend ``k`` to a type ``ty`` of higher bitwidth, all new bits are
|
Sign extend ``k`` to a type ``ty`` of higher bitwidth, all new bits are
|
||||||
0s.
|
0s.
|
||||||
|
|
||||||
``k.fptrunc(ty)``
|
``k.fptrunc(ty)``
|
||||||
~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Truncate floating point constant ``k`` to floating point type ``ty`` of
|
Truncate floating point constant ``k`` to floating point type ``ty`` of
|
||||||
lower size than k's.
|
lower size than k's.
|
||||||
|
|
||||||
``k.fpext(ty)``
|
``k.fpext(ty)``
|
||||||
~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Extend floating point constant ``k`` to floating point type ``ty`` of
|
Extend floating point constant ``k`` to floating point type ``ty`` of
|
||||||
higher size than k's.
|
higher size than k's.
|
||||||
|
|
||||||
``k.uitofp(ty)``
|
``k.uitofp(ty)``
|
||||||
~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Convert an unsigned integer constant ``k`` to floating point constant of
|
Convert an unsigned integer constant ``k`` to floating point constant of
|
||||||
type ``ty``.
|
type ``ty``.
|
||||||
|
|
||||||
``k.sitofp(ty)``
|
``k.sitofp(ty)``
|
||||||
~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Convert a signed integer constant ``k`` to floating point constant of
|
Convert a signed integer constant ``k`` to floating point constant of
|
||||||
type ``ty``.
|
type ``ty``.
|
||||||
|
|
||||||
``k.fptoui(ty)``
|
``k.fptoui(ty)``
|
||||||
~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Convert a floating point constant ``k`` to an unsigned integer constant
|
Convert a floating point constant ``k`` to an unsigned integer constant
|
||||||
of type ``ty``.
|
of type ``ty``.
|
||||||
|
|
||||||
``k.fptosi(ty)``
|
``k.fptosi(ty)``
|
||||||
~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Convert a floating point constant ``k`` to a signed integer constant of
|
Convert a floating point constant ``k`` to a signed integer constant of
|
||||||
type ``ty``.
|
type ``ty``.
|
||||||
|
|
||||||
``k.ptrtoint(ty)``
|
``k.ptrtoint(ty)``
|
||||||
~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Convert a pointer constant ``k`` to an integer constant of type ``ty``.
|
Convert a pointer constant ``k`` to an integer constant of type ``ty``.
|
||||||
|
|
||||||
``k.inttoptr(ty)``
|
``k.inttoptr(ty)``
|
||||||
~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Convert an integer constant ``k`` to a pointer constant of type ``ty``.
|
Convert an integer constant ``k`` to a pointer constant of type ``ty``.
|
||||||
|
|
||||||
``k.bitcast(ty)``
|
``k.bitcast(ty)``
|
||||||
~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Convert ``k`` to a (equal-width) constant of type ``ty``.
|
Convert ``k`` to a (equal-width) constant of type ``ty``.
|
||||||
|
|
||||||
``k.select(cond,k2,k3)``
|
``k.select(cond,k2,k3)``
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Replace value with ``k2`` if the 1-bit integer constant ``cond`` is 1,
|
Replace value with ``k2`` if the 1-bit integer constant ``cond`` is 1,
|
||||||
else with ``k3``.
|
else with ``k3``.
|
||||||
|
|
||||||
``k.extract_element(idx)``
|
``k.extract_element(idx)``
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Extract value at ``idx`` (integer constant) from a vector constant
|
Extract value at ``idx`` (integer constant) from a vector constant
|
||||||
``k``.
|
``k``.
|
||||||
|
|
||||||
``k.insert_element(k2,idx)``
|
``k.insert_element(k2,idx)``
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Insert value ``k2`` (scalar constant) at index ``idx`` (integer
|
Insert value ``k2`` (scalar constant) at index ``idx`` (integer
|
||||||
constant) of vector constant ``k``.
|
constant) of vector constant ``k``.
|
||||||
|
|
||||||
``k.shuffle_vector(k2,mask)``
|
``k.shuffle_vector(k2,mask)``
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Shuffle vector constant ``k`` based on vector constants ``k2`` and
|
Shuffle vector constant ``k`` based on vector constants ``k2`` and
|
||||||
``mask``.
|
``mask``.
|
||||||
|
|
||||||
--------------
|
--------------
|
||||||
|
|
||||||
Other Constant Classes
|
# Other Constant Classes
|
||||||
======================
|
The following subclasses of ``Constant`` do not provide additional
|
||||||
|
methods, **they serve only to provide richer type information.**
|
||||||
|
|
||||||
The following subclasses of ``Constant`` do not provide additional
|
Subclass \| LLVM C++ Class \| Remarks \|
|
||||||
methods, **they serve only to provide richer type information.**
|
---------\|----------------\|---------\| ``ConstantExpr`` \|
|
||||||
|
``llvmConstantExpr`` \| A constant expression \|
|
||||||
|
``ConstantAggregateZero``\ \| ``llvmConstantAggregateZero``\ \| All-zero
|
||||||
|
constant \| ``ConstantInt``\ \| ``llvmConstantInt``\ \| An integer
|
||||||
|
constant \| ``ConstantFP``\ \| ``llvmConstantFP``\ \| A floating-point
|
||||||
|
constant \| ``ConstantArray``\ \| ``llvmConstantArray``\ \| An array
|
||||||
|
constant \| ``ConstantStruct``\ \| ``llvmConstantStruct``\ \| A
|
||||||
|
structure constant \| ``ConstantVector``\ \| ``llvmConstantVector``\ \|
|
||||||
|
A vector constant \| ``ConstantPointerNull``\ \|
|
||||||
|
``llvmConstantPointerNull``\ \| All-zero pointer constant \|
|
||||||
|
``UndefValue``\ \| ``llvmUndefValue``\ \| corresponds to ``undef`` of
|
||||||
|
LLVM IR \|
|
||||||
|
|
||||||
Subclass \| LLVM C++ Class \| Remarks \|
|
These types are helpful in ``isinstance`` checks, like so:
|
||||||
---------\|----------------\|---------\| ``ConstantExpr`` \|
|
|
||||||
``llvmConstantExpr`` \| A constant expression \|
|
|
||||||
``ConstantAggregateZero``\ \| ``llvmConstantAggregateZero``\ \| All-zero
|
|
||||||
constant \| ``ConstantInt``\ \| ``llvmConstantInt``\ \| An integer
|
|
||||||
constant \| ``ConstantFP``\ \| ``llvmConstantFP``\ \| A floating-point
|
|
||||||
constant \| ``ConstantArray``\ \| ``llvmConstantArray``\ \| An array
|
|
||||||
constant \| ``ConstantStruct``\ \| ``llvmConstantStruct``\ \| A
|
|
||||||
structure constant \| ``ConstantVector``\ \| ``llvmConstantVector``\ \|
|
|
||||||
A vector constant \| ``ConstantPointerNull``\ \|
|
|
||||||
``llvmConstantPointerNull``\ \| All-zero pointer constant \|
|
|
||||||
``UndefValue``\ \| ``llvmUndefValue``\ \| corresponds to ``undef`` of
|
|
||||||
LLVM IR \|
|
|
||||||
|
|
||||||
These types are helpful in ``isinstance`` checks, like so:
|
{% highlight python %} ti = Type.int(32) k1 = Constant.int(ti, 42) #
|
||||||
|
int32_t k1 = 42; k2 = Constant.array(ti, [k1, k1]) # int32_t k2[] = {
|
||||||
|
k1, k1 };
|
||||||
|
|
||||||
{% highlight python %} ti = Type.int(32) k1 = Constant.int(ti, 42) #
|
assert isinstance(k1, ConstantInt) assert isinstance(k2, ConstantArray)
|
||||||
int32\_t k1 = 42; k2 = Constant.array(ti, [k1, k1]) # int32\_t k2[] = {
|
|
||||||
k1, k1 };
|
|
||||||
|
|
||||||
assert isinstance(k1, ConstantInt) assert isinstance(k2, ConstantArray)
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
|
||||||
|
|
@ -39,14 +39,10 @@ Returns an iterable object that yields `Type <llvm.core.Type.html>`_
|
||||||
objects that represent, in order, the types of the arguments accepted by
|
objects that represent, in order, the types of the arguments accepted by
|
||||||
the function. Used like this:
|
the function. Used like this:
|
||||||
|
|
||||||
{% highlight python %} func\_type = Type.function( Type.int(), [
|
|
||||||
Type.int(), Type.int() ] ) for arg in func\_type.args: assert arg.kind
|
|
||||||
== TYPE\_INTEGER assert arg == Type.int() assert func\_type.arg\_count
|
|
||||||
== len(func\_type.args) {% endhighlight %}
|
|
||||||
|
|
||||||
``arg_count``
|
.. code-block:: python
|
||||||
~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
func_type = Type.function( Type.int(), [
|
||||||
|
Type.int(), Type.int() ] ) for arg in func_type.args: assert arg.kind
|
||||||
The number of arguments. Same as ``len(obj.args)``, but faster.
|
== TYPE_INTEGER assert arg == Type.int() assert func_type.arg_count
|
||||||
|
== len(func_type.args)
|
||||||
|
|
|
||||||
|
|
@ -11,93 +11,29 @@ marked as constants. Global variables can be created either by using the
|
||||||
``add_global_variable`` method of the `Module <llvm.core.Module.html>`_
|
``add_global_variable`` method of the `Module <llvm.core.Module.html>`_
|
||||||
class, or by using the static method ``GlobalVariable.new``.
|
class, or by using the static method ``GlobalVariable.new``.
|
||||||
|
|
||||||
{% highlight python %} # create a global variable using
|
|
||||||
add\_global\_variable method gv1 =
|
|
||||||
module\_obj.add\_global\_variable(Type.int(), "gv1")
|
|
||||||
|
|
||||||
or equivalently, using a static constructor method
|
.. code-block:: python
|
||||||
==================================================
|
|
||||||
|
|
||||||
gv2 = GlobalVariable.new(module\_obj, Type.int(), "gv2") {% endhighlight
|
# create a global variable using
|
||||||
%}
|
add_global_variable method gv1 =
|
||||||
|
module_obj.add_global_variable(Type.int(), "gv1")
|
||||||
|
|
||||||
Existing global variables of a module can be accessed by name using
|
# or equivalently, using a static constructor method
|
||||||
``module_obj.get_global_variable_named(name)`` or
|
gv2 = GlobalVariable.new(module_obj, Type.int(), "gv2") {% endhighlight
|
||||||
``GlobalVariable.get``. All existing global variables can be enumerated
|
%}
|
||||||
via iterating over the property ``module_obj.global_variables``.
|
|
||||||
|
|
||||||
{% highlight python %} # retrieve a reference to the global variable
|
Existing global variables of a module can be accessed by name using
|
||||||
gv1, # using the get\_global\_variable\_named method gv1 =
|
``module_obj.get_global_variable_named(name)`` or
|
||||||
module\_obj.get\_global\_variable\_named("gv1")
|
``GlobalVariable.get``. All existing global variables can be enumerated
|
||||||
|
via iterating over the property ``module_obj.global_variables``.
|
||||||
|
|
||||||
or equivalently, using the static ``get`` method:
|
{% highlight python %} # retrieve a reference to the global variable
|
||||||
=================================================
|
gv1, # using the get_global_variable_named method gv1 =
|
||||||
|
module_obj.get_global_variable_named("gv1")
|
||||||
|
|
||||||
gv2 = GlobalVariable.get(module\_obj, "gv2")
|
# or equivalently, using the static ``get`` method:
|
||||||
|
gv2 = GlobalVariable.get(module_obj, "gv2")
|
||||||
|
|
||||||
list all global variables in a module
|
# list all global variables in a module
|
||||||
=====================================
|
for gv in module_obj.global_variables: print gv.name, "of type",
|
||||||
|
gv.type
|
||||||
for gv in module\_obj.global\_variables: print gv.name, "of type",
|
|
||||||
gv.type {% endhighlight %}
|
|
||||||
|
|
||||||
The initializer for a global variable can be set by assigning to the
|
|
||||||
``initializer`` property of the object. The ``is_global_constant``
|
|
||||||
property can be used to indicate that the variable is a global constant.
|
|
||||||
|
|
||||||
Global variables can be delete using the ``delete`` method. Do not use
|
|
||||||
the object after calling ``delete`` on it.
|
|
||||||
|
|
||||||
{% highlight python %} # add an initializer 10 (32-bit integer)
|
|
||||||
gv.initializer = Constant.int( Type.int(), 10 )
|
|
||||||
|
|
||||||
delete the global
|
|
||||||
=================
|
|
||||||
|
|
||||||
gv.delete() # DO NOT dereference \`gv' beyond this point! gv = None {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
llvm.core.GlobalVariable
|
|
||||||
========================
|
|
||||||
|
|
||||||
Base Class
|
|
||||||
----------
|
|
||||||
|
|
||||||
- `llvm.core.GlobalValue <llvm.core.GlobalValue.html>`_
|
|
||||||
|
|
||||||
Static Constructors
|
|
||||||
-------------------
|
|
||||||
|
|
||||||
``new(module_obj, ty, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Create a global variable named ``name`` of type ``ty`` in the module
|
|
||||||
``module_obj`` and return a ``GlobalVariable`` object that represents
|
|
||||||
it.
|
|
||||||
|
|
||||||
``get(module_obj, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Return a ``GlobalVariable`` object to represent the global variable
|
|
||||||
named ``name`` in the module ``module_obj`` or raise ``LLVMException``
|
|
||||||
if such a variable does not exist.
|
|
||||||
|
|
||||||
Properties
|
|
||||||
----------
|
|
||||||
|
|
||||||
``initializer``
|
|
||||||
~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
The intializer of the variable. Set to
|
|
||||||
`llvm.core.Constant <llvm.core.Constant.html>`_ (or derived). Gets the
|
|
||||||
initializer constant, or ``None`` if none exists. ``global_constant``
|
|
||||||
``True`` if the variable is a global constant, ``False`` otherwise.
|
|
||||||
|
|
||||||
Methods
|
|
||||||
-------
|
|
||||||
|
|
||||||
``delete()``
|
|
||||||
~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Deletes the global variable from it's module. **Do not hold any
|
|
||||||
references to this object after calling ``delete`` on it.**
|
|
||||||
|
|
|
||||||
|
|
@ -8,226 +8,12 @@ Modules are top-level container objects. You need to create a module
|
||||||
object first, before you can add global variables, aliases or functions.
|
object first, before you can add global variables, aliases or functions.
|
||||||
Modules are created using the static method ``Module.new``:
|
Modules are created using the static method ``Module.new``:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
|
||||||
from llvm import \* from llvm.core import \*
|
.. code-block:: python
|
||||||
|
|
||||||
create a module
|
#!/usr/bin/env python
|
||||||
===============
|
|
||||||
|
|
||||||
my\_module = Module.new('my\_module') {% endhighlight %}
|
from llvm import \* from llvm.core import \*
|
||||||
|
|
||||||
The constructor of the Module class should *not* be used to instantiate
|
# create a module
|
||||||
a Module object. This is a common feature for all llvmpy classes.
|
my_module = Module.new('my_module')
|
||||||
|
|
||||||
**Convention**
|
|
||||||
|
|
||||||
*All* llvmpy objects are instantiated using static methods of
|
|
||||||
corresponding classes. Constructors *should not* be used.
|
|
||||||
|
|
||||||
The argument ``my_module`` is a module identifier (a plain string).
|
|
||||||
A module can also be constructed via deserialization from a bit code
|
|
||||||
file, using the static method ``from_bitcode``. This method takes a
|
|
||||||
file-like object as argument, i.e., it should have a ``read()``
|
|
||||||
method that returns the entire data in a single call, as is the case
|
|
||||||
with the builtin file object. Here is an example:
|
|
||||||
|
|
||||||
{% highlight python %} # create a module from a bit code file bcfile =
|
|
||||||
file("test.bc") my\_module = Module.from\_bitcode(bcfile) {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
There is corresponding serialization method also, called ``to_bitcode``:
|
|
||||||
|
|
||||||
{% highlight python %} # write out a bit code file from the module
|
|
||||||
bcfile = file("test.bc", "w") my\_module.to\_bitcode(bcfile) {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
Modules can also be constructed from LLVM assembly files (``.ll``
|
|
||||||
files). The static method ``from_assembly`` can be used for this.
|
|
||||||
Similar to the ``from_bitcode`` method, this one also takes a file-like
|
|
||||||
object as argument:
|
|
||||||
|
|
||||||
{% highlight python %} # create a module from an assembly file llfile =
|
|
||||||
file("test.ll") my\_module = Module.from\_assembly(llfile) {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
Modules can be converted into their assembly representation by
|
|
||||||
stringifying them (see below).
|
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
llvm.core.Module
|
|
||||||
================
|
|
||||||
|
|
||||||
- This will become a table of contents (this text will be scraped).
|
|
||||||
{:toc}
|
|
||||||
|
|
||||||
Static Constructors
|
|
||||||
-------------------
|
|
||||||
|
|
||||||
``new(module_id)``
|
|
||||||
~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Create a new ``Module`` instance with given ``module_id``. The
|
|
||||||
``module_id`` should be a string.
|
|
||||||
|
|
||||||
``from_bitcode(fileobj)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Create a new ``Module`` instance by deserializing the bitcode file
|
|
||||||
represented by the file-like object ``fileobj``.
|
|
||||||
|
|
||||||
``from_assembly(fileobj)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Create a new ``Module`` instance by parsing the LLVM assembly file
|
|
||||||
represented by the file-like object ``fileobj``.
|
|
||||||
|
|
||||||
Properties
|
|
||||||
----------
|
|
||||||
|
|
||||||
``data_layout``
|
|
||||||
~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
A string representing the ABI of the platform.
|
|
||||||
|
|
||||||
``target``
|
|
||||||
~~~~~~~~~~
|
|
||||||
|
|
||||||
A string like ``i386-pc-linux-gnu`` or ``i386-pc-solaris2.8``.
|
|
||||||
|
|
||||||
``pointer_size``
|
|
||||||
~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
The size in bits of pointers, of the target platform. A value of zero
|
|
||||||
represents ``llvm::Module::AnyPointerSize``.
|
|
||||||
|
|
||||||
``global_variables``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
An iterable that yields
|
|
||||||
`GlobalVariable <llvm.core.GlobalVariable.html>`_ objects, that
|
|
||||||
represent the global variables of the module.
|
|
||||||
|
|
||||||
``functions``
|
|
||||||
~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
An iterable that yields `Function <llvm.core.Function.html>`_ objects,
|
|
||||||
that represent functions in the module.
|
|
||||||
|
|
||||||
``id``
|
|
||||||
~~~~~~
|
|
||||||
|
|
||||||
A string that represents the module identifier (name).
|
|
||||||
|
|
||||||
Methods
|
|
||||||
-------
|
|
||||||
|
|
||||||
``get_type_named(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Return a `StructType <llvm.core.StructType.html>`_ object for the given
|
|
||||||
name.
|
|
||||||
|
|
||||||
The definition of this method was changed to work with LLVM 3.0+, in
|
|
||||||
which the type system was rewritten. See `LLVM
|
|
||||||
Blog <http://blog.llvm.org/2011/11/llvm-30-type-system-rewrite.html>`_.
|
|
||||||
|
|
||||||
{% comment %} ++++++++REMOVED+++++++++++ ### ``add_type_name(name, ty)``
|
|
||||||
|
|
||||||
Add an alias (typedef) for the type ``ty`` with the name ``name``.
|
|
||||||
|
|
||||||
``delete_type_name(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Delete an alias with the name ``name``. ++++++++END-REMOVED+++++++++++
|
|
||||||
{% endcomment %}
|
|
||||||
|
|
||||||
``add_global_variable(ty, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Add a global variable of the type ``ty`` with the name ``name``. Returns
|
|
||||||
a `GlobalVariable <llvm.core.GlobalVariable.html>`_ object.
|
|
||||||
|
|
||||||
``get_global_variable_named(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Get a `GlobalVariable <llvm.core.GlobalVariable.html>`_ object
|
|
||||||
corresponding to the global variable with the name ``name``. Raises
|
|
||||||
``LLVMException`` if such a variable does not exist.
|
|
||||||
|
|
||||||
``add_library(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Add a dependent library to the Module. This only adds a name to a list
|
|
||||||
of dependent library. **No linking is performed**.
|
|
||||||
|
|
||||||
``add_function(ty, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Add a function named ``name`` with the function type ``ty``. ``ty`` must
|
|
||||||
of an object of type `FunctionType <llvm.core.FunctionType.html>`_.
|
|
||||||
|
|
||||||
``get_function_named(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Get a `Function <llvm.core.Function.html>`_ object corresponding to the
|
|
||||||
function with the name ``name``. Raises ``LLVMException`` if such a
|
|
||||||
function does not exist.
|
|
||||||
|
|
||||||
``get_or_insert_function(ty, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Like ``get_function_named``, but adds the function first, if not present
|
|
||||||
(like ``add_function``).
|
|
||||||
|
|
||||||
``verify()``
|
|
||||||
~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Verify the correctness of the module. Raises ``LLVMException`` on
|
|
||||||
errors.
|
|
||||||
|
|
||||||
``to_bitcode(fileobj)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Write the bitcode representation of the module to the file-like object
|
|
||||||
``fileobj``.
|
|
||||||
|
|
||||||
``link_in(other)``
|
|
||||||
~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Link in another module ``other`` into this module. Global variables,
|
|
||||||
functions etc. are matched and resolved. The ``other`` module is no
|
|
||||||
longer valid and should not be used after this operation. This API might
|
|
||||||
be replaced with a full-fledged Linker class in the future.
|
|
||||||
|
|
||||||
Special Methods
|
|
||||||
---------------
|
|
||||||
|
|
||||||
``__str__``
|
|
||||||
~~~~~~~~~~~
|
|
||||||
|
|
||||||
``Module`` objects can be stringified into it's LLVM assembly language
|
|
||||||
representation.
|
|
||||||
|
|
||||||
``__eq__``
|
|
||||||
~~~~~~~~~~
|
|
||||||
|
|
||||||
``Module`` objects can be compared for equality. Internally, this
|
|
||||||
converts both arguments into their LLVM assembly representations and
|
|
||||||
compares the resultant strings.
|
|
||||||
|
|
||||||
**Convention**
|
|
||||||
|
|
||||||
*All* llvmpy objects (where it makes sense), when stringified,
|
|
||||||
return the LLVM assembly representation. ``print module_obj`` for
|
|
||||||
example, prints the LLVM assembly form of the entire module.
|
|
||||||
|
|
||||||
Such objects, when compared for equality, internally compare these
|
|
||||||
string representations.
|
|
||||||
|
|
|
||||||
|
|
@ -1,84 +0,0 @@
|
||||||
+---------------------------------+
|
|
||||||
| layout: page |
|
|
||||||
+---------------------------------+
|
|
||||||
| title: StructType (llvm.core) |
|
|
||||||
+---------------------------------+
|
|
||||||
|
|
||||||
llvm.core.StructType
|
|
||||||
====================
|
|
||||||
|
|
||||||
Base Class
|
|
||||||
----------
|
|
||||||
|
|
||||||
- `llvm.core.Type <llvm.core.Type.html>`_
|
|
||||||
|
|
||||||
Methods
|
|
||||||
-------
|
|
||||||
|
|
||||||
``set_body(self, elems, packed=False)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Define the body for opaque identified structure.
|
|
||||||
|
|
||||||
``elems`` is an iterable of `llvm.core.Type <llvm.core.Type.html>`_ If
|
|
||||||
``packed`` is ``True``, creates a packed structure.
|
|
||||||
|
|
||||||
Properties
|
|
||||||
----------
|
|
||||||
|
|
||||||
``is_identified``
|
|
||||||
~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
``True`` if this is an identified structure.
|
|
||||||
|
|
||||||
``is_literal``
|
|
||||||
~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
``True`` if this is a literal structure.
|
|
||||||
|
|
||||||
``is_opaque``
|
|
||||||
~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
``True`` if this is an opaque structure. Only identified structure can
|
|
||||||
be opaque.
|
|
||||||
|
|
||||||
``packed``
|
|
||||||
~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
``True`` if the structure is packed (no padding between elements).
|
|
||||||
|
|
||||||
``name``
|
|
||||||
~~~~~~~~
|
|
||||||
|
|
||||||
Use in identified structure. If set to empty, the identified structure
|
|
||||||
is removed from the global context.
|
|
||||||
|
|
||||||
``elements``
|
|
||||||
~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
Returns an iterable object that yields `Type <llvm.core.Type.html>`_
|
|
||||||
objects that represent, in order, the types of the elements of the
|
|
||||||
structure. Used like this:
|
|
||||||
|
|
||||||
{% highlight python %} struct\_type = Type.struct( [ Type.int(),
|
|
||||||
Type.int() ] ) for elem in struct\_type.elements: assert elem.kind ==
|
|
||||||
TYPE\_INTEGER assert elem == Type.int() assert
|
|
||||||
struct\_type.element\_count == len(struct\_type.elements) {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
``element_count``
|
|
||||||
~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
The number of elements. Same as ``len(obj.elements)``, but faster.
|
|
||||||
|
|
@ -106,40 +106,23 @@ Properties
|
||||||
A value (enum) representing the "type" of the object. It will be one of
|
A value (enum) representing the "type" of the object. It will be one of
|
||||||
the following constants defined in ``llvm.core``:
|
the following constants defined in ``llvm.core``:
|
||||||
|
|
||||||
{% highlight python %} # Warning: do not rely on actual numerical
|
|
||||||
values! TYPE\_VOID = 0 TYPE\_FLOAT = 1 TYPE\_DOUBLE = 2 TYPE\_X86\_FP80
|
.. code-block:: python
|
||||||
= 3 TYPE\_FP128 = 4 TYPE\_PPC\_FP128 = 5 TYPE\_LABEL = 6 TYPE\_INTEGER =
|
|
||||||
7 TYPE\_FUNCTION = 8 TYPE\_STRUCT = 9 TYPE\_ARRAY = 10 TYPE\_POINTER =
|
# Warning: do not rely on actual numerical
|
||||||
11 TYPE\_OPAQUE = 12 TYPE\_VECTOR = 13 TYPE\_METADATA = 14 TYPE\_UNION =
|
values! TYPE_VOID = 0 TYPE_FLOAT = 1 TYPE_DOUBLE = 2 TYPE_X86_FP80
|
||||||
15 {% endhighlight %}
|
= 3 TYPE_FP128 = 4 TYPE_PPC_FP128 = 5 TYPE_LABEL = 6 TYPE_INTEGER =
|
||||||
|
7 TYPE_FUNCTION = 8 TYPE_STRUCT = 9 TYPE_ARRAY = 10 TYPE_POINTER =
|
||||||
|
11 TYPE_OPAQUE = 12 TYPE_VECTOR = 13 TYPE_METADATA = 14 TYPE_UNION =
|
||||||
|
15
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Example:
|
Example:
|
||||||
^^^^^^^^
|
^^^^^^^^
|
||||||
|
|
||||||
{% highlight python %} assert Type.int().kind == TYPE\_INTEGER assert
|
|
||||||
Type.void().kind == TYPE\_VOID {% endhighlight %}
|
|
||||||
|
|
||||||
Methods
|
.. code-block:: python
|
||||||
-------
|
|
||||||
|
|
||||||
``refine``
|
assert Type.int().kind == TYPE_INTEGER assert
|
||||||
~~~~~~~~~~
|
Type.void().kind == TYPE_VOID
|
||||||
|
|
||||||
Used for constructing self-referencing types. See the documentation of
|
|
||||||
`TypeHandle <llvm.core.TypeHandle.html>`_ objects.
|
|
||||||
|
|
||||||
Special Methods
|
|
||||||
---------------
|
|
||||||
|
|
||||||
``__str__``
|
|
||||||
~~~~~~~~~~~
|
|
||||||
|
|
||||||
``Type`` objects can be stringified into it's LLVM assembly language
|
|
||||||
representation.
|
|
||||||
|
|
||||||
``__eq__``
|
|
||||||
~~~~~~~~~~
|
|
||||||
|
|
||||||
``Type`` objects can be compared for equality. Internally, this converts
|
|
||||||
both arguments into their LLVM assembly representations and compares the
|
|
||||||
resultant strings.
|
|
||||||
|
|
|
||||||
|
|
@ -80,15 +80,8 @@ Pythonically, modules are imported with the statement
|
||||||
``import llvm.core``. However, you might find it more convenient to
|
``import llvm.core``. However, you might find it more convenient to
|
||||||
import llvmpy modules thus:
|
import llvmpy modules thus:
|
||||||
|
|
||||||
{% highlight python %} from llvm import \* from llvm.core import \* from
|
|
||||||
llvm.ee import \* from llvm.passes import \* {% endhighlight %}
|
|
||||||
|
|
||||||
This avoids quite some typing. Both conventions work, however.
|
.. code-block:: python
|
||||||
|
|
||||||
**Tip**
|
from llvm import \* from llvm.core import \* from
|
||||||
|
llvm.ee import \* from llvm.passes import \*
|
||||||
Python-style documentation strings (``__doc__``) are present in
|
|
||||||
llvmpy. You can use the ``help()`` of the interactive Python
|
|
||||||
interpreter or the ``object?`` of
|
|
||||||
`IPython <http://ipython.scipy.org/moin/>`_ to get online help.
|
|
||||||
(Note: not complete yet!)
|
|
||||||
|
|
|
||||||
|
|
@ -60,50 +60,43 @@ An Example
|
||||||
|
|
||||||
Here is an example that demonstrates the creation of types:
|
Here is an example that demonstrates the creation of types:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
|
||||||
integers
|
.. code-block:: python
|
||||||
========
|
|
||||||
|
|
||||||
int\_ty = Type.int() bool\_ty = Type.int(1) int\_64bit = Type.int(64)
|
#!/usr/bin/env python
|
||||||
|
|
||||||
floats
|
# integers
|
||||||
======
|
int_ty = Type.int() bool_ty = Type.int(1) int_64bit = Type.int(64)
|
||||||
|
|
||||||
sprec\_real = Type.float() dprec\_real = Type.double()
|
# floats
|
||||||
|
sprec_real = Type.float() dprec_real = Type.double()
|
||||||
|
|
||||||
arrays and vectors
|
# arrays and vectors
|
||||||
==================
|
intar_ty = Type.array( int_ty, 10 ) # "typedef int intar_ty[10];"
|
||||||
|
twodim = Type.array( intar_ty , 10 ) # "typedef int twodim[10][10];"
|
||||||
|
vec = Type.array( int_ty, 10 )
|
||||||
|
|
||||||
intar\_ty = Type.array( int\_ty, 10 ) # "typedef int intar\_ty[10];"
|
# structures
|
||||||
twodim = Type.array( intar\_ty , 10 ) # "typedef int twodim[10][10];"
|
s1_ty = Type.struct( [ int_ty, sprec_real ] ) # "struct s1_ty { int
|
||||||
vec = Type.array( int\_ty, 10 )
|
v1; float v2; };"
|
||||||
|
|
||||||
structures
|
# pointers
|
||||||
==========
|
intptr_ty = Type.pointer(int_ty) # "typedef int \*intptr_ty;"
|
||||||
|
|
||||||
s1\_ty = Type.struct( [ int\_ty, sprec\_real ] ) # "struct s1\_ty { int
|
# functions
|
||||||
v1; float v2; };"
|
f1 = Type.function( int_ty, [ int_ty ] ) # functions that take 1
|
||||||
|
int_ty and return 1 int_ty
|
||||||
|
|
||||||
pointers
|
f2 = Type.function( Type.void(), [ int_ty, int_ty ] ) # functions that
|
||||||
========
|
take 2 int_tys and return nothing
|
||||||
|
|
||||||
intptr\_ty = Type.pointer(int\_ty) # "typedef int \*intptr\_ty;"
|
f3 = Type.function( Type.void(), ( int_ty, int_ty ) ) # same as f2;
|
||||||
|
any iterable can be used
|
||||||
|
|
||||||
functions
|
fnargs = [ Type.pointer( Type.int(8) ) ] printf = Type.function(
|
||||||
=========
|
Type.int(), fnargs, True ) # variadic function
|
||||||
|
|
||||||
f1 = Type.function( int\_ty, [ int\_ty ] ) # functions that take 1
|
|
||||||
int\_ty and return 1 int\_ty
|
|
||||||
|
|
||||||
f2 = Type.function( Type.void(), [ int\_ty, int\_ty ] ) # functions that
|
|
||||||
take 2 int\_tys and return nothing
|
|
||||||
|
|
||||||
f3 = Type.function( Type.void(), ( int\_ty, int\_ty ) ) # same as f2;
|
|
||||||
any iterable can be used
|
|
||||||
|
|
||||||
fnargs = [ Type.pointer( Type.int(8) ) ] printf = Type.function(
|
|
||||||
Type.int(), fnargs, True ) # variadic function {% endhighlight %}
|
|
||||||
|
|
||||||
--------------
|
--------------
|
||||||
|
|
||||||
|
|
@ -123,16 +116,8 @@ The following code defines a opaque structure, named "mystruct". The
|
||||||
body is defined after the construction using ``StructType.set_body``.
|
body is defined after the construction using ``StructType.set_body``.
|
||||||
The second subtype is a pointer to a "mystruct" type.
|
The second subtype is a pointer to a "mystruct" type.
|
||||||
|
|
||||||
{% highlight python %} ts = Type.opaque('mystruct')
|
|
||||||
ts.set\_body([Type.int(), Type.pointer(ts)]) {% endhighlight %}
|
|
||||||
|
|
||||||
--------------
|
.. code-block:: python
|
||||||
|
|
||||||
**Related Links** `llvm.core.Type <llvm.core.Type.html>`_,
|
ts = Type.opaque('mystruct')
|
||||||
`llvm.core.IntegerType <llvm.core.IntegerType.html>`_,
|
ts.set_body([Type.int(), Type.pointer(ts)])
|
||||||
`llvm.core.FunctionType <llvm.core.FunctionType.html>`_,
|
|
||||||
`llvm.core.StructType <llvm.core.StructType.html>`_,
|
|
||||||
`llvm.core.ArrayType <llvm.core.ArrayType.html>`_,
|
|
||||||
`llvm.core.PointerType <llvm.core.PointerType.html>`_,
|
|
||||||
`llvm.core.VectorType <llvm.core.VectorType.html>`_,
|
|
||||||
`llvm.core.TypeHandle <llvm.core.TypeHandle.html>`_
|
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue