Fix up code-highlighting sections.
This commit is contained in:
parent
415c01f745
commit
ce8884aa33
18 changed files with 5611 additions and 6037 deletions
|
|
@ -13,34 +13,29 @@ References to functions already present in a module can be retrieved via
|
||||||
``Function.get``. All functions in a module can be enumerated by
|
``Function.get``. All functions in a module can be enumerated by
|
||||||
iterating over ``module_obj.functions``.
|
iterating over ``module_obj.functions``.
|
||||||
|
|
||||||
{% highlight python %} # create a type, representing functions that take
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# create a type, representing functions that take
|
||||||
an integer and return # a floating point value. ft = Type.function(
|
an integer and return # a floating point value. ft = Type.function(
|
||||||
Type.float(), [ Type.int() ] )
|
Type.float(), [ Type.int() ] )
|
||||||
|
|
||||||
create a function of this type
|
# create a function of this type
|
||||||
==============================
|
f1 = module_obj.add_function(ft, "func1")
|
||||||
|
|
||||||
f1 = module\_obj.add\_function(ft, "func1")
|
# or equivalently, like this:
|
||||||
|
f2 = Function.new(module_obj, ft, "func2")
|
||||||
|
|
||||||
or equivalently, like this:
|
# get a reference to an existing function
|
||||||
===========================
|
f3 = module_obj.get_function_named("func3")
|
||||||
|
|
||||||
f2 = Function.new(module\_obj, ft, "func2")
|
# or like this:
|
||||||
|
f4 = Function.get(module_obj, "func4")
|
||||||
|
|
||||||
get a reference to an existing function
|
# list all function names in a module
|
||||||
=======================================
|
for f in module_obj.functions: print f.name
|
||||||
|
|
||||||
f3 = module\_obj.get\_function\_named("func3")
|
|
||||||
|
|
||||||
or like this:
|
|
||||||
=============
|
|
||||||
|
|
||||||
f4 = Function.get(module\_obj, "func4")
|
|
||||||
|
|
||||||
list all function names in a module
|
|
||||||
===================================
|
|
||||||
|
|
||||||
for f in module\_obj.functions: print f.name {% endhighlight %}
|
|
||||||
|
|
||||||
Intrinsic
|
Intrinsic
|
||||||
=========
|
=========
|
||||||
|
|
@ -52,13 +47,16 @@ called with a module object, an intrinsic ID (which is a numeric
|
||||||
constant) and a list of the types of arguments (which LLVM uses to
|
constant) and a list of the types of arguments (which LLVM uses to
|
||||||
resolve overloaded intrinsic functions).
|
resolve overloaded intrinsic functions).
|
||||||
|
|
||||||
{% highlight python %} # get a reference to the llvm.bswap intrinsic
|
|
||||||
bswap = Function.intrinsic(mod, INTR\_BSWAP, [Type.int()])
|
|
||||||
|
|
||||||
call it
|
.. code-block:: python
|
||||||
=======
|
|
||||||
|
# get a reference to the llvm.bswap intrinsic
|
||||||
|
bswap = Function.intrinsic(mod, INTR_BSWAP, [Type.int()])
|
||||||
|
|
||||||
|
# call it
|
||||||
|
builder.call(bswap, [value])
|
||||||
|
|
||||||
|
|
||||||
builder.call(bswap, [value]) {% endhighlight %}
|
|
||||||
|
|
||||||
Here, the constant ``INTR_BSWAP``, available from ``llvm.core``,
|
Here, the constant ``INTR_BSWAP``, available from ``llvm.core``,
|
||||||
represents the LLVM intrinsic
|
represents the LLVM intrinsic
|
||||||
|
|
@ -111,13 +109,16 @@ The value objects corresponding to the arguments of a function can be
|
||||||
got using the read-only property ``args``. These can be iterated over,
|
got using the read-only property ``args``. These can be iterated over,
|
||||||
and also be indexed via integers. An example:
|
and also be indexed via integers. An example:
|
||||||
|
|
||||||
{% highlight python %} # list all argument names and types for arg in
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# list all argument names and types for arg in
|
||||||
fn.args: print arg.name, "of type", arg.type
|
fn.args: print arg.name, "of type", arg.type
|
||||||
|
|
||||||
change the name of the first argument
|
# change the name of the first argument
|
||||||
=====================================
|
fn.args[0].name = "objptr"
|
||||||
|
|
||||||
|
|
||||||
fn.args[0].name = "objptr" {% endhighlight %}
|
|
||||||
|
|
||||||
Basic blocks (see later) are contained within functions. When newly
|
Basic blocks (see later) are contained within functions. When newly
|
||||||
created, a function has no basic blocks. They have to be added
|
created, a function has no basic blocks. They have to be added
|
||||||
|
|
@ -130,71 +131,19 @@ blocks can be got via ``basic_block_count`` method. Note that
|
||||||
``get_entry_basic_block`` is slightly faster than ``basic_blocks[0]``
|
``get_entry_basic_block`` is slightly faster than ``basic_blocks[0]``
|
||||||
and so is ``basic_block_count``, over ``len(f.basic_blocks)``.
|
and so is ``basic_block_count``, over ``len(f.basic_blocks)``.
|
||||||
|
|
||||||
{% highlight python %} # add a basic block b1 =
|
|
||||||
fn.append\_basic\_block("entry")
|
|
||||||
|
|
||||||
get the first one
|
.. code-block:: python
|
||||||
=================
|
|
||||||
|
|
||||||
b2 = fn.get\_entry\_basic\_block() b2 = fn.basic\_mdblocks[0] # slower
|
# add a basic block b1 =
|
||||||
|
fn.append_basic_block("entry")
|
||||||
|
|
||||||
|
# get the first one
|
||||||
|
b2 = fn.get_entry_basic_block() b2 = fn.basic_mdblocks[0] # slower
|
||||||
than previous method
|
than previous method
|
||||||
|
|
||||||
print names of all basic blocks
|
# print names of all basic blocks
|
||||||
===============================
|
for b in fn.basic_blocks: print b.name
|
||||||
|
|
||||||
for b in fn.basic\_blocks: print b.name
|
# get number of basic blocks
|
||||||
|
n = fn.basic_block_count n = len(fn.basic_blocks) # slower than
|
||||||
get number of basic blocks
|
previous method
|
||||||
==========================
|
|
||||||
|
|
||||||
n = fn.basic\_block\_count n = len(fn.basic\_blocks) # slower than
|
|
||||||
previous method {% endhighlight %}
|
|
||||||
|
|
||||||
Functions can be deleted using the method ``delete``. This deletes them
|
|
||||||
from their containing module. All references to the function object
|
|
||||||
should be dropped after ``delete`` has been called.
|
|
||||||
|
|
||||||
Functions can be verified with the ``verify`` method. Note that this may
|
|
||||||
not work properly (aborts on errors).
|
|
||||||
|
|
||||||
Function Attributes # {#fnattr}
|
|
||||||
===============================
|
|
||||||
|
|
||||||
Function attributes, as documented
|
|
||||||
`here <http://www.llvm.org/docs/LangRef.html#fnattrs>`_, can be set on
|
|
||||||
functions using the methods ``add_attribute`` and ``remove_attribute``.
|
|
||||||
The following values may be used to refer to the LLVM attributes:
|
|
||||||
|
|
||||||
Value \| Equivalent LLVM Assembly Keyword \|
|
|
||||||
------\|----------------------------------\|
|
|
||||||
``ATTR_ALWAYS_INLINE``\ \|\ ``alwaysinline`` \|
|
|
||||||
``ATTR_INLINE_HINT``\ \|\ ``inlinehint`` \|
|
|
||||||
``ATTR_NO_INLINE``\ \|\ ``noinline`` \|
|
|
||||||
``ATTR_OPTIMIZE_FOR_SIZE``\ \|\ ``optsize`` \|
|
|
||||||
``ATTR_NO_RETURN``\ \|\ ``noreturn`` \|
|
|
||||||
``ATTR_NO_UNWIND``\ \|\ ``nounwind`` \|
|
|
||||||
``ATTR_READ_NONE``\ \|\ ``readnone`` \|
|
|
||||||
``ATTR_READONLY``\ \|\ ``readonly`` \|
|
|
||||||
``ATTR_STACK_PROTECT``\ \|\ ``ssp`` \|
|
|
||||||
``ATTR_STACK_PROTECT_REQ``\ \|\ ``sspreq`` \|
|
|
||||||
``ATTR_NO_REDZONE``\ \|\ ``noredzone`` \|
|
|
||||||
``ATTR_NO_IMPLICIT_FLOAT``\ \|\ ``noimplicitfloat`` \|
|
|
||||||
``ATTR_NAKED``\ \|\ ``naked`` \|
|
|
||||||
|
|
||||||
Here is how attributes can be set and removed:
|
|
||||||
|
|
||||||
{% highlight python %} # create a function ti = Type.int(32) tf =
|
|
||||||
Type.function(ti, [ti, ti]) m = Module.new('mod') f =
|
|
||||||
m.add\_function(tf, 'sum') print f # declare i32 @sum(i32, i32)
|
|
||||||
|
|
||||||
add a couple of attributes
|
|
||||||
==========================
|
|
||||||
|
|
||||||
f.add\_attribute(ATTR\_NO\_UNWIND) f.add\_attribute(ATTR\_READONLY)
|
|
||||||
print f # declare i32 @sum(i32, i32) nounwind readonly {% endhighlight
|
|
||||||
%}
|
|
||||||
|
|
||||||
**Related Links**
|
|
||||||
|
|
||||||
`llvm.core.Function <llvm.core.Function.html>`_,
|
|
||||||
`llvm.core.Argument <llvm.core.Argument.html>`_
|
|
||||||
|
|
|
||||||
|
|
@ -78,8 +78,9 @@ object files be built with the ``-fPIC`` option (generate position
|
||||||
independent code). Be sure to use the ``--enable-pic`` option while
|
independent code). Be sure to use the ``--enable-pic`` option while
|
||||||
configuring LLVM (default is no PIC), like this:
|
configuring LLVM (default is no PIC), like this:
|
||||||
|
|
||||||
{% highlight bash %} ~/llvm$ ./configure --enable-pic --enable-optimized
|
.. code-block:: bash
|
||||||
{% endhighlight %}
|
|
||||||
|
$ ~/llvm ./configure --enable-pic --enable-optimized
|
||||||
|
|
||||||
llvm-config
|
llvm-config
|
||||||
-----------
|
-----------
|
||||||
|
|
@ -103,51 +104,8 @@ LLVM's 'configure'.
|
||||||
|
|
||||||
Get llvmpy and install it:
|
Get llvmpy and install it:
|
||||||
|
|
||||||
{% highlight bash %} $ git clone git@github.com:numba/llvmpy.git $ cd
|
|
||||||
llvmpy $ python setup.py install {% endhighlight %}
|
|
||||||
|
|
||||||
If you need to tell the build script where ``llvm-config`` is, do it
|
.. code-block:: bash
|
||||||
this way:
|
|
||||||
|
|
||||||
{% highlight bash %} $ python setup.py install --user
|
$ git clone git@github.com:numba/llvmpy.git $ cd
|
||||||
--llvm-config=/home/mdevan/llvm/Release/bin/llvm-config {% endhighlight
|
llvmpy $ python setup.py install
|
||||||
%}
|
|
||||||
|
|
||||||
To build a debug version of llvmpy, that links against the debug
|
|
||||||
libraries of LLVM, use this:
|
|
||||||
|
|
||||||
{% highlight bash %} $ python setup.py build -g
|
|
||||||
--llvm-config=/home/mdevan/llvm/Debug/bin/llvm-config $ python setup.py
|
|
||||||
install --user --llvm-config=/home/mdevan/llvm/Debug/bin/llvm-config {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
Be warned that debug binaries will be huge (100MB+) ! They are required
|
|
||||||
only if you need to debug into LLVM also.
|
|
||||||
|
|
||||||
``setup.py`` is a standard Python distutils script. See the Python
|
|
||||||
documentation regarding `Installing Python
|
|
||||||
Modules <http://docs.python.org/inst/inst.html>`_ and `Distributing
|
|
||||||
Python Modules <http://docs.python.org/dist/dist.html>`_ for more
|
|
||||||
information on such scripts.
|
|
||||||
|
|
||||||
|
|
||||||
Uninstall
|
|
||||||
==============
|
|
||||||
|
|
||||||
If you'd installed llvmpy with the ``--user`` option, then llvmpy
|
|
||||||
would be present under ``~/.local/lib/python2.7/site-packages``.
|
|
||||||
Otherwise, it might be under ``/usr/lib/python2.7/site-packages`` or
|
|
||||||
``/usr/local/lib/python2.7/site-packages``. The directory would vary
|
|
||||||
with your Python version and OS flavour. Look around.
|
|
||||||
|
|
||||||
Once you've located the site-packages directory, the modules and the
|
|
||||||
"egg" can be removed like so:
|
|
||||||
|
|
||||||
{% highlight bash %} $ rm -rf /llvm /llvm\_py-.egg-info {% endhighlight
|
|
||||||
%}
|
|
||||||
|
|
||||||
See the `Python
|
|
||||||
documentation <http://docs.python.org/install/index.html>`_ for more
|
|
||||||
information.
|
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
|
||||||
|
|
@ -112,23 +112,36 @@ This gives the language a very nice and simple syntax. For example, the
|
||||||
following simple example computes `Fibonacci
|
following simple example computes `Fibonacci
|
||||||
numbers <http://en.wikipedia.org/wiki/Fibonacci_number>`_:
|
numbers <http://en.wikipedia.org/wiki/Fibonacci_number>`_:
|
||||||
|
|
||||||
{% highlight python %} # Compute the x'th fibonacci number. def fib(x)
|
|
||||||
if x < 3 then 1 else fib(x-1)+fib(x-2)
|
|
||||||
|
|
||||||
This expression will compute the 40th number.
|
.. code-block::
|
||||||
=============================================
|
|
||||||
|
# Compute the x'th fibonacci number.
|
||||||
|
def fib(x):
|
||||||
|
if x < 3:
|
||||||
|
return 1
|
||||||
|
else:
|
||||||
|
return fib(x-1)+fib(x-2)
|
||||||
|
|
||||||
|
# This expression will compute the 40th number.
|
||||||
|
fib(40)
|
||||||
|
|
||||||
|
|
||||||
fib(40) {% endhighlight %}
|
|
||||||
|
|
||||||
We also allow Kaleidoscope to call into standard library functions (the
|
We also allow Kaleidoscope to call into standard library functions (the
|
||||||
LLVM JIT makes this completely trivial). This means that you can use the
|
LLVM JIT makes this completely trivial). This means that you can use the
|
||||||
'extern' keyword to define a function before you use it (this is also
|
'extern' keyword to define a function before you use it (this is also
|
||||||
useful for mutually recursive functions). For example:
|
useful for mutually recursive functions). For example:
|
||||||
|
|
||||||
{% highlight python %} extern sin(arg); extern cos(arg); extern
|
|
||||||
atan2(arg1 arg2);
|
|
||||||
|
|
||||||
atan2(sin(0.4), cos(42)) {% endhighlight %}
|
.. code-block::
|
||||||
|
|
||||||
|
extern sin(arg);
|
||||||
|
extern cos(arg);
|
||||||
|
extern atan2(arg1 arg2);
|
||||||
|
|
||||||
|
atan2(sin(0.4), cos(42))
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
A more interesting example is included in Chapter 6 where we write a
|
A more interesting example is included in Chapter 6 where we write a
|
||||||
little Kaleidoscope application that
|
little Kaleidoscope application that
|
||||||
|
|
@ -150,23 +163,32 @@ traditional way to do this is to use a
|
||||||
the lexer includes a token type and potentially some metadata (e.g. the
|
the lexer includes a token type and potentially some metadata (e.g. the
|
||||||
numeric value of a number). First, we define the possibilities:
|
numeric value of a number). First, we define the possibilities:
|
||||||
|
|
||||||
{% highlight python %} # The lexer yields one of these types for each
|
|
||||||
token. class EOFToken(object): pass
|
.. code-block:: python
|
||||||
|
|
||||||
|
# The lexer yields one of these types for each token.
|
||||||
|
class EOFToken(object): pass
|
||||||
|
|
||||||
class DefToken(object): pass
|
class DefToken(object): pass
|
||||||
|
|
||||||
class ExternToken(object): pass
|
class ExternToken(object): pass
|
||||||
|
|
||||||
class IdentifierToken(object): def **init**\ (self, name): self.name =
|
class IdentifierToken(object):
|
||||||
name
|
def __init__(self, name):
|
||||||
|
self.name = name
|
||||||
|
|
||||||
class NumberToken(object): def **init**\ (self, value): self.value =
|
class NumberToken(object):
|
||||||
value
|
def __init__(self, value):
|
||||||
|
self.value = value
|
||||||
|
|
||||||
|
class CharacterToken(object):
|
||||||
|
def __init__(self, char):
|
||||||
|
self.char = char
|
||||||
|
def __eq__(self, other):
|
||||||
|
return isinstance(other, CharacterToken) and self.char == other.char
|
||||||
|
def __ne__(self, other):
|
||||||
|
return not self == other
|
||||||
|
|
||||||
class CharacterToken(object): def **init**\ (self, char): self.char =
|
|
||||||
char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
|
||||||
== other {% endhighlight %}
|
|
||||||
|
|
||||||
Each token yielded by our lexer will be of one of the above types. For
|
Each token yielded by our lexer will be of one of the above types. For
|
||||||
simple tokens that are always the same, like the "def" keyword, the
|
simple tokens that are always the same, like the "def" keyword, the
|
||||||
|
|
@ -193,82 +215,109 @@ digits. Identifiers (and keywords) are alphanumeric string starting with
|
||||||
a letter and comments are anything between a hash (``#``) and the end of
|
a letter and comments are anything between a hash (``#``) and the end of
|
||||||
the line.
|
the line.
|
||||||
|
|
||||||
{% highlight python %} import re
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
...
|
...
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
=============================================================
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?')
|
||||||
|
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]\ *')
|
||||||
|
REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
Next, let's start defining the ``Tokenize`` function itself. The first
|
Next, let's start defining the ``Tokenize`` function itself. The first
|
||||||
thing we need to do is set up a loop that scans the string, while
|
thing we need to do is set up a loop that scans the string, while
|
||||||
ignoring whitespace between tokens:
|
ignoring whitespace between tokens:
|
||||||
|
|
||||||
{% highlight python %} def Tokenize(string): while string: # Skip
|
|
||||||
whitespace. if string[0].isspace(): string = string[1:] continue
|
.. code-block:: python
|
||||||
|
|
||||||
|
def Tokenize(string):
|
||||||
|
while string: # Skip whitespace.
|
||||||
|
if string[0].isspace():
|
||||||
|
string = string[1:]
|
||||||
|
continue
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
...
|
...
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Next we want to find out what the next token is. For this we run the
|
Next we want to find out what the next token is. For this we run the
|
||||||
regexes we defined above on the remainder of the string. To simplify the
|
regexes we defined above on the remainder of the string. To simplify the
|
||||||
rest of the code, we run all three regexes each time. As mentioned
|
rest of the code, we run all three regexes each time. As mentioned
|
||||||
above, inefficiencies are ignored for the purpose of this tutorial:
|
above, inefficiencies are ignored for the purpose of this tutorial:
|
||||||
|
|
||||||
{% highlight python %} # Run regexes. comment\_match =
|
|
||||||
REGEX\_COMMENT.match(string) number\_match = REGEX\_NUMBER.match(string)
|
|
||||||
identifier\_match = REGEX\_IDENTIFIER.match(string) {% endhighlight %}
|
|
||||||
|
|
||||||
Now se check if any of the regexes matched. For comments, we simply
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Run regexes.
|
||||||
|
comment_match = REGEX_COMMENT.match(string)
|
||||||
|
number_match = REGEX_NUMBER.match(string)
|
||||||
|
identifier_match = REGEX_IDENTIFIER.match(string)
|
||||||
|
|
||||||
|
|
||||||
|
Now we check if any of the regexes matched. For comments, we simply
|
||||||
ignore the captured match:
|
ignore the captured match:
|
||||||
|
|
||||||
{% highlight python %} # Check if any of the regexes matched and yield
|
|
||||||
the appropriate result. if comment\_match: comment =
|
.. code-block:: python
|
||||||
comment\_match.group(0) string = string[len(comment):] {% endhighlight
|
|
||||||
python %}
|
# Check if any of the regexes matched and yield
|
||||||
|
# the appropriate result.
|
||||||
|
if comment_match:
|
||||||
|
comment = comment_match.group(0)
|
||||||
|
string = string[len(comment):]
|
||||||
|
|
||||||
For numbers, we yield the captured match, converted to a float and
|
For numbers, we yield the captured match, converted to a float and
|
||||||
tagged with the appropriate token type:
|
tagged with the appropriate token type:
|
||||||
|
|
||||||
{% highlight python %} elif number\_match: number =
|
.. code-block:: python
|
||||||
number\_match.group(0) yield NumberToken(float(number)) string =
|
|
||||||
string[len(number):] {% endhighlight %}
|
elif number_match:
|
||||||
|
number = number_match.group(0)
|
||||||
|
yield NumberToken(float(number))
|
||||||
|
string = string[len(number):]
|
||||||
|
|
||||||
The identifier case is a little more complex. We have to check for
|
The identifier case is a little more complex. We have to check for
|
||||||
keywords to decide whether we have captured an identifier or a keyword:
|
keywords to decide whether we have captured an identifier or a keyword:
|
||||||
|
|
||||||
{% highlight python %} elif identifier\_match: identifier =
|
.. code-block:: python
|
||||||
identifier\_match.group(0) # Check if we matched a keyword. if
|
|
||||||
identifier == 'def': yield DefToken() elif identifier == 'extern': yield
|
elif identifier_match:
|
||||||
ExternToken() else: yield IdentifierToken(identifier) string =
|
identifier = identifier_match.group(0)
|
||||||
string[len(identifier):] {% endhighlight %}
|
# Check if we matched a keyword.
|
||||||
|
if identifier == 'def':
|
||||||
|
yield DefToken()
|
||||||
|
elif identifier == 'extern':
|
||||||
|
yield ExternToken()
|
||||||
|
else:
|
||||||
|
yield IdentifierToken(identifier)
|
||||||
|
string = string[len(identifier):]
|
||||||
|
|
||||||
|
|
||||||
Finally, if we haven't recognized a comment, a number of an identifier,
|
Finally, if we haven't recognized a comment, a number of an identifier,
|
||||||
we yield the current character as an "unknown character" token. This is
|
we yield the current character as an "unknown character" token. This is
|
||||||
used, for example, for operators like ``+`` or ``*``:
|
used, for example, for operators like ``+`` or ``*``:
|
||||||
|
|
||||||
{% highlight python %} else: # Yield the unknown character. yield
|
|
||||||
CharacterToken(string[0]) string = string[1:] {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
else: # Yield the unknown character.
|
||||||
|
yield CharacterToken(string[0])
|
||||||
|
string = string[1:]
|
||||||
|
|
||||||
|
|
||||||
Once we're done with the loop, we return a final end-of-file token:
|
Once we're done with the loop, we return a final end-of-file token:
|
||||||
|
|
||||||
{% highlight python %} yield EOFToken() {% endhighlight %}
|
|
||||||
|
|
||||||
With this, we have the complete lexer for the basic Kaleidoscope
|
.. code-block:: python
|
||||||
language (the `full code listing <PythonLangImpl2.html#code>`_ for the
|
|
||||||
Lexer is available in the `next chapter <PythonLangImpl2.html>`_ of the
|
|
||||||
tutorial). Next we'll `build a simple parser that uses this to build an
|
|
||||||
Abstract Syntax Tree <PythonLangImpl2.html>`_. When we have that, we'll
|
|
||||||
include a driver so that you can use the lexer and parser together.
|
|
||||||
|
|
||||||
--------------
|
yield EOFToken()
|
||||||
|
|
||||||
**`Next: Implementing a Parser and AST <PythonLangImpl2.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -36,16 +36,19 @@ language, and the AST should closely model the language. In
|
||||||
Kaleidoscope, we have expressions, a prototype, and a function object.
|
Kaleidoscope, we have expressions, a prototype, and a function object.
|
||||||
We'll start with expressions first:
|
We'll start with expressions first:
|
||||||
|
|
||||||
{% highlight python %} # Base class for all expression nodes. class
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Base class for all expression nodes. class
|
||||||
ExpressionNode(object): pass
|
ExpressionNode(object): pass
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
# Expression class for numeric literals like "1.0".
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode): def **init**\ (self, value):
|
class NumberExpressionNode(ExpressionNode): def **init**\ (self, value):
|
||||||
self.value = value
|
self.value = value
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The code above shows the definition of the base ExpressionNode class and
|
The code above shows the definition of the base ExpressionNode class and
|
||||||
one subclass which we use for numeric literals. The important thing to
|
one subclass which we use for numeric literals. The important thing to
|
||||||
|
|
@ -58,22 +61,23 @@ them. It would be very easy to add a virtual method to pretty print the
|
||||||
code, for example. Here are the other expression AST node definitions
|
code, for example. Here are the other expression AST node definitions
|
||||||
that we'll use in the basic form of the Kaleidoscope language:
|
that we'll use in the basic form of the Kaleidoscope language:
|
||||||
|
|
||||||
{% highlight python %} # Expression class for referencing a variable,
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Expression class for referencing a variable,
|
||||||
like "a". class VariableExpressionNode(ExpressionNode): def
|
like "a". class VariableExpressionNode(ExpressionNode): def
|
||||||
**init**\ (self, name): self.name = name
|
**init**\ (self, name): self.name = name
|
||||||
|
|
||||||
Expression class for a binary operator.
|
# Expression class for a binary operator.
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode): def **init**\ (self,
|
class BinaryOperatorExpressionNode(ExpressionNode): def **init**\ (self,
|
||||||
operator, left, right): self.operator = operator self.left = left
|
operator, left, right): self.operator = operator self.left = left
|
||||||
self.right = right
|
self.right = right
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode): def **init**\ (self, callee,
|
class CallExpressionNode(ExpressionNode): def **init**\ (self, callee,
|
||||||
args): self.callee = callee self.args = args {% endhighlight %}
|
args): self.callee = callee self.args = args
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This is all (intentionally) rather straight-forward: variables capture
|
This is all (intentionally) rather straight-forward: variables capture
|
||||||
the variable name, binary operators capture their opcode (e.g. '+'), and
|
the variable name, binary operators capture their opcode (e.g. '+'), and
|
||||||
|
|
@ -89,17 +93,20 @@ Turing-complete; we'll fix that in a later installment. The two things
|
||||||
we need next are a way to talk about the interface to a function, and a
|
we need next are a way to talk about the interface to a function, and a
|
||||||
way to talk about functions themselves:
|
way to talk about functions themselves:
|
||||||
|
|
||||||
{% highlight python %} # This class represents the "prototype" for a
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# This class represents the "prototype" for a
|
||||||
function, which captures its name, # and its argument names (thus
|
function, which captures its name, # and its argument names (thus
|
||||||
implicitly the number of arguments the function # takes). class
|
implicitly the number of arguments the function # takes). class
|
||||||
PrototypeNode(object): def **init**\ (self, name, args): self.name =
|
PrototypeNode(object): def **init**\ (self, name, args): self.name =
|
||||||
name self.args = args
|
name self.args = args
|
||||||
|
|
||||||
This class represents a function definition itself.
|
# This class represents a function definition itself.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class FunctionNode(object): def **init**\ (self, prototype, body):
|
class FunctionNode(object): def **init**\ (self, prototype, body):
|
||||||
self.prototype = prototype self.body = body {% endhighlight %}
|
self.prototype = prototype self.body = body
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In Kaleidoscope, functions are typed with just a count of their
|
In Kaleidoscope, functions are typed with just a count of their
|
||||||
arguments. Since all values are double precision floating point, the
|
arguments. Since all values are double precision floating point, the
|
||||||
|
|
@ -120,22 +127,32 @@ build it. The idea here is that we want to parse something like
|
||||||
``x + y`` (which is returned as three tokens by the lexer) into an AST
|
``x + y`` (which is returned as three tokens by the lexer) into an AST
|
||||||
that could be generated with calls like this:
|
that could be generated with calls like this:
|
||||||
|
|
||||||
{% highlight python %} x = VariableExpressionNode('x') y =
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
x = VariableExpressionNode('x') y =
|
||||||
VariableExpressionNode('y') result = BinaryOperatorExpressionNode('+',
|
VariableExpressionNode('y') result = BinaryOperatorExpressionNode('+',
|
||||||
x, y) {% endhighlight %}
|
x, y)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In order to do this, we'll start by defining a lightweight ``Parser``
|
In order to do this, we'll start by defining a lightweight ``Parser``
|
||||||
class with some basic helper routines:
|
class with some basic helper routines:
|
||||||
|
|
||||||
{% highlight python %} class Parser(object):
|
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
.. code-block:: python
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
|
||||||
|
class Parser(object):
|
||||||
|
|
||||||
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
and # updates Parser.current with its results. def Next(self):
|
and # updates Parser.current with its results. def Next(self):
|
||||||
self.current = self.tokens.next() {% endhighlight %}
|
self.current = self.tokens.next()
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This implements a simple token buffer around the lexer. This allows us
|
This implements a simple token buffer around the lexer. This allows us
|
||||||
to look one token ahead at what the lexer is returning. Every function
|
to look one token ahead at what the lexer is returning. Every function
|
||||||
|
|
@ -157,9 +174,14 @@ We start with numeric literals, because they are the simplest to
|
||||||
process. For each production in our grammar, we'll define a function
|
process. For each production in our grammar, we'll define a function
|
||||||
which parses that production. For numeric literals, we have:
|
which parses that production. For numeric literals, we have:
|
||||||
|
|
||||||
{% highlight python %} # numberexpr ::= number def
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# numberexpr ::= number def
|
||||||
ParseNumberExpr(self): result = NumberExpressionNode(self.current.value)
|
ParseNumberExpr(self): result = NumberExpressionNode(self.current.value)
|
||||||
self.Next() # consume the number. return result {% endhighlight %}
|
self.Next() # consume the number. return result
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This method is very simple: it expects to be called when the current
|
This method is very simple: it expects to be called when the current
|
||||||
token is a ``NumberToken``. It takes the current number value, creates a
|
token is a ``NumberToken``. It takes the current number value, creates a
|
||||||
|
|
@ -173,7 +195,10 @@ not part of the grammar production) ready to go. This is a fairly
|
||||||
standard way to go for recursive descent parsers. For a better example,
|
standard way to go for recursive descent parsers. For a better example,
|
||||||
the parenthesis operator is defined like this:
|
the parenthesis operator is defined like this:
|
||||||
|
|
||||||
{% highlight python %} # parenexpr ::= '(' expression ')' def
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# parenexpr ::= '(' expression ')' def
|
||||||
ParseParenExpr(self): self.Next() # eat '('.
|
ParseParenExpr(self): self.Next() # eat '('.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -186,7 +211,9 @@ ParseParenExpr(self): self.Next() # eat '('.
|
||||||
|
|
||||||
return contents
|
return contents
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This function illustrates an interesting aspect of the parser. The
|
This function illustrates an interesting aspect of the parser. The
|
||||||
function uses recursion by calling ``ParseExpression`` (we will soon see
|
function uses recursion by calling ``ParseExpression`` (we will soon see
|
||||||
|
|
@ -201,8 +228,11 @@ needed.
|
||||||
The next simple production is for handling variable references and
|
The next simple production is for handling variable references and
|
||||||
function calls:
|
function calls:
|
||||||
|
|
||||||
{% highlight python %} # identifierexpr ::= identifier \| identifier '('
|
|
||||||
expression\* ')' def ParseIdentifierExpr(self): identifier\_name =
|
.. code-block:: python
|
||||||
|
|
||||||
|
# identifierexpr ::= identifier \| identifier '('
|
||||||
|
expression\* ')' def ParseIdentifierExpr(self): identifier_name =
|
||||||
self.current.name self.Next() # eat identifier.
|
self.current.name self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -225,7 +255,9 @@ self.current.name self.Next() # eat identifier.
|
||||||
self.Next() # eat ')'.
|
self.Next() # eat ')'.
|
||||||
return CallExpressionNode(identifier_name, args)
|
return CallExpressionNode(identifier_name, args)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This routine follows the same style as the other routines. It expects to
|
This routine follows the same style as the other routines. It expects to
|
||||||
be called if the current token is an ``IdentifierToken``. It also has
|
be called if the current token is an ``IdentifierToken``. It also has
|
||||||
|
|
@ -243,13 +275,18 @@ that will become more clear `later in the
|
||||||
tutorial <PythonLangImpl6.html#unary>`_. In order to parse an arbitrary
|
tutorial <PythonLangImpl6.html#unary>`_. In order to parse an arbitrary
|
||||||
primary expression, we need to determine what sort of expression it is:
|
primary expression, we need to determine what sort of expression it is:
|
||||||
|
|
||||||
{% highlight python %} # primary ::= identifierexpr \| numberexpr \|
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# primary ::= identifierexpr \| numberexpr \|
|
||||||
parenexpr def ParsePrimary(self): if isinstance(self.current,
|
parenexpr def ParsePrimary(self): if isinstance(self.current,
|
||||||
IdentifierToken): return self.ParseIdentifierExpr() elif
|
IdentifierToken): return self.ParseIdentifierExpr() elif
|
||||||
isinstance(self.current, NumberToken): return self.ParseNumberExpr();
|
isinstance(self.current, NumberToken): return self.ParseNumberExpr();
|
||||||
elif self.current == CharacterToken('('): return self.ParseParenExpr()
|
elif self.current == CharacterToken('('): return self.ParseParenExpr()
|
||||||
else: raise RuntimeError('Unknown token when expecting an expression.')
|
else: raise RuntimeError('Unknown token when expecting an expression.')
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Now that you see the definition of this function, it is more obvious why
|
Now that you see the definition of this function, it is more obvious why
|
||||||
we can assume the state of ``Parser.current`` in the various functions.
|
we can assume the state of ``Parser.current`` in the various functions.
|
||||||
|
|
@ -278,9 +315,12 @@ recursion. To start with, we need a table of precedences. Remember the
|
||||||
``binop_precedence`` parameter we passed to the ``Parser`` constructor?
|
``binop_precedence`` parameter we passed to the ``Parser`` constructor?
|
||||||
Now is the time to use it:
|
Now is the time to use it:
|
||||||
|
|
||||||
{% highlight python %} def main(): # Install standard binary operators.
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def main(): # Install standard binary operators.
|
||||||
# 1 is lowest possible precedence. 40 is the highest.
|
# 1 is lowest possible precedence. 40 is the highest.
|
||||||
operator\_precedence = { '<': 10, '+': 20, '-': 20, '\*': 40 }
|
operator_precedence = { '<': 10, '+': 20, '-': 20, '\*': 40 }
|
||||||
|
|
||||||
# Run the main ``interpreter loop``. while True:
|
# Run the main ``interpreter loop``. while True:
|
||||||
|
|
||||||
|
|
@ -290,7 +330,9 @@ operator\_precedence = { '<': 10, '+': 20, '-': 20, '\*': 40 }
|
||||||
|
|
||||||
parser = Parser(Tokenize(raw), operator_precedence)
|
parser = Parser(Tokenize(raw), operator_precedence)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
For the basic form of Kaleidoscope, we will only support 4 binary
|
For the basic form of Kaleidoscope, we will only support 4 binary
|
||||||
operators (this can obviously be extended by you, our brave and intrepid
|
operators (this can obviously be extended by you, our brave and intrepid
|
||||||
|
|
@ -302,11 +344,16 @@ hardcode the comparisons.
|
||||||
We also define a helper function to get the precedence of the current
|
We also define a helper function to get the precedence of the current
|
||||||
token, or -1 if the token is not a binary operator:
|
token, or -1 if the token is not a binary operator:
|
||||||
|
|
||||||
{% highlight python %} # Gets the precedence of the current token, or -1
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Gets the precedence of the current token, or -1
|
||||||
if the token is not a binary # operator. def
|
if the token is not a binary # operator. def
|
||||||
GetCurrentTokenPrecedence(self): if isinstance(self.current,
|
GetCurrentTokenPrecedence(self): if isinstance(self.current,
|
||||||
CharacterToken): return self.binop\_precedence.get(self.current.char,
|
CharacterToken): return self.binop_precedence.get(self.current.char,
|
||||||
-1) else: return -1 {% endhighlight %}
|
-1) else: return -1
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
With the helper above defined, we can now start parsing binary
|
With the helper above defined, we can now start parsing binary
|
||||||
expressions. The basic idea of operator precedence parsing is to break
|
expressions. The basic idea of operator precedence parsing is to break
|
||||||
|
|
@ -322,9 +369,14 @@ doesn't need to worry about nested subexpressions like (c+d) at all.
|
||||||
To start, an expression is a primary expression potentially followed by
|
To start, an expression is a primary expression potentially followed by
|
||||||
a sequence of ``[binop,primaryexpr]`` pairs:
|
a sequence of ``[binop,primaryexpr]`` pairs:
|
||||||
|
|
||||||
{% highlight python %} # expression ::= primary binoprhs def
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# expression ::= primary binoprhs def
|
||||||
ParseExpression(self): left = self.ParsePrimary() return
|
ParseExpression(self): left = self.ParsePrimary() return
|
||||||
self.ParseBinOpRHS(left, 0) {% endhighlight %}
|
self.ParseBinOpRHS(left, 0)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
``ParseBinOpRHS`` is the function that parses the sequence of pairs for
|
``ParseBinOpRHS`` is the function that parses the sequence of pairs for
|
||||||
us. It takes a precedence and a pointer to an expression for the part
|
us. It takes a precedence and a pointer to an expression for the part
|
||||||
|
|
@ -341,8 +393,11 @@ is passed in a precedence of 40, it will not consume any tokens (because
|
||||||
the precedence of '+' is only 20). With this in mind, ``ParseBinOpRHS``
|
the precedence of '+' is only 20). With this in mind, ``ParseBinOpRHS``
|
||||||
starts with:
|
starts with:
|
||||||
|
|
||||||
{% highlight python %} # binoprhs ::= (operator primary)\* def
|
|
||||||
ParseBinOpRHS(self, left, left\_precedence): # If this is a binary
|
.. code-block:: python
|
||||||
|
|
||||||
|
# binoprhs ::= (operator primary)\* def
|
||||||
|
ParseBinOpRHS(self, left, left_precedence): # If this is a binary
|
||||||
operator, find its precedence. while True: precedence =
|
operator, find its precedence. while True: precedence =
|
||||||
self.GetCurrentTokenPrecedence()
|
self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
|
|
@ -353,7 +408,9 @@ self.GetCurrentTokenPrecedence()
|
||||||
if precedence < left_precedence:
|
if precedence < left_precedence:
|
||||||
return left
|
return left
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code gets the precedence of the current token and checks to see if
|
This code gets the precedence of the current token and checks to see if
|
||||||
if is too low. Because we defined invalid tokens to have a precedence of
|
if is too low. Because we defined invalid tokens to have a precedence of
|
||||||
|
|
@ -362,7 +419,10 @@ stream runs out of binary operators. If this check succeeds, we know
|
||||||
that the token is a binary operator and that it will be included in this
|
that the token is a binary operator and that it will be included in this
|
||||||
expression:
|
expression:
|
||||||
|
|
||||||
{% highlight python %} binary\_operator = self.current.char self.Next()
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
binary_operator = self.current.char self.Next()
|
||||||
# eat the operator.
|
# eat the operator.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -370,7 +430,9 @@ expression:
|
||||||
# Parse the primary expression after the binary operator.
|
# Parse the primary expression after the binary operator.
|
||||||
right = self.ParsePrimary()
|
right = self.ParsePrimary()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
As such, this code eats (and remembers) the binary operator and then
|
As such, this code eats (and remembers) the binary operator and then
|
||||||
parses the primary expression that follows. This builds up the whole
|
parses the primary expression that follows. This builds up the whole
|
||||||
|
|
@ -383,10 +445,15 @@ In particular, we could have ``(a+b) binop unparsed`` or
|
||||||
``binop`` to determine its precedence and compare it to BinOp's
|
``binop`` to determine its precedence and compare it to BinOp's
|
||||||
precedence (which is '+' in this case):
|
precedence (which is '+' in this case):
|
||||||
|
|
||||||
{% highlight python %} # If binary\_operator binds less tightly with
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# If binary_operator binds less tightly with
|
||||||
right than the operator after # right, let the pending operator take
|
right than the operator after # right, let the pending operator take
|
||||||
right as its left. next\_precedence = self.GetCurrentTokenPrecedence()
|
right as its left. next_precedence = self.GetCurrentTokenPrecedence()
|
||||||
if precedence < next\_precedence: {% endhighlight %}
|
if precedence < next_precedence:
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
If the precedence of the binop to the right of ``RHS`` is lower or equal
|
If the precedence of the binop to the right of ``RHS`` is lower or equal
|
||||||
to the precedence of our current operator, then we know that the
|
to the precedence of our current operator, then we know that the
|
||||||
|
|
@ -395,7 +462,10 @@ current operator is ``+`` and the next operator is ``+``, we know that
|
||||||
they have the same precedence. In this case we'll create the AST node
|
they have the same precedence. In this case we'll create the AST node
|
||||||
for ``a+b``, and then continue parsing:
|
for ``a+b``, and then continue parsing:
|
||||||
|
|
||||||
{% highlight python %} if precedence < next\_precedence: ... if body
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
if precedence < next_precedence: ... if body
|
||||||
omitted ...
|
omitted ...
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -403,7 +473,9 @@ omitted ...
|
||||||
# Merge left/right.
|
# Merge left/right.
|
||||||
left = BinaryOperatorExpressionNode(binary_operator, left, right);
|
left = BinaryOperatorExpressionNode(binary_operator, left, right);
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In our example above, this will turn ``a+b+`` into ``(a+b)`` and execute
|
In our example above, this will turn ``a+b+`` into ``(a+b)`` and execute
|
||||||
the next iteration of the loop, with ``+`` as the current token. The
|
the next iteration of the loop, with ``+`` as the current token. The
|
||||||
|
|
@ -420,10 +492,13 @@ all of ``( c + d ) * e * f`` as the RHS expression variable. The code to
|
||||||
do this is surprisingly simple (code from the above two blocks
|
do this is surprisingly simple (code from the above two blocks
|
||||||
duplicated for context):
|
duplicated for context):
|
||||||
|
|
||||||
{% highlight python %} # If binary\_operator binds less tightly with
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# If binary_operator binds less tightly with
|
||||||
right than the operator after # right, let the pending operator take
|
right than the operator after # right, let the pending operator take
|
||||||
right as its left. next\_precedence = self.GetCurrentTokenPrecedence()
|
right as its left. next_precedence = self.GetCurrentTokenPrecedence()
|
||||||
if precedence < next\_precedence: right = self.ParseBinOpRHS(right,
|
if precedence < next_precedence: right = self.ParseBinOpRHS(right,
|
||||||
precedence + 1)
|
precedence + 1)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -431,7 +506,9 @@ precedence + 1)
|
||||||
# Merge left/right.
|
# Merge left/right.
|
||||||
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
left = BinaryOperatorExpressionNode(binary_operator, left, right)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
At this point, we know that the binary operator to the RHS of our
|
At this point, we know that the binary operator to the RHS of our
|
||||||
primary has higher precedence than the binop we are currently parsing.
|
primary has higher precedence than the binop we are currently parsing.
|
||||||
|
|
@ -466,7 +543,10 @@ well as function body definitions. The code to do this is
|
||||||
straight-forward and not very interesting (once you've survived
|
straight-forward and not very interesting (once you've survived
|
||||||
expressions):
|
expressions):
|
||||||
|
|
||||||
{% highlight python %} # prototype ::= id '(' id\* ')' def
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# prototype ::= id '(' id\* ')' def
|
||||||
ParsePrototype(self): if not isinstance(self.current, IdentifierToken):
|
ParsePrototype(self): if not isinstance(self.current, IdentifierToken):
|
||||||
raise RuntimeError('Expected function name in prototype.')
|
raise RuntimeError('Expected function name in prototype.')
|
||||||
|
|
||||||
|
|
@ -492,31 +572,48 @@ raise RuntimeError('Expected function name in prototype.')
|
||||||
|
|
||||||
return PrototypeNode(function_name, arg_names)
|
return PrototypeNode(function_name, arg_names)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Given this, a function definition is very simple, just a prototype plus
|
Given this, a function definition is very simple, just a prototype plus
|
||||||
an expression to implement the body:
|
an expression to implement the body:
|
||||||
|
|
||||||
{% highlight python %} # definition ::= 'def' prototype expression def
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# definition ::= 'def' prototype expression def
|
||||||
ParseDefinition(self): self.Next() # eat def. proto =
|
ParseDefinition(self): self.Next() # eat def. proto =
|
||||||
self.ParsePrototype() body = self.ParseExpression() return
|
self.ParsePrototype() body = self.ParseExpression() return
|
||||||
FunctionNode(proto, body) {% endhighlight %}
|
FunctionNode(proto, body)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In addition, we support 'extern' to declare functions like 'sin' and
|
In addition, we support 'extern' to declare functions like 'sin' and
|
||||||
'cos' as well as to support forward declaration of user functions. These
|
'cos' as well as to support forward declaration of user functions. These
|
||||||
'extern's are just prototypes with no body:
|
'extern's are just prototypes with no body:
|
||||||
|
|
||||||
{% highlight python %} # external ::= 'extern' prototype def
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# external ::= 'extern' prototype def
|
||||||
ParseExtern(self): self.Next() # eat extern. return
|
ParseExtern(self): self.Next() # eat extern. return
|
||||||
self.ParsePrototype() {% endhighlight %}
|
self.ParsePrototype()
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Finally, we'll also let the user type in arbitrary top-level expressions
|
Finally, we'll also let the user type in arbitrary top-level expressions
|
||||||
and evaluate them on the fly. We will handle this by defining anonymous
|
and evaluate them on the fly. We will handle this by defining anonymous
|
||||||
nullary (zero argument) functions for them:
|
nullary (zero argument) functions for them:
|
||||||
|
|
||||||
{% highlight python %} # toplevelexpr ::= expression def
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# toplevelexpr ::= expression def
|
||||||
ParseTopLevelExpr(self): proto = PrototypeNode('', []) return
|
ParseTopLevelExpr(self): proto = PrototypeNode('', []) return
|
||||||
FunctionNode(proto, self.ParseExpression()) {% endhighlight %}
|
FunctionNode(proto, self.ParseExpression())
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Now that we have all the pieces, let's build a little driver that will
|
Now that we have all the pieces, let's build a little driver that will
|
||||||
let us actually *execute* this code we've built!
|
let us actually *execute* this code we've built!
|
||||||
|
|
@ -530,8 +627,11 @@ The driver for this simply invokes all of the parsing pieces with a
|
||||||
top-level dispatch loop. There isn't much interesting here, so I'll just
|
top-level dispatch loop. There isn't much interesting here, so I'll just
|
||||||
include the top-level loop. See `below <#code>`_ for full code.
|
include the top-level loop. See `below <#code>`_ for full code.
|
||||||
|
|
||||||
{% highlight python %} # Run the main "interpreter loop". while True:
|
|
||||||
print 'ready>', try: raw = raw\_input() except KeyboardInterrupt: return
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Run the main "interpreter loop". while True:
|
||||||
|
print 'ready>', try: raw = raw_input() except KeyboardInterrupt: return
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -547,7 +647,9 @@ print 'ready>', try: raw = raw\_input() except KeyboardInterrupt: return
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Here we create a new ``Parser`` for each line read, and try to parse out
|
Here we create a new ``Parser`` for each line read, and try to parse out
|
||||||
all the expressions, declarations and definitions in the line. We also
|
all the expressions, declarations and definitions in the line. We also
|
||||||
|
|
@ -564,12 +666,17 @@ lexer, parser, and AST builder. With this done, the executable will
|
||||||
validate Kaleidoscope code and tell us if it is grammatically invalid.
|
validate Kaleidoscope code and tell us if it is grammatically invalid.
|
||||||
For example, here is a sample interaction:
|
For example, here is a sample interaction:
|
||||||
|
|
||||||
{% highlight python %} $ python kaleidoscope.py ready> def foo(x y)
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
$ python kaleidoscope.py ready> def foo(x y)
|
||||||
x+foo(y, 4.0) Parsed a function definition. ready> def foo(x y) x+y y
|
x+foo(y, 4.0) Parsed a function definition. ready> def foo(x y) x+y y
|
||||||
Parsed a function definition. Parsed a top-level expression. ready> def
|
Parsed a function definition. Parsed a top-level expression. ready> def
|
||||||
foo(x y) x+y ) Parsed a function definition. Error: Unknown token when
|
foo(x y) x+y ) Parsed a function definition. Error: Unknown token when
|
||||||
expecting an expression. ready> extern sin(a); Parsed an extern. ready>
|
expecting an expression. ready> extern sin(a); Parsed an extern. ready>
|
||||||
^C $ {% endhighlight %}
|
^C $
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
There is a lot of room for extension here. You can define new AST nodes,
|
There is a lot of room for extension here. You can define new AST nodes,
|
||||||
extend the language in many ways, etc. In the `next
|
extend the language in many ways, etc. In the `next
|
||||||
|
|
@ -585,16 +692,17 @@ Here is the complete code listing for this and the previous chapter.
|
||||||
Note that it is fully self-contained: you don't need LLVM or any
|
Note that it is fully self-contained: you don't need LLVM or any
|
||||||
external libraries at all for this.
|
external libraries at all for this.
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
import re
|
import re
|
||||||
|
|
||||||
Lexer
|
Lexer
|
||||||
-----
|
-----
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
# The lexer yields one of these types for each token.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class EOFToken(object): pass
|
class EOFToken(object): pass
|
||||||
|
|
||||||
class DefToken(object): pass
|
class DefToken(object): pass
|
||||||
|
|
@ -612,11 +720,9 @@ char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
== other
|
== other
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
=============================================================
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
string[0].isspace(): string = string[1:] continue
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
@ -656,51 +762,33 @@ yield EOFToken()
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
# Expression class for numeric literals like "1.0".
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode): def **init**\ (self, value):
|
class NumberExpressionNode(ExpressionNode): def **init**\ (self, value):
|
||||||
self.value = value
|
self.value = value
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
# Expression class for referencing a variable, like "a".
|
||||||
======================================================
|
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode): def **init**\ (self,
|
class VariableExpressionNode(ExpressionNode): def **init**\ (self,
|
||||||
name): self.name = name
|
name): self.name = name
|
||||||
|
|
||||||
Expression class for a binary operator.
|
# Expression class for a binary operator.
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode): def **init**\ (self,
|
class BinaryOperatorExpressionNode(ExpressionNode): def **init**\ (self,
|
||||||
operator, left, right): self.operator = operator self.left = left
|
operator, left, right): self.operator = operator self.left = left
|
||||||
self.right = right
|
self.right = right
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode): def **init**\ (self, callee,
|
class CallExpressionNode(ExpressionNode): def **init**\ (self, callee,
|
||||||
args): self.callee = callee self.args = args
|
args): self.callee = callee self.args = args
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
==============================================================================
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes).
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
|
||||||
============================================================================
|
|
||||||
|
|
||||||
takes).
|
|
||||||
=======
|
|
||||||
|
|
||||||
class PrototypeNode(object): def **init**\ (self, name, args): self.name
|
class PrototypeNode(object): def **init**\ (self, name, args): self.name
|
||||||
= name self.args = args
|
= name self.args = args
|
||||||
|
|
||||||
This class represents a function definition itself.
|
# This class represents a function definition itself.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class FunctionNode(object): def **init**\ (self, prototype, body):
|
class FunctionNode(object): def **init**\ (self, prototype, body):
|
||||||
self.prototype = prototype self.body = body
|
self.prototype = prototype self.body = body
|
||||||
|
|
||||||
|
|
@ -709,8 +797,8 @@ Parser
|
||||||
|
|
||||||
class Parser(object):
|
class Parser(object):
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
|
|
@ -720,10 +808,10 @@ self.current = self.tokens.next()
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
isinstance(self.current, CharacterToken): return
|
isinstance(self.current, CharacterToken): return
|
||||||
self.binop\_precedence.get(self.current.char, -1) else: return -1
|
self.binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
self.Next() # eat identifier.
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -771,7 +859,7 @@ return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
||||||
when expecting an expression.')
|
when expecting an expression.')
|
||||||
|
|
||||||
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -852,11 +940,11 @@ Main driver code.
|
||||||
-----------------
|
-----------------
|
||||||
|
|
||||||
def main(): # Install standard binary operators. # 1 is lowest possible
|
def main(): # Install standard binary operators. # 1 is lowest possible
|
||||||
precedence. 40 is the highest. operator\_precedence = { '<': 10, '+':
|
precedence. 40 is the highest. operator_precedence = { '<': 10, '+':
|
||||||
20, '-': 20, '\*': 40 }
|
20, '-': 20, '\*': 40 }
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: return
|
= raw_input() except KeyboardInterrupt: return
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -872,9 +960,4 @@ precedence. 40 is the highest. operator\_precedence = { '<': 10, '+':
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Implementing Code Generation to LLVM
|
|
||||||
IR <PythonLangImpl3.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -31,23 +31,26 @@ Code Generation Setup # {#basics}
|
||||||
In order to generate LLVM IR, we want some simple setup to get started.
|
In order to generate LLVM IR, we want some simple setup to get started.
|
||||||
First we define code generation methods in each AST node class:
|
First we define code generation methods in each AST node class:
|
||||||
|
|
||||||
{% highlight python %} # Expression class for numeric literals like
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Expression class for numeric literals like
|
||||||
"1.0". class NumberExpressionNode(ExpressionNode):
|
"1.0". class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
def **init**\ (self, value): self.value = value
|
||||||
|
|
||||||
def CodeGen(self): ...
|
def CodeGen(self): ...
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
# Expression class for referencing a variable, like "a".
|
||||||
======================================================
|
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
def CodeGen(self): ...
|
def CodeGen(self): ...
|
||||||
|
|
||||||
... {% endhighlight %}
|
...
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The ``CodeGen`` method says to emit IR for that AST node along with all
|
The ``CodeGen`` method says to emit IR for that AST node along with all
|
||||||
the things it depends on, and they all return an LLVM Value object.
|
the things it depends on, and they all return an LLVM Value object.
|
||||||
|
|
@ -64,21 +67,20 @@ Assignment <http://en.wikipedia.org/wiki/Static_single_assignment_form>`_
|
||||||
We will also need to define some global variables which we will be used
|
We will also need to define some global variables which we will be used
|
||||||
during code generation:
|
during code generation:
|
||||||
|
|
||||||
{% highlight python %} # The LLVM module, which holds all the IR code.
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
.. code-block:: python
|
||||||
=========================================================================
|
|
||||||
|
|
||||||
g\_llvm\_builder = None
|
# The LLVM module, which holds all the IR code.
|
||||||
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
==============================================================================
|
g_llvm_builder = None
|
||||||
|
|
||||||
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
|
|
||||||
and what their LLVM representation is.
|
|
||||||
======================================
|
|
||||||
|
|
||||||
g\_named\_values = {} {% endhighlight %}
|
|
||||||
|
|
||||||
``g_llvm_module`` is the LLVM construct that contains all of the
|
``g_llvm_module`` is the LLVM construct that contains all of the
|
||||||
functions and global variables in a chunk of code. In many ways, it is
|
functions and global variables in a chunk of code. In many ways, it is
|
||||||
|
|
@ -112,8 +114,13 @@ Generating LLVM code for expression nodes is very straightforward: less
|
||||||
than 35 lines of commented code for all four of our expression nodes.
|
than 35 lines of commented code for all four of our expression nodes.
|
||||||
First we'll do numeric literals:
|
First we'll do numeric literals:
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): return
|
|
||||||
Constant.real(Type.double(), self.value) {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): return
|
||||||
|
Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In llvmpy, floating point numeric constants are represented with the
|
In llvmpy, floating point numeric constants are represented with the
|
||||||
``llvm.core.ConstantFP`` class. To create one, we can use the static
|
``llvm.core.ConstantFP`` class. To create one, we can use the static
|
||||||
|
|
@ -123,9 +130,14 @@ LLVM IR constants are all uniqued together and shared. For this reason,
|
||||||
we create the constant through a factory method instead of instantiating
|
we create the constant through a factory method instead of instantiating
|
||||||
one directly.
|
one directly.
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): if self.name in
|
|
||||||
g\_named\_values: return g\_named\_values[self.name] else: raise
|
.. code-block:: python
|
||||||
RuntimeError('Unknown variable name: ' + self.name) {% endhighlight %}
|
|
||||||
|
def CodeGen(self): if self.name in
|
||||||
|
g_named_values: return g_named_values[self.name] else: raise
|
||||||
|
RuntimeError('Unknown variable name: ' + self.name)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
References to variables are also quite simple using LLVM. In the simple
|
References to variables are also quite simple using LLVM. In the simple
|
||||||
version of Kaleidoscope, we assume that the variable has already been
|
version of Kaleidoscope, we assume that the variable has already been
|
||||||
|
|
@ -137,7 +149,10 @@ the value for it. In future chapters, we'll add support for `loop
|
||||||
induction variables <PythonLangImpl5.html#for>`_ in the symbol table,
|
induction variables <PythonLangImpl5.html#for>`_ in the symbol table,
|
||||||
and for `local variables <PythonLangImpl7.html#localvars>`_.
|
and for `local variables <PythonLangImpl7.html#localvars>`_.
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): left = self.left.CodeGen()
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): left = self.left.CodeGen()
|
||||||
right = self.right.CodeGen()
|
right = self.right.CodeGen()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -155,7 +170,9 @@ right = self.right.CodeGen()
|
||||||
else:
|
else:
|
||||||
raise RuntimeError('Unknown binary operator.')
|
raise RuntimeError('Unknown binary operator.')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Binary operators start to get more interesting. The basic idea here is
|
Binary operators start to get more interesting. The basic idea here is
|
||||||
that we recursively emit code for the left-hand side of the expression,
|
that we recursively emit code for the left-hand side of the expression,
|
||||||
|
|
@ -193,9 +210,12 @@ treating the input as an unsigned value. In contrast, if we used the
|
||||||
the Kaleidoscope ``<`` operator would return 0.0 and -1.0, depending on
|
the Kaleidoscope ``<`` operator would return 0.0 and -1.0, depending on
|
||||||
the input value.
|
the input value.
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): # Look up the name in the
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): # Look up the name in the
|
||||||
global module table. callee =
|
global module table. callee =
|
||||||
g\_llvm\_module.get\_function\_named(self.callee)
|
g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -207,7 +227,9 @@ g\_llvm\_module.get\_function\_named(self.callee)
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Code generation for function calls is quite straightforward with LLVM.
|
Code generation for function calls is quite straightforward with LLVM.
|
||||||
The code above initially does a function name lookup in the LLVM
|
The code above initially does a function name lookup in the LLVM
|
||||||
|
|
@ -242,15 +264,20 @@ let's talk about code generation for prototypes: they are used both for
|
||||||
function bodies and external function declarations. The code starts
|
function bodies and external function declarations. The code starts
|
||||||
with:
|
with:
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): # Make the function type, eg.
|
|
||||||
double(double,double). funct\_type = Type.function( Type.double(),
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): # Make the function type, eg.
|
||||||
|
double(double,double). funct_type = Type.function( Type.double(),
|
||||||
[Type.double()] \* len(self.args), False)
|
[Type.double()] \* len(self.args), False)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
function = Function.new(g_llvm_module, funct_type, self.name)
|
function = Function.new(g_llvm_module, funct_type, self.name)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The call to ``Type.function`` creates the ``FunctionType`` that should
|
The call to ``Type.function`` creates the ``FunctionType`` that should
|
||||||
be used for a given Prototype. Since all function arguments in
|
be used for a given Prototype. Since all function arguments in
|
||||||
|
|
@ -272,11 +299,16 @@ the name the user specified: since ``g_llvm_module`` is specified, this
|
||||||
name is registered in ``g_llvm_module``'s symbol table, which is used by
|
name is registered in ``g_llvm_module``'s symbol table, which is used by
|
||||||
the function call code above.
|
the function call code above.
|
||||||
|
|
||||||
{% highlight python %} # If the name conflicted, there was already
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# If the name conflicted, there was already
|
||||||
something with the same name. # If it has a body, don't allow
|
something with the same name. # If it has a body, don't allow
|
||||||
redefinition or reextern. if function.name != self.name:
|
redefinition or reextern. if function.name != self.name:
|
||||||
function.delete() function =
|
function.delete() function =
|
||||||
g\_llvm\_module.get\_function\_named(self.name) {% endhighlight %}
|
g_llvm_module.get_function_named(self.name)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The Module symbol table works just like the Function symbol table when
|
The Module symbol table works just like the Function symbol table when
|
||||||
it comes to name conflicts: if a new function is created with a name was
|
it comes to name conflicts: if a new function is created with a name was
|
||||||
|
|
@ -298,8 +330,11 @@ function we just created (by calling ``delete``) and then calling
|
||||||
``get_function_named`` to get the existing function with the specified
|
``get_function_named`` to get the existing function with the specified
|
||||||
name.
|
name.
|
||||||
|
|
||||||
{% highlight python %} # If the function already has a body, reject
|
|
||||||
this. if not function.is\_declaration: raise RuntimeError('Redefinition
|
.. code-block:: python
|
||||||
|
|
||||||
|
# If the function already has a body, reject
|
||||||
|
this. if not function.is_declaration: raise RuntimeError('Redefinition
|
||||||
of function.')
|
of function.')
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -309,7 +344,9 @@ of function.')
|
||||||
raise RuntimeError('Redeclaration of a function with different number '
|
raise RuntimeError('Redeclaration of a function with different number '
|
||||||
'of args.')
|
'of args.')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In order to verify the logic above, we first check to see if the
|
In order to verify the logic above, we first check to see if the
|
||||||
pre-existing function is a forward declaration. Since we don't allow
|
pre-existing function is a forward declaration. Since we don't allow
|
||||||
|
|
@ -318,16 +355,21 @@ case. If the previous reference to a function was an 'extern', we simply
|
||||||
verify that the number of arguments for that definition and this one
|
verify that the number of arguments for that definition and this one
|
||||||
match up. If not, we emit an error.
|
match up. If not, we emit an error.
|
||||||
|
|
||||||
{% highlight python %} # Set names for all arguments and add them to the
|
|
||||||
variables symbol table. for arg, arg\_name in zip(function.args,
|
.. code-block:: python
|
||||||
self.args): arg.name = arg\_name # Add arguments to variable symbol
|
|
||||||
table. g\_named\_values[arg\_name] = arg
|
# Set names for all arguments and add them to the
|
||||||
|
variables symbol table. for arg, arg_name in zip(function.args,
|
||||||
|
self.args): arg.name = arg_name # Add arguments to variable symbol
|
||||||
|
table. g_named_values[arg_name] = arg
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The last bit of code for prototypes loops over all of the arguments in
|
The last bit of code for prototypes loops over all of the arguments in
|
||||||
the function, setting the name of the LLVM Argument objects to match,
|
the function, setting the name of the LLVM Argument objects to match,
|
||||||
|
|
@ -338,15 +380,20 @@ would be very straight-forward with the mechanics we have already used
|
||||||
above. Once this is all set up, it returns the Function object to the
|
above. Once this is all set up, it returns the Function object to the
|
||||||
caller.
|
caller.
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): # Clear scope.
|
|
||||||
g\_named\_values.clear()
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): # Clear scope.
|
||||||
|
g_named_values.clear()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
# Create a function object.
|
# Create a function object.
|
||||||
function = self.prototype.CodeGen()
|
function = self.prototype.CodeGen()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Code generation for function definitions starts out simply enough: we
|
Code generation for function definitions starts out simply enough: we
|
||||||
just clear out the ``g_named_values`` dictionary to make sure that there
|
just clear out the ``g_named_values`` dictionary to make sure that there
|
||||||
|
|
@ -354,9 +401,12 @@ isn't anything in it from the last function we compiled and codegen the
|
||||||
prototype. Code generation of the prototype ensures that there is an
|
prototype. Code generation of the prototype ensures that there is an
|
||||||
LLVM Function object that is ready to go for us.
|
LLVM Function object that is ready to go for us.
|
||||||
|
|
||||||
{% highlight python %} # Create a new basic block to start insertion
|
|
||||||
into. block = function.append\_basic\_block('entry') global
|
.. code-block:: python
|
||||||
g\_llvm\_builder g\_llvm\_builder = Builder.new(block) {% endhighlight
|
|
||||||
|
# Create a new basic block to start insertion
|
||||||
|
into. block = function.append_basic_block('entry') global
|
||||||
|
g_llvm_builder g_llvm_builder = Builder.new(block) {% endhighlight
|
||||||
%}
|
%}
|
||||||
|
|
||||||
Now we get to the point where ``g_llvm_builder`` is set up. The first
|
Now we get to the point where ``g_llvm_builder`` is set up. The first
|
||||||
|
|
@ -371,15 +421,17 @@ Graph <http://en.wikipedia.org/wiki/Control_flow_graph>`_. Since we
|
||||||
don't have any control flow, our functions will only contain one block
|
don't have any control flow, our functions will only contain one block
|
||||||
at this point. We'll fix this in `Chapter 5 <PythonLangImpl5.html>`_ :).
|
at this point. We'll fix this in `Chapter 5 <PythonLangImpl5.html>`_ :).
|
||||||
|
|
||||||
{% highlight python %} # Finish off the function. try: return\_value =
|
{% highlight python %} # Finish off the function. try: return_value =
|
||||||
self.body.CodeGen() g\_llvm\_builder.ret(return\_value)
|
self.body.CodeGen() g_llvm_builder.ret(return_value)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
# Validate the generated code, checking for consistency.
|
# Validate the generated code, checking for consistency.
|
||||||
function.verify()
|
function.verify()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Once the insertion point is set up, we call the ``CodeGen`` method for
|
Once the insertion point is set up, we call the ``CodeGen`` method for
|
||||||
the root expression of the function. If no error happens, this emits
|
the root expression of the function. If no error happens, this emits
|
||||||
|
|
@ -392,13 +444,18 @@ checks on the generated code, to determine if our compiler is doing
|
||||||
everything right. Using this is important: it can catch a lot of bugs.
|
everything right. Using this is important: it can catch a lot of bugs.
|
||||||
Once the function is finished and validated, we return it.
|
Once the function is finished and validated, we return it.
|
||||||
|
|
||||||
{% highlight python %} except: function.delete() raise
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
except: function.delete() raise
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The only piece left here is handling of the error case. For simplicity,
|
The only piece left here is handling of the error case. For simplicity,
|
||||||
we handle this by merely deleting the function we produced with the
|
we handle this by merely deleting the function we produced with the
|
||||||
|
|
@ -411,9 +468,14 @@ can return a previously defined forward declaration, our code can
|
||||||
actually delete a forward declaration. There are a number of ways to fix
|
actually delete a forward declaration. There are a number of ways to fix
|
||||||
this bug; see what you can come up with! Here is a testcase:
|
this bug; see what you can come up with! Here is a testcase:
|
||||||
|
|
||||||
{% highlight python %} extern foo(a b) # ok, defines foo. def foo(a b) c
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
extern foo(a b) # ok, defines foo. def foo(a b) c
|
||||||
# error, 'c' is invalid. def bar() foo(1, 2) # error, unknown function
|
# error, 'c' is invalid. def bar() foo(1, 2) # error, unknown function
|
||||||
"foo" {% endhighlight %}
|
"foo"
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
--------------
|
--------------
|
||||||
|
|
||||||
|
|
@ -426,8 +488,13 @@ CodeGen into the ``Handle*`` functions, and then dumps out the LLVM IR.
|
||||||
This gives a nice way to look at the LLVM IR for simple functions. For
|
This gives a nice way to look at the LLVM IR for simple functions. For
|
||||||
example:
|
example:
|
||||||
|
|
||||||
{% highlight bash %} ready> 4+5 Read a top-level expression: define
|
|
||||||
double @0() { entry: ret double 9.000000e+00 } {% endhighlight %}
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> 4+5 Read a top-level expression: define
|
||||||
|
double @0() { entry: ret double 9.000000e+00 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Note how the parser turns the top-level expression into anonymous
|
Note how the parser turns the top-level expression into anonymous
|
||||||
functions for us. This will be handy when we add JIT support in the next
|
functions for us. This will be handy when we add JIT support in the next
|
||||||
|
|
@ -435,37 +502,55 @@ chapter. Also note that the code is very literally transcribed, no
|
||||||
optimizations are being performed except simple constant folding done by
|
optimizations are being performed except simple constant folding done by
|
||||||
the Builder. We will add optimizations explicitly in the next chapter.
|
the Builder. We will add optimizations explicitly in the next chapter.
|
||||||
|
|
||||||
{% highlight bash %} ready> def foo(a b) a\ *a + 2*\ a\ *b + b*\ b Read
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> def foo(a b) a\ *a + 2*\ a\ *b + b*\ b Read
|
||||||
a function definition: define double @foo(double %a, double %b) { entry:
|
a function definition: define double @foo(double %a, double %b) { entry:
|
||||||
%multmp = fmul double %a, %a ; [#uses=1] %multmp1 = fmul double
|
%multmp = fmul double %a, %a ; [#uses=1] %multmp1 = fmul double
|
||||||
2.000000e+00, %a ; [#uses=1] %multmp2 = fmul double %multmp1, %b ;
|
2.000000e+00, %a ; [#uses=1] %multmp2 = fmul double %multmp1, %b ;
|
||||||
[#uses=1] %addtmp = fadd double %multmp, %multmp2 ; [#uses=1] %multmp3 =
|
[#uses=1] %addtmp = fadd double %multmp, %multmp2 ; [#uses=1] %multmp3 =
|
||||||
fmul double %b, %b ; [#uses=1] %addtmp4 = fadd double %addtmp, %multmp3
|
fmul double %b, %b ; [#uses=1] %addtmp4 = fadd double %addtmp, %multmp3
|
||||||
; [#uses=1] ret double %addtmp4 } {% endhighlight %}
|
; [#uses=1] ret double %addtmp4 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This shows some simple arithmetic. Notice the striking similarity to the
|
This shows some simple arithmetic. Notice the striking similarity to the
|
||||||
LLVM builder calls that we use to create the instructions.
|
LLVM builder calls that we use to create the instructions.
|
||||||
|
|
||||||
{% highlight bash %} ready> def bar(a) foo(a, 4.0) + bar(31337) Read a
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> def bar(a) foo(a, 4.0) + bar(31337) Read a
|
||||||
function definition: define double @bar(double %a) { entry: %calltmp =
|
function definition: define double @bar(double %a) { entry: %calltmp =
|
||||||
call double @foo(double %a, double 4.000000e+00) ; [#uses=1] %calltmp1 =
|
call double @foo(double %a, double 4.000000e+00) ; [#uses=1] %calltmp1 =
|
||||||
call double @bar(double 3.133700e+04) ; [#uses=1] %addtmp = fadd double
|
call double @bar(double 3.133700e+04) ; [#uses=1] %addtmp = fadd double
|
||||||
%calltmp, %calltmp1 ; [#uses=1] ret double %addtmp } {% endhighlight %}
|
%calltmp, %calltmp1 ; [#uses=1] ret double %addtmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This shows some function calls. Note that this function will take a long
|
This shows some function calls. Note that this function will take a long
|
||||||
time to execute if you call it. In the future we'll add conditional
|
time to execute if you call it. In the future we'll add conditional
|
||||||
control flow to actually make recursion useful :).
|
control flow to actually make recursion useful :).
|
||||||
|
|
||||||
{% highlight bash %} ready> extern cos(x) Read extern: declare double
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> extern cos(x) Read extern: declare double
|
||||||
@cos(double)
|
@cos(double)
|
||||||
|
|
||||||
ready> cos(1.234) Read a top-level expression: define double @1() {
|
ready> cos(1.234) Read a top-level expression: define double @1() {
|
||||||
entry: %calltmp = call double @cos(double 1.234000e+00) ; [#uses=1] ret
|
entry: %calltmp = call double @cos(double 1.234000e+00) ; [#uses=1] ret
|
||||||
double %calltmp } {% endhighlight %}
|
double %calltmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This shows an extern for the libm "cos" function, and a call to it.
|
This shows an extern for the libm "cos" function, and a call to it.
|
||||||
|
|
||||||
{% highlight bash %} ready> ^C ; ModuleID = 'my cool jit'
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> ^C ; ModuleID = 'my cool jit'
|
||||||
|
|
||||||
define double @0() { entry: ret double 9.000000e+00 }
|
define double @0() { entry: ret double 9.000000e+00 }
|
||||||
|
|
||||||
|
|
@ -484,7 +569,9 @@ define double @bar(double %a) { entry: %calltmp = call double
|
||||||
declare double @cos(double)
|
declare double @cos(double)
|
||||||
|
|
||||||
define double @1() { entry: %calltmp = call double @cos(double
|
define double @1() { entry: %calltmp = call double @cos(double
|
||||||
1.234000e+00) ; [#uses=1] ret double %calltmp } {% endhighlight %}
|
1.234000e+00) ; [#uses=1] ret double %calltmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
When you quit the current demo, it dumps out the IR for the entire
|
When you quit the current demo, it dumps out the IR for the entire
|
||||||
module generated. Here you can see the big picture with all the
|
module generated. Here you can see the big picture with all the
|
||||||
|
|
@ -505,38 +592,31 @@ the LLVM code generator. Because this uses the llvmpy libraries, you
|
||||||
need to `download <../download.html>`_ and
|
need to `download <../download.html>`_ and
|
||||||
`install <../userguide.html#install>`_ them.
|
`install <../userguide.html#install>`_ them.
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
import re from llvm.core import Module, Constant, Type, Function,
|
import re from llvm.core import Module, Constant, Type, Function,
|
||||||
Builder, FCMP\_ULT
|
Builder, FCMP_ULT
|
||||||
|
|
||||||
Globals
|
Globals
|
||||||
-------
|
-------
|
||||||
|
|
||||||
The LLVM module, which holds all the IR code.
|
# The LLVM module, which holds all the IR code.
|
||||||
=============================================
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
|
g_llvm_builder = None
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
=========================================================================
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
g\_llvm\_builder = None
|
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
|
||||||
==============================================================================
|
|
||||||
|
|
||||||
and what their LLVM representation is.
|
|
||||||
======================================
|
|
||||||
|
|
||||||
g\_named\_values = {}
|
|
||||||
|
|
||||||
Lexer
|
Lexer
|
||||||
-----
|
-----
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
# The lexer yields one of these types for each token.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class EOFToken(object): pass
|
class EOFToken(object): pass
|
||||||
|
|
||||||
class DefToken(object): pass
|
class DefToken(object): pass
|
||||||
|
|
@ -554,11 +634,9 @@ char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
== other
|
== other
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
=============================================================
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
string[0].isspace(): string = string[1:] continue
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
@ -598,34 +676,26 @@ yield EOFToken()
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
# Expression class for numeric literals like "1.0".
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode):
|
class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
def **init**\ (self, value): self.value = value
|
||||||
|
|
||||||
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
# Expression class for referencing a variable, like "a".
|
||||||
======================================================
|
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
def CodeGen(self): if self.name in g\_named\_values: return
|
def CodeGen(self): if self.name in g_named_values: return
|
||||||
g\_named\_values[self.name] else: raise RuntimeError('Unknown variable
|
g_named_values[self.name] else: raise RuntimeError('Unknown variable
|
||||||
name: ' + self.name)
|
name: ' + self.name)
|
||||||
|
|
||||||
Expression class for a binary operator.
|
# Expression class for a binary operator.
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode):
|
class BinaryOperatorExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, operator, left, right): self.operator = operator
|
def **init**\ (self, operator, left, right): self.operator = operator
|
||||||
|
|
@ -649,16 +719,14 @@ self.right.CodeGen()
|
||||||
else:
|
else:
|
||||||
raise RuntimeError('Unknown binary operator.')
|
raise RuntimeError('Unknown binary operator.')
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode):
|
class CallExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, callee, args): self.callee = callee self.args =
|
def **init**\ (self, callee, args): self.callee = callee self.args =
|
||||||
args
|
args
|
||||||
|
|
||||||
def CodeGen(self): # Look up the name in the global module table. callee
|
def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
= g\_llvm\_module.get\_function\_named(self.callee)
|
= g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -670,21 +738,15 @@ def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
==============================================================================
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes).
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
|
||||||
============================================================================
|
|
||||||
|
|
||||||
takes).
|
|
||||||
=======
|
|
||||||
|
|
||||||
class PrototypeNode(object):
|
class PrototypeNode(object):
|
||||||
|
|
||||||
def **init**\ (self, name, args): self.name = name self.args = args
|
def **init**\ (self, name, args): self.name = name self.args = args
|
||||||
|
|
||||||
def CodeGen(self): # Make the function type, eg. double(double,double).
|
def CodeGen(self): # Make the function type, eg. double(double,double).
|
||||||
funct\_type = Type.function( Type.double(), [Type.double()] \*
|
funct_type = Type.function( Type.double(), [Type.double()] \*
|
||||||
len(self.args), False)
|
len(self.args), False)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -714,15 +776,13 @@ len(self.args), False)
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
This class represents a function definition itself.
|
# This class represents a function definition itself.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class FunctionNode(object):
|
class FunctionNode(object):
|
||||||
|
|
||||||
def **init**\ (self, prototype, body): self.prototype = prototype
|
def **init**\ (self, prototype, body): self.prototype = prototype
|
||||||
self.body = body
|
self.body = body
|
||||||
|
|
||||||
def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
def CodeGen(self): # Clear scope. g_named_values.clear()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -752,8 +812,8 @@ Parser
|
||||||
|
|
||||||
class Parser(object):
|
class Parser(object):
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
|
|
@ -763,10 +823,10 @@ self.current = self.tokens.next()
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
isinstance(self.current, CharacterToken): return
|
isinstance(self.current, CharacterToken): return
|
||||||
self.binop\_precedence.get(self.current.char, -1) else: return -1
|
self.binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
self.Next() # eat identifier.
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -814,7 +874,7 @@ return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
||||||
when expecting an expression.')
|
when expecting an expression.')
|
||||||
|
|
||||||
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -894,11 +954,11 @@ Main driver code.
|
||||||
-----------------
|
-----------------
|
||||||
|
|
||||||
def main(): # Install standard binary operators. # 1 is lowest possible
|
def main(): # Install standard binary operators. # 1 is lowest possible
|
||||||
precedence. 40 is the highest. operator\_precedence = { '<': 10, '+':
|
precedence. 40 is the highest. operator_precedence = { '<': 10, '+':
|
||||||
20, '-': 20, '\*': 40 }
|
20, '-': 20, '\*': 40 }
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: break
|
= raw_input() except KeyboardInterrupt: break
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -914,10 +974,6 @@ precedence. 40 is the highest. operator\_precedence = { '<': 10, '+':
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
# Print out all of the generated code. print '', g\_llvm\_module
|
# Print out all of the generated code. print '', g_llvm_module
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Adding JIT and Optimizer Support <PythonLangImpl4.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -25,17 +25,27 @@ Our demonstration for Chapter 3 is elegant and easy to extend.
|
||||||
Unfortunately, it does not produce wonderful code. The LLVM Builder,
|
Unfortunately, it does not produce wonderful code. The LLVM Builder,
|
||||||
however, does give us obvious optimizations when compiling simple code:
|
however, does give us obvious optimizations when compiling simple code:
|
||||||
|
|
||||||
{% highlight bash %} ready> def test(x) 1+2+x Read function definition:
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> def test(x) 1+2+x Read function definition:
|
||||||
define double @test(double %x) { entry: %addtmp = fadd double
|
define double @test(double %x) { entry: %addtmp = fadd double
|
||||||
3.000000e+00, %x ret double %addtmp } {% endhighlight %}
|
3.000000e+00, %x ret double %addtmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code is not a literal transcription of the AST built by parsing the
|
This code is not a literal transcription of the AST built by parsing the
|
||||||
input. That would be:
|
input. That would be:
|
||||||
|
|
||||||
{% highlight bash %} ready> def test(x) 1+2+x Read function definition:
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> def test(x) 1+2+x Read function definition:
|
||||||
define double @test(double %x) { entry: %addtmp = fadd double
|
define double @test(double %x) { entry: %addtmp = fadd double
|
||||||
2.000000e+00, 1.000000e+00 %addtmp1 = fadd double %addtmp, %x ret double
|
2.000000e+00, 1.000000e+00 %addtmp1 = fadd double %addtmp, %x ret double
|
||||||
%addtmp1 } {% endhighlight %}
|
%addtmp1 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Constant folding, as seen above, in particular, is a very common and
|
Constant folding, as seen above, in particular, is a very common and
|
||||||
very important optimization: so much so that many language implementors
|
very important optimization: so much so that many language implementors
|
||||||
|
|
@ -58,11 +68,16 @@ On the other hand, the ``Builder`` is limited by the fact that it does
|
||||||
all of its analysis inline with the code as it is built. If you take a
|
all of its analysis inline with the code as it is built. If you take a
|
||||||
slightly more complex example:
|
slightly more complex example:
|
||||||
|
|
||||||
{% highlight bash %} ready> def test(x) (1+2+x)\*(x+(1+2)) Read a
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> def test(x) (1+2+x)\*(x+(1+2)) Read a
|
||||||
function definition: define double @test(double %x) { entry: %addtmp =
|
function definition: define double @test(double %x) { entry: %addtmp =
|
||||||
fadd double 3.000000e+00, %x ; [#uses=1] %addtmp1 = fadd double %x,
|
fadd double 3.000000e+00, %x ; [#uses=1] %addtmp1 = fadd double %x,
|
||||||
3.000000e+00 ; [#uses=1] %multmp = fmul double %addtmp, %addtmp1 ;
|
3.000000e+00 ; [#uses=1] %multmp = fmul double %addtmp, %addtmp1 ;
|
||||||
[#uses=1] ret double %multmp } {% endhighlight %}
|
[#uses=1] ret double %multmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In this case, the LHS and RHS of the multiplication are the same value.
|
In this case, the LHS and RHS of the multiplication are the same value.
|
||||||
We'd really like to see this generate"``tmp = x+3; result = tmp*tmp;``
|
We'd really like to see this generate"``tmp = x+3; result = tmp*tmp;``
|
||||||
|
|
@ -112,27 +127,30 @@ to hold and organize the LLVM optimizations that we want to run. Once we
|
||||||
have that, we can add a set of optimizations to run. The code looks like
|
have that, we can add a set of optimizations to run. The code looks like
|
||||||
this:
|
this:
|
||||||
|
|
||||||
{% highlight python %} # The function optimization passes manager.
|
|
||||||
g\_llvm\_pass\_manager = FunctionPassManager.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
The LLVM execution engine.
|
.. code-block:: python
|
||||||
==========================
|
|
||||||
|
|
||||||
g\_llvm\_executor = ExecutionEngine.new(g\_llvm\_module)
|
# The function optimization passes manager.
|
||||||
|
g_llvm_pass_manager = FunctionPassManager.new(g_llvm_module)
|
||||||
|
|
||||||
|
# The LLVM execution engine.
|
||||||
|
g_llvm_executor = ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
...
|
...
|
||||||
|
|
||||||
def main(): # Set up the optimizer pipeline. Start with registering info
|
def main(): # Set up the optimizer pipeline. Start with registering info
|
||||||
about how the # target lays out data structures.
|
about how the # target lays out data structures.
|
||||||
g\_llvm\_pass\_manager.add(g\_llvm\_executor.target\_data) # Do simple
|
g_llvm_pass_manager.add(g_llvm_executor.target_data) # Do simple
|
||||||
"peephole" optimizations and bit-twiddling optzns.
|
"peephole" optimizations and bit-twiddling optzns.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_INSTRUCTION\_COMBINING) # Reassociate
|
g_llvm_pass_manager.add(PASS_INSTRUCTION_COMBINING) # Reassociate
|
||||||
expressions. g\_llvm\_pass\_manager.add(PASS\_REASSOCIATE) # Eliminate
|
expressions. g_llvm_pass_manager.add(PASS_REASSOCIATE) # Eliminate
|
||||||
Common SubExpressions. g\_llvm\_pass\_manager.add(PASS\_GVN) # Simplify
|
Common SubExpressions. g_llvm_pass_manager.add(PASS_GVN) # Simplify
|
||||||
the control flow graph (deleting unreachable blocks, etc).
|
the control flow graph (deleting unreachable blocks, etc).
|
||||||
g\_llvm\_pass\_manager.add(PASS\_CFG\_SIMPLIFICATION)
|
g_llvm_pass_manager.add(PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
|
g_llvm_pass_manager.initialize()
|
||||||
|
|
||||||
|
|
||||||
g\_llvm\_pass\_manager.initialize() {% endhighlight %}
|
|
||||||
|
|
||||||
This code defines a ``FunctionPassManager``, ``g_llvm_pass_manager``.
|
This code defines a ``FunctionPassManager``, ``g_llvm_pass_manager``.
|
||||||
Once it is set up, we use a series of "add" calls to add a bunch of LLVM
|
Once it is set up, we use a series of "add" calls to add a bunch of LLVM
|
||||||
|
|
@ -149,8 +167,11 @@ Once the pass manager is set up, we need to make use of it. We do this
|
||||||
by running it after our newly created function is constructed (in
|
by running it after our newly created function is constructed (in
|
||||||
``FunctionNode.CodeGen``), but before it is returned to the client:
|
``FunctionNode.CodeGen``), but before it is returned to the client:
|
||||||
|
|
||||||
{% highlight python %} return\_value = self.body.CodeGen()
|
|
||||||
g\_llvm\_builder.ret(return\_value)
|
.. code-block:: python
|
||||||
|
|
||||||
|
return_value = self.body.CodeGen()
|
||||||
|
g_llvm_builder.ret(return_value)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -160,17 +181,24 @@ g\_llvm\_builder.ret(return\_value)
|
||||||
# Optimize the function.
|
# Optimize the function.
|
||||||
g_llvm_pass_manager.run(function)
|
g_llvm_pass_manager.run(function)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
As you can see, this is pretty straightforward. The
|
As you can see, this is pretty straightforward. The
|
||||||
``FunctionPassManager`` optimizes and updates the LLVM Function in
|
``FunctionPassManager`` optimizes and updates the LLVM Function in
|
||||||
place, improving (hopefully) its body. With this in place, we can try
|
place, improving (hopefully) its body. With this in place, we can try
|
||||||
our test above again:
|
our test above again:
|
||||||
|
|
||||||
{% highlight bash %} ready> def test(x) (1+2+x)\*(x+(1+2)) Read a
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> def test(x) (1+2+x)\*(x+(1+2)) Read a
|
||||||
function definition: define double @test(double %x) { entry: %addtmp =
|
function definition: define double @test(double %x) { entry: %addtmp =
|
||||||
fadd double %x, 3.000000e+00 ; [#uses=2] %multmp = fmul double %addtmp,
|
fadd double %x, 3.000000e+00 ; [#uses=2] %multmp = fmul double %addtmp,
|
||||||
%addtmp ; [#uses=1] ret double %multmp } {% endhighlight %}
|
%addtmp ; [#uses=1] ret double %multmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
As expected, we now get our nicely optimized code, saving a floating
|
As expected, we now get our nicely optimized code, saving a floating
|
||||||
point add instruction from every execution of this function.
|
point add instruction from every execution of this function.
|
||||||
|
|
@ -208,8 +236,13 @@ be able to call it from the command line.
|
||||||
In order to do this, we first declare and initialize the JIT. This is
|
In order to do this, we first declare and initialize the JIT. This is
|
||||||
done by adding and initializing a global variable:
|
done by adding and initializing a global variable:
|
||||||
|
|
||||||
{% highlight python %} # The LLVM execution engine. g\_llvm\_executor =
|
|
||||||
ExecutionEngine.new(g\_llvm\_module) {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
# The LLVM execution engine. g_llvm_executor =
|
||||||
|
ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This creates an abstract "Execution Engine" which can be either a JIT
|
This creates an abstract "Execution Engine" which can be either a JIT
|
||||||
compiler or the LLVM interpreter. LLVM will automatically pick a JIT
|
compiler or the LLVM interpreter. LLVM will automatically pick a JIT
|
||||||
|
|
@ -222,10 +255,13 @@ compiled function and get its return value. In our case, this means that
|
||||||
we can change the code that parses a top-level expression to look like
|
we can change the code that parses a top-level expression to look like
|
||||||
this:
|
this:
|
||||||
|
|
||||||
{% highlight python %} def HandleTopLevelExpression(self): try: function
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def HandleTopLevelExpression(self): try: function
|
||||||
= self.ParseTopLevelExpr().CodeGen() result =
|
= self.ParseTopLevelExpr().CodeGen() result =
|
||||||
g\_llvm\_executor.run\_function(function, []) print 'Evaluated to:',
|
g_llvm_executor.run_function(function, []) print 'Evaluated to:',
|
||||||
result.as\_real(Type.double()) except Exception, e: print 'Error:', e
|
result.as_real(Type.double()) except Exception, e: print 'Error:', e
|
||||||
try: self.Next() # Skip for error recovery. except: pass {% endhighlight
|
try: self.Next() # Skip for error recovery. except: pass {% endhighlight
|
||||||
%}
|
%}
|
||||||
|
|
||||||
|
|
@ -237,14 +273,19 @@ With just these two changes, lets see how Kaleidoscope works now!
|
||||||
{% highlight python %} ready> 4+5 Read a top level expression: define
|
{% highlight python %} ready> 4+5 Read a top level expression: define
|
||||||
double @0() { entry: ret double 9.000000e+00 }
|
double @0() { entry: ret double 9.000000e+00 }
|
||||||
|
|
||||||
Evaluated to: 9.0 {% endhighlight %}
|
Evaluated to: 9.0
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Well this looks like it is basically working. The dump of the function
|
Well this looks like it is basically working. The dump of the function
|
||||||
shows the "no argument function that always returns double" that we
|
shows the "no argument function that always returns double" that we
|
||||||
synthesize for each top-level expression that is typed in. This
|
synthesize for each top-level expression that is typed in. This
|
||||||
demonstrates very basic functionality, but can we do more?
|
demonstrates very basic functionality, but can we do more?
|
||||||
|
|
||||||
{% highlight python %} ready> def testfunc(x y) x + y\*2 Read a function
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
ready> def testfunc(x y) x + y\*2 Read a function
|
||||||
definition: define double @testfunc(double %x, double %y) { entry:
|
definition: define double @testfunc(double %x, double %y) { entry:
|
||||||
%multmp = fmul double %y, 2.000000e+00 ; [#uses=1] %addtmp = fadd double
|
%multmp = fmul double %y, 2.000000e+00 ; [#uses=1] %addtmp = fadd double
|
||||||
%multmp, %x ; [#uses=1] ret double %addtmp }
|
%multmp, %x ; [#uses=1] ret double %addtmp }
|
||||||
|
|
@ -253,7 +294,9 @@ ready> testfunc(4, 10) Read a top level expression: define double @0() {
|
||||||
entry: %calltmp = call double @testfunc(double 4.000000e+00, double
|
entry: %calltmp = call double @testfunc(double 4.000000e+00, double
|
||||||
1.000000e+01) ; [#uses=1] ret double %calltmp }
|
1.000000e+01) ; [#uses=1] ret double %calltmp }
|
||||||
|
|
||||||
*Evaluated to: 24.0* {% endhighlight %}
|
*Evaluated to: 24.0*
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This illustrates that we can now call user code, but there is something
|
This illustrates that we can now call user code, but there is something
|
||||||
a bit subtle going on here. Note that we only invoke the JIT on the
|
a bit subtle going on here. Note that we only invoke the JIT on the
|
||||||
|
|
@ -269,7 +312,10 @@ etc. However, even with this simple code, we get some surprisingly
|
||||||
powerful capabilities - check this out (I removed the dump of the
|
powerful capabilities - check this out (I removed the dump of the
|
||||||
anonymous functions, you should get the idea by now :) :
|
anonymous functions, you should get the idea by now :) :
|
||||||
|
|
||||||
{% highlight bash %} ready> extern sin(x) Read an extern: declare double
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> extern sin(x) Read an extern: declare double
|
||||||
@sin(double)
|
@sin(double)
|
||||||
|
|
||||||
ready> extern cos(x) Read an extern: declare double @cos(double)
|
ready> extern cos(x) Read an extern: declare double @cos(double)
|
||||||
|
|
@ -285,7 +331,9 @@ double @cos(double %x) ; [#uses=1] %multmp4 = fmul double %calltmp2,
|
||||||
%calltmp3 ; [#uses=1] %addtmp = fadd double %multmp, %multmp4 ;
|
%calltmp3 ; [#uses=1] %addtmp = fadd double %multmp, %multmp4 ;
|
||||||
[#uses=1] ret double %addtmp }
|
[#uses=1] ret double %addtmp }
|
||||||
|
|
||||||
ready> foo(4.0) *Evaluated to: 1.000000* {% endhighlight %}
|
ready> foo(4.0) *Evaluated to: 1.000000*
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Whoa, how does the JIT know about sin and cos? The answer is
|
Whoa, how does the JIT know about sin and cos? The answer is
|
||||||
surprisingly simple: in this example, the JIT started execution of a
|
surprisingly simple: in this example, the JIT started execution of a
|
||||||
|
|
@ -301,7 +349,10 @@ One interesting application of this is that we can now extend the
|
||||||
language by writing arbitrary C++ code to implement operations. For
|
language by writing arbitrary C++ code to implement operations. For
|
||||||
example, we can create a C file with the following simple function:
|
example, we can create a C file with the following simple function:
|
||||||
|
|
||||||
{% highlight c %} #include
|
|
||||||
|
.. code-block:: c
|
||||||
|
|
||||||
|
#include
|
||||||
|
|
||||||
double putchard(double x) { putchar((char)x); return 0; } {%
|
double putchard(double x) { putchar((char)x); return 0; } {%
|
||||||
endhighlight %}
|
endhighlight %}
|
||||||
|
|
@ -316,12 +367,14 @@ Now we can load this library into the Python process using
|
||||||
to produce simple output to the console:
|
to produce simple output to the console:
|
||||||
|
|
||||||
{% highlight python %} >>> import llvm.core >>>
|
{% highlight python %} >>> import llvm.core >>>
|
||||||
llvm.core.load\_library\_permanently('/home/max/llvmpy-tutorial/putchard.so')
|
llvm.core.load_library_permanently('/home/max/llvmpy-tutorial/putchard.so')
|
||||||
>>> import kaleidoscope >>> kaleidoscope.main() ready> extern
|
>>> import kaleidoscope >>> kaleidoscope.main() ready> extern
|
||||||
putchard(x) Read an extern: declare double @putchard(double)
|
putchard(x) Read an extern: declare double @putchard(double)
|
||||||
|
|
||||||
ready> putchard(65) + putchard(66) + putchard(67) + putchard(10) *ABC*
|
ready> putchard(65) + putchard(66) + putchard(67) + putchard(10) *ABC*
|
||||||
Evaluated to: 0.0 {% endhighlight %}
|
Evaluated to: 0.0
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Similar code could be used to implement file I/O, console input, and
|
Similar code could be used to implement file I/O, console input, and
|
||||||
many other capabilities in Kaleidoscope.
|
many other capabilities in Kaleidoscope.
|
||||||
|
|
@ -341,51 +394,40 @@ Full Code Listing # {#code}
|
||||||
Here is the complete code listing for our running example, enhanced with
|
Here is the complete code listing for our running example, enhanced with
|
||||||
the LLVM JIT and optimizer:
|
the LLVM JIT and optimizer:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
import re from llvm.core import Module, Constant, Type, Function,
|
import re from llvm.core import Module, Constant, Type, Function,
|
||||||
Builder, FCMP\_ULT from llvm.ee import ExecutionEngine, TargetData from
|
Builder, FCMP_ULT from llvm.ee import ExecutionEngine, TargetData from
|
||||||
llvm.passes import FunctionPassManager from llvm.passes import
|
llvm.passes import FunctionPassManager from llvm.passes import
|
||||||
(PASS\_INSTRUCTION\_COMBINING, PASS\_REASSOCIATE, PASS\_GVN,
|
(PASS_INSTRUCTION_COMBINING, PASS_REASSOCIATE, PASS_GVN,
|
||||||
PASS\_CFG\_SIMPLIFICATION)
|
PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
Globals
|
Globals
|
||||||
-------
|
-------
|
||||||
|
|
||||||
The LLVM module, which holds all the IR code.
|
# The LLVM module, which holds all the IR code.
|
||||||
=============================================
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
|
g_llvm_builder = None
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
=========================================================================
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
|
|
||||||
g\_llvm\_builder = None
|
# The function optimization passes manager.
|
||||||
|
g_llvm_pass_manager = FunctionPassManager.new(g_llvm_module)
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
# The LLVM execution engine.
|
||||||
==============================================================================
|
g_llvm_executor = ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
and what their LLVM representation is.
|
|
||||||
======================================
|
|
||||||
|
|
||||||
g\_named\_values = {}
|
|
||||||
|
|
||||||
The function optimization passes manager.
|
|
||||||
=========================================
|
|
||||||
|
|
||||||
g\_llvm\_pass\_manager = FunctionPassManager.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
The LLVM execution engine.
|
|
||||||
==========================
|
|
||||||
|
|
||||||
g\_llvm\_executor = ExecutionEngine.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
Lexer
|
Lexer
|
||||||
-----
|
-----
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
# The lexer yields one of these types for each token.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class EOFToken(object): pass
|
class EOFToken(object): pass
|
||||||
|
|
||||||
class DefToken(object): pass
|
class DefToken(object): pass
|
||||||
|
|
@ -403,11 +445,9 @@ char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
== other
|
== other
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
=============================================================
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
string[0].isspace(): string = string[1:] continue
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
@ -447,34 +487,26 @@ yield EOFToken()
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
# Expression class for numeric literals like "1.0".
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode):
|
class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
def **init**\ (self, value): self.value = value
|
||||||
|
|
||||||
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
# Expression class for referencing a variable, like "a".
|
||||||
======================================================
|
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
def CodeGen(self): if self.name in g\_named\_values: return
|
def CodeGen(self): if self.name in g_named_values: return
|
||||||
g\_named\_values[self.name] else: raise RuntimeError('Unknown variable
|
g_named_values[self.name] else: raise RuntimeError('Unknown variable
|
||||||
name: ' + self.name)
|
name: ' + self.name)
|
||||||
|
|
||||||
Expression class for a binary operator.
|
# Expression class for a binary operator.
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode):
|
class BinaryOperatorExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, operator, left, right): self.operator = operator
|
def **init**\ (self, operator, left, right): self.operator = operator
|
||||||
|
|
@ -498,16 +530,14 @@ self.right.CodeGen()
|
||||||
else:
|
else:
|
||||||
raise RuntimeError('Unknown binary operator.')
|
raise RuntimeError('Unknown binary operator.')
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode):
|
class CallExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, callee, args): self.callee = callee self.args =
|
def **init**\ (self, callee, args): self.callee = callee self.args =
|
||||||
args
|
args
|
||||||
|
|
||||||
def CodeGen(self): # Look up the name in the global module table. callee
|
def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
= g\_llvm\_module.get\_function\_named(self.callee)
|
= g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -519,21 +549,15 @@ def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
==============================================================================
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes).
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
|
||||||
============================================================================
|
|
||||||
|
|
||||||
takes).
|
|
||||||
=======
|
|
||||||
|
|
||||||
class PrototypeNode(object):
|
class PrototypeNode(object):
|
||||||
|
|
||||||
def **init**\ (self, name, args): self.name = name self.args = args
|
def **init**\ (self, name, args): self.name = name self.args = args
|
||||||
|
|
||||||
def CodeGen(self): # Make the function type, eg. double(double,double).
|
def CodeGen(self): # Make the function type, eg. double(double,double).
|
||||||
funct\_type = Type.function( Type.double(), [Type.double()] \*
|
funct_type = Type.function( Type.double(), [Type.double()] \*
|
||||||
len(self.args), False)
|
len(self.args), False)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -563,15 +587,13 @@ len(self.args), False)
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
This class represents a function definition itself.
|
# This class represents a function definition itself.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class FunctionNode(object):
|
class FunctionNode(object):
|
||||||
|
|
||||||
def **init**\ (self, prototype, body): self.prototype = prototype
|
def **init**\ (self, prototype, body): self.prototype = prototype
|
||||||
self.body = body
|
self.body = body
|
||||||
|
|
||||||
def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
def CodeGen(self): # Clear scope. g_named_values.clear()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -604,8 +626,8 @@ Parser
|
||||||
|
|
||||||
class Parser(object):
|
class Parser(object):
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
|
|
@ -615,10 +637,10 @@ self.current = self.tokens.next()
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
isinstance(self.current, CharacterToken): return
|
isinstance(self.current, CharacterToken): return
|
||||||
self.binop\_precedence.get(self.current.char, -1) else: return -1
|
self.binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
self.Next() # eat identifier.
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -666,7 +688,7 @@ return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
||||||
when expecting an expression.')
|
when expecting an expression.')
|
||||||
|
|
||||||
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -737,8 +759,8 @@ def HandleExtern(self): self.Handle(self.ParseExtern, 'Read an extern:')
|
||||||
|
|
||||||
def HandleTopLevelExpression(self): try: function =
|
def HandleTopLevelExpression(self): try: function =
|
||||||
self.ParseTopLevelExpr().CodeGen() result =
|
self.ParseTopLevelExpr().CodeGen() result =
|
||||||
g\_llvm\_executor.run\_function(function, []) print 'Evaluated to:',
|
g_llvm_executor.run_function(function, []) print 'Evaluated to:',
|
||||||
result.as\_real(Type.double()) except Exception, e: print 'Error:', e
|
result.as_real(Type.double()) except Exception, e: print 'Error:', e
|
||||||
try: self.Next() # Skip for error recovery. except: pass
|
try: self.Next() # Skip for error recovery. except: pass
|
||||||
|
|
||||||
def Handle(self, function, message): try: print message,
|
def Handle(self, function, message): try: print message,
|
||||||
|
|
@ -750,22 +772,22 @@ Main driver code.
|
||||||
|
|
||||||
def main(): # Set up the optimizer pipeline. Start with registering info
|
def main(): # Set up the optimizer pipeline. Start with registering info
|
||||||
about how the # target lays out data structures.
|
about how the # target lays out data structures.
|
||||||
g\_llvm\_pass\_manager.add(g\_llvm\_executor.target\_data) # Do simple
|
g_llvm_pass_manager.add(g_llvm_executor.target_data) # Do simple
|
||||||
"peephole" optimizations and bit-twiddling optzns.
|
"peephole" optimizations and bit-twiddling optzns.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_INSTRUCTION\_COMBINING) # Reassociate
|
g_llvm_pass_manager.add(PASS_INSTRUCTION_COMBINING) # Reassociate
|
||||||
expressions. g\_llvm\_pass\_manager.add(PASS\_REASSOCIATE) # Eliminate
|
expressions. g_llvm_pass_manager.add(PASS_REASSOCIATE) # Eliminate
|
||||||
Common SubExpressions. g\_llvm\_pass\_manager.add(PASS\_GVN) # Simplify
|
Common SubExpressions. g_llvm_pass_manager.add(PASS_GVN) # Simplify
|
||||||
the control flow graph (deleting unreachable blocks, etc).
|
the control flow graph (deleting unreachable blocks, etc).
|
||||||
g\_llvm\_pass\_manager.add(PASS\_CFG\_SIMPLIFICATION)
|
g_llvm_pass_manager.add(PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
g\_llvm\_pass\_manager.initialize()
|
g_llvm_pass_manager.initialize()
|
||||||
|
|
||||||
# Install standard binary operators. # 1 is lowest possible precedence.
|
# Install standard binary operators. # 1 is lowest possible precedence.
|
||||||
40 is the highest. operator\_precedence = { '<': 10, '+': 20, '-': 20,
|
40 is the highest. operator_precedence = { '<': 10, '+': 20, '-': 20,
|
||||||
'\*': 40 }
|
'\*': 40 }
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: break
|
= raw_input() except KeyboardInterrupt: break
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -781,10 +803,6 @@ g\_llvm\_pass\_manager.initialize()
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
# Print out all of the generated code. print '', g\_llvm\_module
|
# Print out all of the generated code. print '', g_llvm_module
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Extending the language: control flow <PythonLangImpl5.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -34,8 +34,13 @@ Before we get going on "how" we add this extension, lets talk about
|
||||||
"what" we want. The basic idea is that we want to be able to write this
|
"what" we want. The basic idea is that we want to be able to write this
|
||||||
sort of thing:
|
sort of thing:
|
||||||
|
|
||||||
{% highlight python %} def fib(x) if x < 3 then 1 else fib(x-1) +
|
|
||||||
fib(x-2) {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
def fib(x) if x < 3 then 1 else fib(x-1) +
|
||||||
|
fib(x-2)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In Kaleidoscope, every construct is an expression: there are no
|
In Kaleidoscope, every construct is an expression: there are no
|
||||||
statements. As such, the if/then/else expression needs to return a value
|
statements. As such, the if/then/else expression needs to return a value
|
||||||
|
|
@ -61,31 +66,46 @@ Lexer Extensions for If/Then/Else ## {#iflexer}
|
||||||
The lexer extensions are straightforward. First we add new token classes
|
The lexer extensions are straightforward. First we add new token classes
|
||||||
for the relevant tokens:
|
for the relevant tokens:
|
||||||
|
|
||||||
{% highlight python %} class IfToken(object): pass class
|
|
||||||
ThenToken(object): pass class ElseToken(object): pass {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
class IfToken(object): pass class
|
||||||
|
ThenToken(object): pass class ElseToken(object): pass
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Once we have that, we recognize the new keywords in the lexer. This is
|
Once we have that, we recognize the new keywords in the lexer. This is
|
||||||
pretty simple stuff:
|
pretty simple stuff:
|
||||||
|
|
||||||
{% highlight python %} ... if identifier == 'def': yield DefToken() elif
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
... if identifier == 'def': yield DefToken() elif
|
||||||
identifier == 'extern': yield ExternToken() elif identifier == 'if':
|
identifier == 'extern': yield ExternToken() elif identifier == 'if':
|
||||||
yield IfToken() elif identifier == 'then': yield ThenToken() elif
|
yield IfToken() elif identifier == 'then': yield ThenToken() elif
|
||||||
identifier == 'else': yield ElseToken() else: yield
|
identifier == 'else': yield ElseToken() else: yield
|
||||||
IdentifierToken(identifier) {% endhighlight %}
|
IdentifierToken(identifier)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
AST Extensions for If/Then/Else ## {#ifast}
|
AST Extensions for If/Then/Else ## {#ifast}
|
||||||
-------------------------------------------
|
-------------------------------------------
|
||||||
|
|
||||||
To represent the new expression we add a new AST node for it:
|
To represent the new expression we add a new AST node for it:
|
||||||
|
|
||||||
{% highlight python %} # Expression class for if/then/else. class
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Expression class for if/then/else. class
|
||||||
IfExpressionNode(ExpressionNode):
|
IfExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, condition, then\_branch, else\_branch):
|
def **init**\ (self, condition, then_branch, else_branch):
|
||||||
self.condition = condition self.then\_branch = then\_branch
|
self.condition = condition self.then_branch = then_branch
|
||||||
self.else\_branch = else\_branch
|
self.else_branch = else_branch
|
||||||
|
|
||||||
|
def CodeGen(self): ...
|
||||||
|
|
||||||
|
|
||||||
def CodeGen(self): ... {% endhighlight %}
|
|
||||||
|
|
||||||
The AST node just has pointers to the various subexpressions.
|
The AST node just has pointers to the various subexpressions.
|
||||||
|
|
||||||
|
|
@ -96,7 +116,10 @@ Now that we have the relevant tokens coming from the lexer and we have
|
||||||
the AST node to build, our parsing logic is relatively straightforward.
|
the AST node to build, our parsing logic is relatively straightforward.
|
||||||
First we define a new parsing function:
|
First we define a new parsing function:
|
||||||
|
|
||||||
{% highlight python %} # ifexpr ::= 'if' expression 'then' expression
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# ifexpr ::= 'if' expression 'then' expression
|
||||||
'else' expression def ParseIfExpr(self): self.Next() # eat the if.
|
'else' expression def ParseIfExpr(self): self.Next() # eat the if.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -118,17 +141,24 @@ First we define a new parsing function:
|
||||||
|
|
||||||
return IfExpressionNode(condition, then_branch, else_branch)
|
return IfExpressionNode(condition, then_branch, else_branch)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Next we hook it up as a primary expression:
|
Next we hook it up as a primary expression:
|
||||||
|
|
||||||
{% highlight python %} def ParsePrimary(self): if
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def ParsePrimary(self): if
|
||||||
isinstance(self.current, IdentifierToken): return
|
isinstance(self.current, IdentifierToken): return
|
||||||
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
||||||
return self.ParseNumberExpr(); elif isinstance(self.current, IfToken):
|
return self.ParseNumberExpr(); elif isinstance(self.current, IfToken):
|
||||||
return self.ParseIfExpr() elif self.current == CharacterToken('('):
|
return self.ParseIfExpr() elif self.current == CharacterToken('('):
|
||||||
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
||||||
when expecting an expression.') {% endhighlight %}
|
when expecting an expression.')
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
LLVM IR for If/Then/Else ## {#ifir}
|
LLVM IR for If/Then/Else ## {#ifir}
|
||||||
-----------------------------------
|
-----------------------------------
|
||||||
|
|
@ -142,19 +172,29 @@ described in previous chapters.
|
||||||
To motivate the code we want to produce, lets take a look at a simple
|
To motivate the code we want to produce, lets take a look at a simple
|
||||||
example. Consider:
|
example. Consider:
|
||||||
|
|
||||||
{% highlight python %} extern foo(); extern bar(); def baz(x) if x then
|
|
||||||
foo() else bar(); {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
extern foo(); extern bar(); def baz(x) if x then
|
||||||
|
foo() else bar();
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
If you disable optimizations, the code you'll (soon) get from
|
If you disable optimizations, the code you'll (soon) get from
|
||||||
Kaleidoscope looks something like this:
|
Kaleidoscope looks something like this:
|
||||||
|
|
||||||
{% highlight llvm %} declare double @foo() declare double @bar() define
|
|
||||||
|
.. code-block:: llvm
|
||||||
|
|
||||||
|
declare double @foo() declare double @bar() define
|
||||||
double @baz(double %x) { entry: %ifcond = fcmp one double %x,
|
double @baz(double %x) { entry: %ifcond = fcmp one double %x,
|
||||||
0.000000e+00 br i1 %ifcond, label %then, label %else then: ; preds =
|
0.000000e+00 br i1 %ifcond, label %then, label %else then: ; preds =
|
||||||
%entry %calltmp1 = call double @bar() else: ; preds = %entry %calltmp1 =
|
%entry %calltmp1 = call double @bar() else: ; preds = %entry %calltmp1 =
|
||||||
call double @bar() br label %ifcont ifcont: ; preds = %else, %then
|
call double @bar() br label %ifcont ifcont: ; preds = %else, %then
|
||||||
%iftmp = phi double [ %calltmp, %then ], [ %calltmp1, %else ] ret double
|
%iftmp = phi double [ %calltmp, %then ], [ %calltmp1, %else ] ret double
|
||||||
%iftmp } {% endhighlight %}
|
%iftmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
To visualize the control flow graph, you can use a nifty feature of the
|
To visualize the control flow graph, you can use a nifty feature of the
|
||||||
LLVM `opt <http://llvm.org/cmds/opt.html>`_ tool. If you put this LLVM
|
LLVM `opt <http://llvm.org/cmds/opt.html>`_ tool. If you put this LLVM
|
||||||
|
|
@ -226,7 +266,10 @@ Code Generation for If/Then/Else ## {#ifcodegen}
|
||||||
In order to generate code for this, we implement the ``Codegen`` method
|
In order to generate code for this, we implement the ``Codegen`` method
|
||||||
for ``IfExpressionNode``:
|
for ``IfExpressionNode``:
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): condition =
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): condition =
|
||||||
self.condition.CodeGen()
|
self.condition.CodeGen()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -235,13 +278,18 @@ self.condition.CodeGen()
|
||||||
condition_bool = g_llvm_builder.fcmp(
|
condition_bool = g_llvm_builder.fcmp(
|
||||||
FCMP_ONE, condition, Constant.real(Type.double(), 0), 'ifcond')
|
FCMP_ONE, condition, Constant.real(Type.double(), 0), 'ifcond')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code is straightforward and similar to what we saw before. We emit
|
This code is straightforward and similar to what we saw before. We emit
|
||||||
the expression for the condition, then compare that value to zero to get
|
the expression for the condition, then compare that value to zero to get
|
||||||
a truth value as a 1-bit (bool) value.
|
a truth value as a 1-bit (bool) value.
|
||||||
|
|
||||||
{% highlight python %} function = g\_llvm\_builder.basic\_block.function
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
function = g_llvm_builder.basic_block.function
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -253,7 +301,9 @@ a truth value as a 1-bit (bool) value.
|
||||||
|
|
||||||
g_llvm_builder.cbranch(condition_bool, then_block, else_block)
|
g_llvm_builder.cbranch(condition_bool, then_block, else_block)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code creates the basic blocks that are related to the if/then/else
|
This code creates the basic blocks that are related to the if/then/else
|
||||||
statement, and correspond directly to the blocks in the example above.
|
statement, and correspond directly to the blocks in the example above.
|
||||||
|
|
@ -268,9 +318,12 @@ can emit the conditional branch that chooses between them. Note that
|
||||||
creating new blocks does not implicitly affect the Builder, so it is
|
creating new blocks does not implicitly affect the Builder, so it is
|
||||||
still inserting into the block that the condition went into.
|
still inserting into the block that the condition went into.
|
||||||
|
|
||||||
{% highlight python %} # Emit then value.
|
|
||||||
g\_llvm\_builder.position\_at\_end(then\_block) then\_value =
|
.. code-block:: python
|
||||||
self.then\_branch.CodeGen() g\_llvm\_builder.branch(merge\_block)
|
|
||||||
|
# Emit then value.
|
||||||
|
g_llvm_builder.position_at_end(then_block) then_value =
|
||||||
|
self.then_branch.CodeGen() g_llvm_builder.branch(merge_block)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -278,7 +331,9 @@ self.then\_branch.CodeGen() g\_llvm\_builder.branch(merge\_block)
|
||||||
# PHI node.
|
# PHI node.
|
||||||
then_block = g_llvm_builder.basic_block
|
then_block = g_llvm_builder.basic_block
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
After the conditional branch is inserted, we move the builder to start
|
After the conditional branch is inserted, we move the builder to start
|
||||||
inserting into the "then" block. Strictly speaking, this call moves the
|
inserting into the "then" block. Strictly speaking, this call moves the
|
||||||
|
|
@ -310,9 +365,12 @@ expression. Because calling Codegen recursively could arbitrarily change
|
||||||
the notion of the current block, we are required to get an up-to-date
|
the notion of the current block, we are required to get an up-to-date
|
||||||
value for code that will set up the Phi node.
|
value for code that will set up the Phi node.
|
||||||
|
|
||||||
{% highlight python %} # Emit else block.
|
|
||||||
g\_llvm\_builder.position\_at\_end(else\_block) else\_value =
|
.. code-block:: python
|
||||||
self.else\_branch.CodeGen() g\_llvm\_builder.branch(merge\_block)
|
|
||||||
|
# Emit else block.
|
||||||
|
g_llvm_builder.position_at_end(else_block) else_value =
|
||||||
|
self.else_branch.CodeGen() g_llvm_builder.branch(merge_block)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -320,7 +378,9 @@ self.else\_branch.CodeGen() g\_llvm\_builder.branch(merge\_block)
|
||||||
# PHI node.
|
# PHI node.
|
||||||
else_block = g_llvm_builder.basic_block
|
else_block = g_llvm_builder.basic_block
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Code generation for the 'else' block is basically identical to codegen
|
Code generation for the 'else' block is basically identical to codegen
|
||||||
for the 'then' block. The only significant difference is the first line,
|
for the 'then' block. The only significant difference is the first line,
|
||||||
|
|
@ -329,17 +389,22 @@ which adds the 'else' block to the function. Recall previously that the
|
||||||
'then' and 'else' blocks are emitted, we can finish up with the merge
|
'then' and 'else' blocks are emitted, we can finish up with the merge
|
||||||
code:
|
code:
|
||||||
|
|
||||||
{% highlight python %} # Emit merge block.
|
|
||||||
g\_llvm\_builder.position\_at\_end(merge\_block) phi =
|
.. code-block:: python
|
||||||
g\_llvm\_builder.phi(Type.double(), 'iftmp')
|
|
||||||
phi.add\_incoming(then\_value, then\_block)
|
# Emit merge block.
|
||||||
phi.add\_incoming(else\_value, else\_block)
|
g_llvm_builder.position_at_end(merge_block) phi =
|
||||||
|
g_llvm_builder.phi(Type.double(), 'iftmp')
|
||||||
|
phi.add_incoming(then_value, then_block)
|
||||||
|
phi.add_incoming(else_value, else_block)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
return phi
|
return phi
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The first line changes the insertion point so that newly created code
|
The first line changes the insertion point so that newly created code
|
||||||
will go into the "merge" block. Once that is done, we need to create the
|
will go into the "merge" block. Once that is done, we need to create the
|
||||||
|
|
@ -365,7 +430,10 @@ Now that we know how to add basic control flow constructs to the
|
||||||
language, we have the tools to add more powerful things. Lets add
|
language, we have the tools to add more powerful things. Lets add
|
||||||
something more aggressive, a 'for' expression:
|
something more aggressive, a 'for' expression:
|
||||||
|
|
||||||
{% highlight python %} extern putchard(char) def printstar(n) for i = 1,
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
extern putchard(char) def printstar(n) for i = 1,
|
||||||
i < n, 1.0 in putchard(42) # ascii 42 = '\*'
|
i < n, 1.0 in putchard(42) # ascii 42 = '\*'
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -373,7 +441,9 @@ i < n, 1.0 in putchard(42) # ascii 42 = '\*'
|
||||||
# print 100 '*' characters
|
# print 100 '*' characters
|
||||||
printstar(100)
|
printstar(100)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This expression defines a new variable (``i`` in this case) which
|
This expression defines a new variable (``i`` in this case) which
|
||||||
iterates from a starting value, while the condition (``i < n`` in this
|
iterates from a starting value, while the condition (``i < n`` in this
|
||||||
|
|
@ -391,7 +461,10 @@ Lexer Extensions for the 'for' Loop ## {#forlexer}
|
||||||
|
|
||||||
The lexer extensions are the same sort of thing as for if/then/else:
|
The lexer extensions are the same sort of thing as for if/then/else:
|
||||||
|
|
||||||
{% highlight python %} ...
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
...
|
||||||
|
|
||||||
class ThenToken(object): pass class ElseToken(object): pass class
|
class ThenToken(object): pass class ElseToken(object): pass class
|
||||||
ForToken(object): pass class InToken(object): pass
|
ForToken(object): pass class InToken(object): pass
|
||||||
|
|
@ -413,7 +486,9 @@ def Tokenize(string):
|
||||||
else:
|
else:
|
||||||
yield IdentifierToken(identifier)
|
yield IdentifierToken(identifier)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
AST Extensions for the 'for' Loop ## {#forast}
|
AST Extensions for the 'for' Loop ## {#forast}
|
||||||
----------------------------------------------
|
----------------------------------------------
|
||||||
|
|
@ -421,14 +496,19 @@ AST Extensions for the 'for' Loop ## {#forast}
|
||||||
The AST node is just as simple. It basically boils down to capturing the
|
The AST node is just as simple. It basically boils down to capturing the
|
||||||
variable name and the constituent expressions in the node.
|
variable name and the constituent expressions in the node.
|
||||||
|
|
||||||
{% highlight python %} # Expression class for for/in. class
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Expression class for for/in. class
|
||||||
ForExpressionNode(ExpressionNode):
|
ForExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, loop\_variable, start, end, step, body):
|
def **init**\ (self, loop_variable, start, end, step, body):
|
||||||
self.loop\_variable = loop\_variable self.start = start self.end = end
|
self.loop_variable = loop_variable self.start = start self.end = end
|
||||||
self.step = step self.body = body
|
self.step = step self.body = body
|
||||||
|
|
||||||
def CodeGen(self): ... {% endhighlight %}
|
def CodeGen(self): ...
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Parser Extensions for the 'for' Loop ## {#forparser}
|
Parser Extensions for the 'for' Loop ## {#forparser}
|
||||||
----------------------------------------------------
|
----------------------------------------------------
|
||||||
|
|
@ -438,7 +518,10 @@ is handling of the optional step value. The parser code handles it by
|
||||||
checking to see if the second comma is present. If not, it sets the step
|
checking to see if the second comma is present. If not, it sets the step
|
||||||
value to null in the AST node:
|
value to null in the AST node:
|
||||||
|
|
||||||
{% highlight python %} # forexpr ::= 'for' identifier '=' expr ',' expr
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# forexpr ::= 'for' identifier '=' expr ',' expr
|
||||||
(',' expr)? 'in' expression def ParseForExpr(self): self.Next() # eat
|
(',' expr)? 'in' expression def ParseForExpr(self): self.Next() # eat
|
||||||
the for.
|
the for.
|
||||||
|
|
||||||
|
|
@ -477,7 +560,9 @@ the for.
|
||||||
|
|
||||||
return ForExpressionNode(loop_variable, start, end, step, body)
|
return ForExpressionNode(loop_variable, start, end, step, body)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
LLVM IR for the 'for' Loop ## {#forir}
|
LLVM IR for the 'for' Loop ## {#forir}
|
||||||
--------------------------------------
|
--------------------------------------
|
||||||
|
|
@ -486,7 +571,10 @@ Now we get to the good part: the LLVM IR we want to generate for this
|
||||||
thing. With the simple example above, we get this LLVM IR (note that
|
thing. With the simple example above, we get this LLVM IR (note that
|
||||||
this dump is generated with optimizations disabled for clarity):
|
this dump is generated with optimizations disabled for clarity):
|
||||||
|
|
||||||
{% highlight llvm %} declare double @putchard(double) define double
|
|
||||||
|
.. code-block:: llvm
|
||||||
|
|
||||||
|
declare double @putchard(double) define double
|
||||||
@printstar(double %n) { entry: ; initial value = 1.0 (inlined into phi)
|
@printstar(double %n) { entry: ; initial value = 1.0 (inlined into phi)
|
||||||
br label %loop loop: ; preds = %loop, %entry %i = phi double [
|
br label %loop loop: ; preds = %loop, %entry %i = phi double [
|
||||||
1.000000e+00, %entry ], [ %nextvar, %loop ] ; body %calltmp = call
|
1.000000e+00, %entry ], [ %nextvar, %loop ] ; body %calltmp = call
|
||||||
|
|
@ -495,7 +583,9 @@ double @putchard(double 4.200000e+01) ; increment %nextvar = fadd double
|
||||||
%booltmp = uitofp i1 %cmptmp to double %loopcond = fcmp one double
|
%booltmp = uitofp i1 %cmptmp to double %loopcond = fcmp one double
|
||||||
%booltmp, 0.000000e+00 br i1 %loopcond, label %loop, label %afterloop
|
%booltmp, 0.000000e+00 br i1 %loopcond, label %loop, label %afterloop
|
||||||
afterloop: ; preds = %loop ; loop always returns 0.0 ret double
|
afterloop: ; preds = %loop ; loop always returns 0.0 ret double
|
||||||
0.000000e+00 } {% endhighlight %}
|
0.000000e+00 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This loop contains all the same constructs we saw before: a phi node,
|
This loop contains all the same constructs we saw before: a phi node,
|
||||||
several expressions, and some basic blocks. Lets see how this fits
|
several expressions, and some basic blocks. Lets see how this fits
|
||||||
|
|
@ -507,8 +597,11 @@ Code Generation for the 'for' Loop ## {#forcodegen}
|
||||||
The first part of Codegen is very simple: we just output the start
|
The first part of Codegen is very simple: we just output the start
|
||||||
expression for the loop value:
|
expression for the loop value:
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): # Emit the start code first,
|
|
||||||
without 'variable' in scope. start\_value = self.start.CodeGen() {%
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): # Emit the start code first,
|
||||||
|
without 'variable' in scope. start_value = self.start.CodeGen() {%
|
||||||
endhighlight %}
|
endhighlight %}
|
||||||
|
|
||||||
With this out of the way, the next step is to set up the LLVM basic
|
With this out of the way, the next step is to set up the LLVM basic
|
||||||
|
|
@ -519,16 +612,18 @@ expression).
|
||||||
|
|
||||||
{% highlight python %} # Make the new basic block for the loop header,
|
{% highlight python %} # Make the new basic block for the loop header,
|
||||||
inserting after current # block. function =
|
inserting after current # block. function =
|
||||||
g\_llvm\_builder.basic\_block.function pre\_header\_block =
|
g_llvm_builder.basic_block.function pre_header_block =
|
||||||
g\_llvm\_builder.basic\_block loop\_block =
|
g_llvm_builder.basic_block loop_block =
|
||||||
function.append\_basic\_block('loop')
|
function.append_basic_block('loop')
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
# Insert an explicit fallthrough from the current block to the loop_block.
|
# Insert an explicit fallthrough from the current block to the loop_block.
|
||||||
g_llvm_builder.branch(loop_block)
|
g_llvm_builder.branch(loop_block)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code is similar to what we saw for if/then/else. Because we will
|
This code is similar to what we saw for if/then/else. Because we will
|
||||||
need it to create the Phi node, we remember the block that falls through
|
need it to create the Phi node, we remember the block that falls through
|
||||||
|
|
@ -536,8 +631,11 @@ into the loop. Once we have that, we create the actual block that starts
|
||||||
the loop and create an unconditional branch for the fall-through between
|
the loop and create an unconditional branch for the fall-through between
|
||||||
the two blocks.
|
the two blocks.
|
||||||
|
|
||||||
{% highlight python %} # Start insertion in loop\_block.
|
|
||||||
g\_llvm\_builder.position\_at\_end(loop\_block);
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Start insertion in loop_block.
|
||||||
|
g_llvm_builder.position_at_end(loop_block);
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -545,7 +643,9 @@ g\_llvm\_builder.position\_at\_end(loop\_block);
|
||||||
variable_phi = g_llvm_builder.phi(Type.double(), self.loop_variable)
|
variable_phi = g_llvm_builder.phi(Type.double(), self.loop_variable)
|
||||||
variable_phi.add_incoming(start_value, pre_header_block)
|
variable_phi.add_incoming(start_value, pre_header_block)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Now that the "pre\_header\_block" for the loop is set up, we switch to
|
Now that the "pre\_header\_block" for the loop is set up, we switch to
|
||||||
emitting code for the loop body. To begin with, we move the insertion
|
emitting code for the loop body. To begin with, we move the insertion
|
||||||
|
|
@ -554,11 +654,14 @@ already know the incoming value for the starting value, we add it to the
|
||||||
Phi node. Note that the Phi will eventually get a second value for the
|
Phi node. Note that the Phi will eventually get a second value for the
|
||||||
backedge, but we can't set it up yet (because it doesn't exist!).
|
backedge, but we can't set it up yet (because it doesn't exist!).
|
||||||
|
|
||||||
{% highlight python %} # Within the loop, the variable is defined equal
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Within the loop, the variable is defined equal
|
||||||
to the PHI node. If it # shadows an existing variable, we have to
|
to the PHI node. If it # shadows an existing variable, we have to
|
||||||
restore it, so save it now. old\_value =
|
restore it, so save it now. old_value =
|
||||||
g\_named\_values.get(self.loop\_variable, None)
|
g_named_values.get(self.loop_variable, None)
|
||||||
g\_named\_values[self.loop\_variable] = variable\_phi
|
g_named_values[self.loop_variable] = variable_phi
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -566,7 +669,9 @@ g\_named\_values[self.loop\_variable] = variable\_phi
|
||||||
# current BB. Note that we ignore the value computed by the body.
|
# current BB. Note that we ignore the value computed by the body.
|
||||||
self.body.CodeGen()
|
self.body.CodeGen()
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Now the code starts to get more interesting. Our 'for' loop introduces a
|
Now the code starts to get more interesting. Our 'for' loop introduces a
|
||||||
new variable to the symbol table. This means that our symbol table can
|
new variable to the symbol table. This means that our symbol table can
|
||||||
|
|
@ -585,33 +690,46 @@ recursively codegen's the body. This allows the body to use the loop
|
||||||
variable: any references to it will naturally find it in the symbol
|
variable: any references to it will naturally find it in the symbol
|
||||||
table.
|
table.
|
||||||
|
|
||||||
{% highlight python %} # Emit the step value. if self.step: step\_value
|
|
||||||
= self.step.CodeGen() else: # If not specified, use 1.0. step\_value =
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Emit the step value. if self.step: step_value
|
||||||
|
= self.step.CodeGen() else: # If not specified, use 1.0. step_value =
|
||||||
Constant.real(Type.double(), 1)
|
Constant.real(Type.double(), 1)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
next_value = g_llvm_builder.fadd(variable_phi, step_value, 'next')
|
next_value = g_llvm_builder.fadd(variable_phi, step_value, 'next')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Now that the body is emitted, we compute the next value of the iteration
|
Now that the body is emitted, we compute the next value of the iteration
|
||||||
variable by adding the step value, or 1.0 if it isn't present.
|
variable by adding the step value, or 1.0 if it isn't present.
|
||||||
``next_value`` will be the value of the loop variable on the next
|
``next_value`` will be the value of the loop variable on the next
|
||||||
iteration of the loop.
|
iteration of the loop.
|
||||||
|
|
||||||
{% highlight python %} # Compute the end condition and convert it to a
|
|
||||||
bool by comparing to 0.0. end\_condition = self.end.CodeGen()
|
.. code-block:: python
|
||||||
end\_condition\_bool = g\_llvm\_builder.fcmp( FCMP\_ONE, end\_condition,
|
|
||||||
Constant.real(Type.double(), 0), 'loopcond') {% endhighlight %}
|
# Compute the end condition and convert it to a
|
||||||
|
bool by comparing to 0.0. end_condition = self.end.CodeGen()
|
||||||
|
end_condition_bool = g_llvm_builder.fcmp( FCMP_ONE, end_condition,
|
||||||
|
Constant.real(Type.double(), 0), 'loopcond')
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Finally, we evaluate the exit value of the loop, to determine whether
|
Finally, we evaluate the exit value of the loop, to determine whether
|
||||||
the loop should exit. This mirrors the condition evaluation for the
|
the loop should exit. This mirrors the condition evaluation for the
|
||||||
if/then/else statement.
|
if/then/else statement.
|
||||||
|
|
||||||
{% highlight python %} # Create the "after loop" block and insert it.
|
|
||||||
loop\_end\_block = g\_llvm\_builder.basic\_block after\_block =
|
.. code-block:: python
|
||||||
function.append\_basic\_block('afterloop')
|
|
||||||
|
# Create the "after loop" block and insert it.
|
||||||
|
loop_end_block = g_llvm_builder.basic_block after_block =
|
||||||
|
function.append_basic_block('afterloop')
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -621,7 +739,9 @@ function.append\_basic\_block('afterloop')
|
||||||
# Any new code will be inserted in after_block.
|
# Any new code will be inserted in after_block.
|
||||||
g_llvm_builder.position_at_end(after_block)
|
g_llvm_builder.position_at_end(after_block)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
With the code for the body of the loop complete, we just need to finish
|
With the code for the body of the loop complete, we just need to finish
|
||||||
up the control flow for it. This code remembers the end block (for the
|
up the control flow for it. This code remembers the end block (for the
|
||||||
|
|
@ -631,8 +751,11 @@ chooses between executing the loop again and exiting the loop. Any
|
||||||
future code is emitted in the "afterloop" block, so it sets the
|
future code is emitted in the "afterloop" block, so it sets the
|
||||||
insertion position to it.
|
insertion position to it.
|
||||||
|
|
||||||
{% highlight python %} # Add a new entry to the PHI node for the
|
|
||||||
backedge. variable\_phi.add\_incoming(next\_value, loop\_end\_block)
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Add a new entry to the PHI node for the
|
||||||
|
backedge. variable_phi.add_incoming(next_value, loop_end_block)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -645,7 +768,9 @@ backedge. variable\_phi.add\_incoming(next\_value, loop\_end\_block)
|
||||||
# for expr always returns 0.0.
|
# for expr always returns 0.0.
|
||||||
return Constant.real(Type.double(), 0)
|
return Constant.real(Type.double(), 0)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The final code handles various cleanups: now that we have the
|
The final code handles various cleanups: now that we have the
|
||||||
"next\_value", we can add the incoming value to the loop PHI node. After
|
"next\_value", we can add the incoming value to the loop PHI node. After
|
||||||
|
|
@ -669,53 +794,42 @@ Full Code Listing # {#code}
|
||||||
Here is the complete code listing for our running example, enhanced with
|
Here is the complete code listing for our running example, enhanced with
|
||||||
the if/then/else and for expressions:
|
the if/then/else and for expressions:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
import re from llvm.core import Module, Constant, Type, Function,
|
import re from llvm.core import Module, Constant, Type, Function,
|
||||||
Builder from llvm.ee import ExecutionEngine, TargetData from llvm.passes
|
Builder from llvm.ee import ExecutionEngine, TargetData from llvm.passes
|
||||||
import FunctionPassManager
|
import FunctionPassManager
|
||||||
|
|
||||||
from llvm.core import FCMP\_ULT, FCMP\_ONE from llvm.passes import
|
from llvm.core import FCMP_ULT, FCMP_ONE from llvm.passes import
|
||||||
(PASS\_INSTRUCTION\_COMBINING, PASS\_REASSOCIATE, PASS\_GVN,
|
(PASS_INSTRUCTION_COMBINING, PASS_REASSOCIATE, PASS_GVN,
|
||||||
PASS\_CFG\_SIMPLIFICATION)
|
PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
Globals
|
Globals
|
||||||
-------
|
-------
|
||||||
|
|
||||||
The LLVM module, which holds all the IR code.
|
# The LLVM module, which holds all the IR code.
|
||||||
=============================================
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
|
g_llvm_builder = None
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
=========================================================================
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
|
|
||||||
g\_llvm\_builder = None
|
# The function optimization passes manager.
|
||||||
|
g_llvm_pass_manager = FunctionPassManager.new(g_llvm_module)
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
# The LLVM execution engine.
|
||||||
==============================================================================
|
g_llvm_executor = ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
and what their LLVM representation is.
|
|
||||||
======================================
|
|
||||||
|
|
||||||
g\_named\_values = {}
|
|
||||||
|
|
||||||
The function optimization passes manager.
|
|
||||||
=========================================
|
|
||||||
|
|
||||||
g\_llvm\_pass\_manager = FunctionPassManager.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
The LLVM execution engine.
|
|
||||||
==========================
|
|
||||||
|
|
||||||
g\_llvm\_executor = ExecutionEngine.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
Lexer
|
Lexer
|
||||||
-----
|
-----
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
# The lexer yields one of these types for each token.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class EOFToken(object): pass class DefToken(object): pass class
|
class EOFToken(object): pass class DefToken(object): pass class
|
||||||
ExternToken(object): pass class IfToken(object): pass class
|
ExternToken(object): pass class IfToken(object): pass class
|
||||||
ThenToken(object): pass class ElseToken(object): pass class
|
ThenToken(object): pass class ElseToken(object): pass class
|
||||||
|
|
@ -732,11 +846,9 @@ char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
== other
|
== other
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
=============================================================
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
string[0].isspace(): string = string[1:] continue
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
@ -786,34 +898,26 @@ yield EOFToken()
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
# Expression class for numeric literals like "1.0".
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode):
|
class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
def **init**\ (self, value): self.value = value
|
||||||
|
|
||||||
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
# Expression class for referencing a variable, like "a".
|
||||||
======================================================
|
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
def CodeGen(self): if self.name in g\_named\_values: return
|
def CodeGen(self): if self.name in g_named_values: return
|
||||||
g\_named\_values[self.name] else: raise RuntimeError('Unknown variable
|
g_named_values[self.name] else: raise RuntimeError('Unknown variable
|
||||||
name: ' + self.name)
|
name: ' + self.name)
|
||||||
|
|
||||||
Expression class for a binary operator.
|
# Expression class for a binary operator.
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode):
|
class BinaryOperatorExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, operator, left, right): self.operator = operator
|
def **init**\ (self, operator, left, right): self.operator = operator
|
||||||
|
|
@ -837,16 +941,14 @@ self.right.CodeGen()
|
||||||
else:
|
else:
|
||||||
raise RuntimeError('Unknown binary operator.')
|
raise RuntimeError('Unknown binary operator.')
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode):
|
class CallExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, callee, args): self.callee = callee self.args =
|
def **init**\ (self, callee, args): self.callee = callee self.args =
|
||||||
args
|
args
|
||||||
|
|
||||||
def CodeGen(self): # Look up the name in the global module table. callee
|
def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
= g\_llvm\_module.get\_function\_named(self.callee)
|
= g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -858,14 +960,12 @@ def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
Expression class for if/then/else.
|
# Expression class for if/then/else.
|
||||||
==================================
|
|
||||||
|
|
||||||
class IfExpressionNode(ExpressionNode):
|
class IfExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, condition, then\_branch, else\_branch):
|
def **init**\ (self, condition, then_branch, else_branch):
|
||||||
self.condition = condition self.then\_branch = then\_branch
|
self.condition = condition self.then_branch = then_branch
|
||||||
self.else\_branch = else\_branch
|
self.else_branch = else_branch
|
||||||
|
|
||||||
def CodeGen(self): condition = self.condition.CodeGen()
|
def CodeGen(self): condition = self.condition.CodeGen()
|
||||||
|
|
||||||
|
|
@ -911,13 +1011,11 @@ def CodeGen(self): condition = self.condition.CodeGen()
|
||||||
|
|
||||||
return phi
|
return phi
|
||||||
|
|
||||||
Expression class for for/in.
|
# Expression class for for/in.
|
||||||
============================
|
|
||||||
|
|
||||||
class ForExpressionNode(ExpressionNode):
|
class ForExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, loop\_variable, start, end, step, body):
|
def **init**\ (self, loop_variable, start, end, step, body):
|
||||||
self.loop\_variable = loop\_variable self.start = start self.end = end
|
self.loop_variable = loop_variable self.start = start self.end = end
|
||||||
self.step = step self.body = body
|
self.step = step self.body = body
|
||||||
|
|
||||||
def CodeGen(self): # Output this as: # ... # start = startexpr # goto
|
def CodeGen(self): # Output this as: # ... # start = startexpr # goto
|
||||||
|
|
@ -992,21 +1090,15 @@ endloop # outloop:
|
||||||
# for expr always returns 0.0.
|
# for expr always returns 0.0.
|
||||||
return Constant.real(Type.double(), 0)
|
return Constant.real(Type.double(), 0)
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
==============================================================================
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes).
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
|
||||||
============================================================================
|
|
||||||
|
|
||||||
takes).
|
|
||||||
=======
|
|
||||||
|
|
||||||
class PrototypeNode(object):
|
class PrototypeNode(object):
|
||||||
|
|
||||||
def **init**\ (self, name, args): self.name = name self.args = args
|
def **init**\ (self, name, args): self.name = name self.args = args
|
||||||
|
|
||||||
def CodeGen(self): # Make the function type, eg. double(double,double).
|
def CodeGen(self): # Make the function type, eg. double(double,double).
|
||||||
funct\_type = Type.function( Type.double(), [Type.double()] \*
|
funct_type = Type.function( Type.double(), [Type.double()] \*
|
||||||
len(self.args), False)
|
len(self.args), False)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1036,15 +1128,13 @@ len(self.args), False)
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
This class represents a function definition itself.
|
# This class represents a function definition itself.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class FunctionNode(object):
|
class FunctionNode(object):
|
||||||
|
|
||||||
def **init**\ (self, prototype, body): self.prototype = prototype
|
def **init**\ (self, prototype, body): self.prototype = prototype
|
||||||
self.body = body
|
self.body = body
|
||||||
|
|
||||||
def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
def CodeGen(self): # Clear scope. g_named_values.clear()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -1077,8 +1167,8 @@ Parser
|
||||||
|
|
||||||
class Parser(object):
|
class Parser(object):
|
||||||
|
|
||||||
def **init**\ (self, tokens, binop\_precedence): self.tokens = tokens
|
def **init**\ (self, tokens, binop_precedence): self.tokens = tokens
|
||||||
self.binop\_precedence = binop\_precedence self.Next()
|
self.binop_precedence = binop_precedence self.Next()
|
||||||
|
|
||||||
# Provide a simple token buffer. Parser.current is the current token the
|
# Provide a simple token buffer. Parser.current is the current token the
|
||||||
# parser is looking at. Parser.Next() reads another token from the lexer
|
# parser is looking at. Parser.Next() reads another token from the lexer
|
||||||
|
|
@ -1088,10 +1178,10 @@ self.current = self.tokens.next()
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
isinstance(self.current, CharacterToken): return
|
isinstance(self.current, CharacterToken): return
|
||||||
self.binop\_precedence.get(self.current.char, -1) else: return -1
|
self.binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
self.Next() # eat identifier.
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1201,7 +1291,7 @@ self.current == CharacterToken('('): return self.ParseParenExpr() else:
|
||||||
raise RuntimeError('Unknown token when expecting an expression.')
|
raise RuntimeError('Unknown token when expecting an expression.')
|
||||||
|
|
||||||
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (operator primary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1272,8 +1362,8 @@ def HandleExtern(self): self.Handle(self.ParseExtern, 'Read an extern:')
|
||||||
|
|
||||||
def HandleTopLevelExpression(self): try: function =
|
def HandleTopLevelExpression(self): try: function =
|
||||||
self.ParseTopLevelExpr().CodeGen() result =
|
self.ParseTopLevelExpr().CodeGen() result =
|
||||||
g\_llvm\_executor.run\_function(function, []) print 'Evaluated to:',
|
g_llvm_executor.run_function(function, []) print 'Evaluated to:',
|
||||||
result.as\_real(Type.double()) except Exception, e: print 'Error:', e
|
result.as_real(Type.double()) except Exception, e: print 'Error:', e
|
||||||
try: self.Next() # Skip for error recovery. except: pass
|
try: self.Next() # Skip for error recovery. except: pass
|
||||||
|
|
||||||
def Handle(self, function, message): try: print message,
|
def Handle(self, function, message): try: print message,
|
||||||
|
|
@ -1285,22 +1375,22 @@ Main driver code.
|
||||||
|
|
||||||
def main(): # Set up the optimizer pipeline. Start with registering info
|
def main(): # Set up the optimizer pipeline. Start with registering info
|
||||||
about how the # target lays out data structures.
|
about how the # target lays out data structures.
|
||||||
g\_llvm\_pass\_manager.add(g\_llvm\_executor.target\_data) # Do simple
|
g_llvm_pass_manager.add(g_llvm_executor.target_data) # Do simple
|
||||||
"peephole" optimizations and bit-twiddling optzns.
|
"peephole" optimizations and bit-twiddling optzns.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_INSTRUCTION\_COMBINING) # Reassociate
|
g_llvm_pass_manager.add(PASS_INSTRUCTION_COMBINING) # Reassociate
|
||||||
expressions. g\_llvm\_pass\_manager.add(PASS\_REASSOCIATE) # Eliminate
|
expressions. g_llvm_pass_manager.add(PASS_REASSOCIATE) # Eliminate
|
||||||
Common SubExpressions. g\_llvm\_pass\_manager.add(PASS\_GVN) # Simplify
|
Common SubExpressions. g_llvm_pass_manager.add(PASS_GVN) # Simplify
|
||||||
the control flow graph (deleting unreachable blocks, etc).
|
the control flow graph (deleting unreachable blocks, etc).
|
||||||
g\_llvm\_pass\_manager.add(PASS\_CFG\_SIMPLIFICATION)
|
g_llvm_pass_manager.add(PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
g\_llvm\_pass\_manager.initialize()
|
g_llvm_pass_manager.initialize()
|
||||||
|
|
||||||
# Install standard binary operators. # 1 is lowest possible precedence.
|
# Install standard binary operators. # 1 is lowest possible precedence.
|
||||||
40 is the highest. operator\_precedence = { '<': 10, '+': 20, '-': 20,
|
40 is the highest. operator_precedence = { '<': 10, '+': 20, '-': 20,
|
||||||
'\*': 40 }
|
'\*': 40 }
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: break
|
= raw_input() except KeyboardInterrupt: break
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -1316,11 +1406,6 @@ g\_llvm\_pass\_manager.initialize()
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
# Print out all of the generated code. print '', g\_llvm\_module
|
# Print out all of the generated code. print '', g_llvm_module
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Extending the language: user-defined
|
|
||||||
operators <PythonLangImpl6.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -50,23 +50,22 @@ The two specific features we'll add are programmable unary operators
|
||||||
(right now, Kaleidoscope has no unary operators at all) as well as
|
(right now, Kaleidoscope has no unary operators at all) as well as
|
||||||
binary operators. An example of this is:
|
binary operators. An example of this is:
|
||||||
|
|
||||||
{% highlight python %} # Logical unary not. def unary!(v) if v then 0
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Logical unary not. def unary!(v) if v then 0
|
||||||
else 1
|
else 1
|
||||||
|
|
||||||
Define > with the same precedence as <.
|
# Define > with the same precedence as <.
|
||||||
=======================================
|
|
||||||
|
|
||||||
def binary> 10 (LHS RHS) RHS < LHS
|
def binary> 10 (LHS RHS) RHS < LHS
|
||||||
|
|
||||||
Binary "logical or", (note that it does not "short circuit").
|
# Binary "logical or", (note that it does not "short circuit").
|
||||||
=============================================================
|
|
||||||
|
|
||||||
def binary\| 5 (LHS RHS) if LHS then 1 else if RHS then 1 else 0
|
def binary\| 5 (LHS RHS) if LHS then 1 else if RHS then 1 else 0
|
||||||
|
|
||||||
Define = with slightly lower precedence than relationals.
|
# Define = with slightly lower precedence than relationals.
|
||||||
=========================================================
|
def binary= 9 (LHS RHS) !(LHS < RHS \| LHS > RHS)
|
||||||
|
|
||||||
|
|
||||||
def binary= 9 (LHS RHS) !(LHS < RHS \| LHS > RHS) {% endhighlight %}
|
|
||||||
|
|
||||||
Many languages aspire to being able to implement their standard runtime
|
Many languages aspire to being able to implement their standard runtime
|
||||||
library in the language itself. In Kaleidoscope, we can implement
|
library in the language itself. In Kaleidoscope, we can implement
|
||||||
|
|
@ -85,7 +84,10 @@ Adding support for user-defined binary operators is pretty simple with
|
||||||
our current framework. We'll first add support for the unary/binary
|
our current framework. We'll first add support for the unary/binary
|
||||||
keywords:
|
keywords:
|
||||||
|
|
||||||
{% highlight python %} class InToken(object): pass class
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
class InToken(object): pass class
|
||||||
BinaryToken(object): pass class UnaryToken(object): pass ... def
|
BinaryToken(object): pass class UnaryToken(object): pass ... def
|
||||||
Tokenize(string): ... elif identifier == 'in': yield InToken() elif
|
Tokenize(string): ... elif identifier == 'in': yield InToken() elif
|
||||||
identifier == 'binary': yield BinaryToken() elif identifier == 'unary':
|
identifier == 'binary': yield BinaryToken() elif identifier == 'unary':
|
||||||
|
|
@ -111,15 +113,17 @@ function, which captures its name, # and its argument names (thus
|
||||||
implicitly the number of arguments the function # takes), as well as if
|
implicitly the number of arguments the function # takes), as well as if
|
||||||
it is an operator. class PrototypeNode(object):
|
it is an operator. class PrototypeNode(object):
|
||||||
|
|
||||||
def **init**\ (self, name, args, is\_operator=False, precedence=0):
|
def **init**\ (self, name, args, is_operator=False, precedence=0):
|
||||||
self.name = name self.args = args self.is\_operator = is\_operator
|
self.name = name self.args = args self.is_operator = is_operator
|
||||||
self.precedence = precedence
|
self.precedence = precedence
|
||||||
|
|
||||||
def IsBinaryOp(self): return self.is\_operator and len(self.args) == 2
|
def IsBinaryOp(self): return self.is_operator and len(self.args) == 2
|
||||||
|
|
||||||
|
def GetOperatorName(self): assert self.is_operator return self.name[-1]
|
||||||
|
|
||||||
|
def CodeGen(self): ...
|
||||||
|
|
||||||
def GetOperatorName(self): assert self.is\_operator return self.name[-1]
|
|
||||||
|
|
||||||
def CodeGen(self): ... {% endhighlight %}
|
|
||||||
|
|
||||||
Basically, in addition to knowing a name for the prototype, we now keep
|
Basically, in addition to knowing a name for the prototype, we now keep
|
||||||
track of whether it was an operator, and if it was, what precedence
|
track of whether it was an operator, and if it was, what precedence
|
||||||
|
|
@ -128,14 +132,17 @@ operators (as you'll see below, it just doesn't apply for unary
|
||||||
operators). Now that we have a way to represent the prototype for a
|
operators). Now that we have a way to represent the prototype for a
|
||||||
user-defined operator, we need to parse it:
|
user-defined operator, we need to parse it:
|
||||||
|
|
||||||
{% highlight python %} # prototype # ::= id '(' id\* ')' # ::= binary
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# prototype # ::= id '(' id\* ')' # ::= binary
|
||||||
LETTER number? (id, id) # ::= unary LETTER (id) def
|
LETTER number? (id, id) # ::= unary LETTER (id) def
|
||||||
ParsePrototype(self): precedence = None if isinstance(self.current,
|
ParsePrototype(self): precedence = None if isinstance(self.current,
|
||||||
IdentifierToken): kind = 'normal' function\_name = self.current.name
|
IdentifierToken): kind = 'normal' function_name = self.current.name
|
||||||
self.Next() # eat function name. elif isinstance(self.current,
|
self.Next() # eat function name. elif isinstance(self.current,
|
||||||
BinaryToken): kind = 'binary' self.Next() # eat 'binary'. if not
|
BinaryToken): kind = 'binary' self.Next() # eat 'binary'. if not
|
||||||
isinstance(self.current, CharacterToken): raise RuntimeError('Expected
|
isinstance(self.current, CharacterToken): raise RuntimeError('Expected
|
||||||
an operator after "binary".') function\_name = 'binary' +
|
an operator after "binary".') function_name = 'binary' +
|
||||||
self.current.char self.Next() # eat the operator. if
|
self.current.char self.Next() # eat the operator. if
|
||||||
isinstance(self.current, NumberToken): if not 1 <= self.current.value <=
|
isinstance(self.current, NumberToken): if not 1 <= self.current.value <=
|
||||||
100: raise RuntimeError('Invalid precedence: must be in range [1,
|
100: raise RuntimeError('Invalid precedence: must be in range [1,
|
||||||
|
|
@ -165,7 +172,9 @@ precedence. else: raise RuntimeError('Expected function name, "unary" or
|
||||||
|
|
||||||
return PrototypeNode(function_name, arg_names, kind != 'normal', precedence)
|
return PrototypeNode(function_name, arg_names, kind != 'normal', precedence)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This is all fairly straightforward parsing code, and we have already
|
This is all fairly straightforward parsing code, and we have already
|
||||||
seen a lot of similar code in the past. One interesting part about the
|
seen a lot of similar code in the past. One interesting part about the
|
||||||
|
|
@ -178,7 +187,10 @@ The next interesting thing to add, is codegen support for these binary
|
||||||
operators. Given our current structure, this is a simple addition of a
|
operators. Given our current structure, this is a simple addition of a
|
||||||
default case for our existing binary operator node:
|
default case for our existing binary operator node:
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): left = self.left.CodeGen()
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def CodeGen(self): left = self.left.CodeGen()
|
||||||
right = self.right.CodeGen()
|
right = self.right.CodeGen()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -197,7 +209,9 @@ right = self.right.CodeGen()
|
||||||
function = g_llvm_module.get_function_named('binary' + self.operator)
|
function = g_llvm_module.get_function_named('binary' + self.operator)
|
||||||
return g_llvm_builder.call(function, [left, right], 'binop')
|
return g_llvm_builder.call(function, [left, right], 'binop')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
As you can see above, the new code is actually really simple. It just
|
As you can see above, the new code is actually really simple. It just
|
||||||
does a lookup for the appropriate operator in the symbol table and
|
does a lookup for the appropriate operator in the symbol table and
|
||||||
|
|
@ -209,8 +223,11 @@ The final piece of code we are missing, is a bit of top-level magic. We
|
||||||
will need to make the dinary precedence map global and modify it
|
will need to make the dinary precedence map global and modify it
|
||||||
whenever we define a new binary operator:
|
whenever we define a new binary operator:
|
||||||
|
|
||||||
{% highlight python %} # The binary operator precedence chart.
|
|
||||||
g\_binop\_precedence = {} ... class FunctionNode(object): ... def
|
.. code-block:: python
|
||||||
|
|
||||||
|
# The binary operator precedence chart.
|
||||||
|
g_binop_precedence = {} ... class FunctionNode(object): ... def
|
||||||
CodeGen(self): ... # Create a function object. function =
|
CodeGen(self): ... # Create a function object. function =
|
||||||
self.prototype.CodeGen()
|
self.prototype.CodeGen()
|
||||||
|
|
||||||
|
|
@ -232,9 +249,11 @@ self.prototype.CodeGen()
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
... def main(): ... g\_binop\_precedence['<'] = 10
|
... def main(): ... g_binop_precedence['<'] = 10
|
||||||
g\_binop\_precedence['+'] = 20 g\_binop\_precedence['-'] = 20
|
g_binop_precedence['+'] = 20 g_binop_precedence['-'] = 20
|
||||||
g\_binop\_precedence['\*'] = 40 ... {% endhighlight %}
|
g_binop_precedence['\*'] = 40 ...
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Basically, before CodeGening a function, if it is a user-defined
|
Basically, before CodeGening a function, if it is a user-defined
|
||||||
operator, we register it in the precedence table. This allows the binary
|
operator, we register it in the precedence table. This allows the binary
|
||||||
|
|
@ -255,20 +274,28 @@ language, we'll need to add everything to support them. Above, we added
|
||||||
simple support for the 'unary' keyword to the lexer. In addition to
|
simple support for the 'unary' keyword to the lexer. In addition to
|
||||||
that, we need an AST node:
|
that, we need an AST node:
|
||||||
|
|
||||||
{% highlight python %} # Expression class for a unary operator. class
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Expression class for a unary operator. class
|
||||||
UnaryExpressionNode(ExpressionNode):
|
UnaryExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, operator, operand): self.operator = operator
|
def **init**\ (self, operator, operand): self.operator = operator
|
||||||
self.operand = operand
|
self.operand = operand
|
||||||
|
|
||||||
def CodeGen(self): ... {% endhighlight %}
|
def CodeGen(self): ...
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This AST node is very simple and obvious by now. It directly mirrors the
|
This AST node is very simple and obvious by now. It directly mirrors the
|
||||||
binary operator AST node, except that it only has one child. With this,
|
binary operator AST node, except that it only has one child. With this,
|
||||||
we need to add the parsing logic. Parsing a unary operator is pretty
|
we need to add the parsing logic. Parsing a unary operator is pretty
|
||||||
simple: we'll add a new function to do it:
|
simple: we'll add a new function to do it:
|
||||||
|
|
||||||
{% highlight python %} # unary ::= primary \| unary\_operator unary def
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# unary ::= primary \| unary_operator unary def
|
||||||
ParseUnary(self): # If the current token is not an operator, it must be
|
ParseUnary(self): # If the current token is not an operator, it must be
|
||||||
a primary expression. if (not isinstance(self.current, CharacterToken)
|
a primary expression. if (not isinstance(self.current, CharacterToken)
|
||||||
or self.current in [CharacterToken('('), CharacterToken(',')]): return
|
or self.current in [CharacterToken('('), CharacterToken(',')]): return
|
||||||
|
|
@ -281,7 +308,9 @@ self.ParsePrimary()
|
||||||
self.Next() # eat the operator.
|
self.Next() # eat the operator.
|
||||||
return UnaryExpressionNode(operator, self.ParseUnary())
|
return UnaryExpressionNode(operator, self.ParseUnary())
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The grammar we add is pretty straightforward here. If we see a unary
|
The grammar we add is pretty straightforward here. If we see a unary
|
||||||
operator when parsing a primary operator, we eat the operator as a
|
operator when parsing a primary operator, we eat the operator as a
|
||||||
|
|
@ -294,47 +323,62 @@ The problem with this function, is that we need to call ParseUnary from
|
||||||
somewhere. To do this, we change previous callers of ParsePrimary to
|
somewhere. To do this, we change previous callers of ParsePrimary to
|
||||||
call ParseUnary instead:
|
call ParseUnary instead:
|
||||||
|
|
||||||
{% highlight python %} # binoprhs ::= (binary\_operator unary)\* def
|
|
||||||
ParseBinOpRHS(self, left, left\_precedence): ... # Parse the unary
|
.. code-block:: python
|
||||||
|
|
||||||
|
# binoprhs ::= (binary_operator unary)\* def
|
||||||
|
ParseBinOpRHS(self, left, left_precedence): ... # Parse the unary
|
||||||
expression after the binary operator. right = self.ParseUnary() ...
|
expression after the binary operator. right = self.ParseUnary() ...
|
||||||
|
|
||||||
# expression ::= unary binoprhs def ParseExpression(self): left =
|
# expression ::= unary binoprhs def ParseExpression(self): left =
|
||||||
self.ParseUnary() return self.ParseBinOpRHS(left, 0) {% endhighlight %}
|
self.ParseUnary() return self.ParseBinOpRHS(left, 0)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
With these two simple changes, we are now able to parse unary operators
|
With these two simple changes, we are now able to parse unary operators
|
||||||
and build the AST for them. Next up, we need to add parser support for
|
and build the AST for them. Next up, we need to add parser support for
|
||||||
prototypes, to parse the unary operator prototype. We extend the binary
|
prototypes, to parse the unary operator prototype. We extend the binary
|
||||||
operator code above with:
|
operator code above with:
|
||||||
|
|
||||||
{% highlight python %} # prototype # ::= id '(' id\* ')' # ::= binary
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# prototype # ::= id '(' id\* ')' # ::= binary
|
||||||
LETTER number? (id, id) # ::= unary LETTER (id) def
|
LETTER number? (id, id) # ::= unary LETTER (id) def
|
||||||
ParsePrototype(self): precedence = None if isinstance(self.current,
|
ParsePrototype(self): precedence = None if isinstance(self.current,
|
||||||
IdentifierToken): ... elif isinstance(self.current, UnaryToken): kind =
|
IdentifierToken): ... elif isinstance(self.current, UnaryToken): kind =
|
||||||
'unary' self.Next() # eat 'unary'. if not isinstance(self.current,
|
'unary' self.Next() # eat 'unary'. if not isinstance(self.current,
|
||||||
CharacterToken): raise RuntimeError('Expected an operator after
|
CharacterToken): raise RuntimeError('Expected an operator after
|
||||||
"unary".') function\_name = 'unary' + self.current.char self.Next() #
|
"unary".') function_name = 'unary' + self.current.char self.Next() #
|
||||||
eat the operator. elif isinstance(self.current, BinaryToken): ... else:
|
eat the operator. elif isinstance(self.current, BinaryToken): ... else:
|
||||||
raise RuntimeError('Expected function name, "unary" or "binary" in '
|
raise RuntimeError('Expected function name, "unary" or "binary" in '
|
||||||
'prototype.') ... if kind == 'unary' and len(arg\_names) != 1: raise
|
'prototype.') ... if kind == 'unary' and len(arg_names) != 1: raise
|
||||||
RuntimeError('Invalid number of arguments for a unary operator.') elif
|
RuntimeError('Invalid number of arguments for a unary operator.') elif
|
||||||
kind == 'binary' and len(arg\_names) != 2: raise RuntimeError('Invalid
|
kind == 'binary' and len(arg_names) != 2: raise RuntimeError('Invalid
|
||||||
number of arguments for a binary operator.')
|
number of arguments for a binary operator.')
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
return PrototypeNode(function_name, arg_names, kind != 'normal', precedence)
|
return PrototypeNode(function_name, arg_names, kind != 'normal', precedence)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
As with binary operators, we name unary operators with a name that
|
As with binary operators, we name unary operators with a name that
|
||||||
includes the operator character. This assists us at code generation
|
includes the operator character. This assists us at code generation
|
||||||
time. Speaking of, the final piece we need to add is codegen support for
|
time. Speaking of, the final piece we need to add is codegen support for
|
||||||
unary operators. It looks like this:
|
unary operators. It looks like this:
|
||||||
|
|
||||||
{% highlight python %} class UnaryExpressionNode(ExpressionNode): ...
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
class UnaryExpressionNode(ExpressionNode): ...
|
||||||
def CodeGen(self): operand = self.operand.CodeGen() function =
|
def CodeGen(self): operand = self.operand.CodeGen() function =
|
||||||
g\_llvm\_module.get\_function\_named('unary' + self.operator) return
|
g_llvm_module.get_function_named('unary' + self.operator) return
|
||||||
g\_llvm\_builder.call(function, [operand], 'unop') {% endhighlight %}
|
g_llvm_builder.call(function, [operand], 'unop')
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code is similar to, but simpler than, the code for binary
|
This code is similar to, but simpler than, the code for binary
|
||||||
operators. It is simpler primarily because it doesn't need to handle any
|
operators. It is simpler primarily because it doesn't need to handle any
|
||||||
|
|
@ -351,49 +395,52 @@ this, we can do a lot of interesting things, including I/O, math, and a
|
||||||
bunch of other things. For example, we can now add a nice sequencing
|
bunch of other things. For example, we can now add a nice sequencing
|
||||||
operator (assuming we import ``putchard`` as described in Chapter 4):
|
operator (assuming we import ``putchard`` as described in Chapter 4):
|
||||||
|
|
||||||
{% highlight python %} ready> def binary : 1 (x y) 0 # Low-precedence
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
ready> def binary : 1 (x y) 0 # Low-precedence
|
||||||
operator that ignores operands. ... ready> extern putchard(x) ... ready>
|
operator that ignores operands. ... ready> extern putchard(x) ... ready>
|
||||||
def printd(x) putchard(x) : putchard(10) .. ready> printd(65) :
|
def printd(x) putchard(x) : putchard(10) .. ready> printd(65) :
|
||||||
printd(66) : printd(67) A B C Evaluated to: 0.0 {% endhighlight %}
|
printd(66) : printd(67) A B C Evaluated to: 0.0
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
We can also define a bunch of other "primitive" operations, such as:
|
We can also define a bunch of other "primitive" operations, such as:
|
||||||
|
|
||||||
{% highlight python %} # Logical unary not. def unary!(v) if v then 0
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Logical unary not. def unary!(v) if v then 0
|
||||||
else 1
|
else 1
|
||||||
|
|
||||||
Unary negate.
|
# Unary negate.
|
||||||
=============
|
|
||||||
|
|
||||||
def unary-(v) 0-v
|
def unary-(v) 0-v
|
||||||
|
|
||||||
Define > with the same precedence as <.
|
# Define > with the same precedence as <.
|
||||||
=======================================
|
|
||||||
|
|
||||||
def binary> 10 (LHS RHS) RHS < LHS
|
def binary> 10 (LHS RHS) RHS < LHS
|
||||||
|
|
||||||
Binary logical or, which does not short circuit.
|
# Binary logical or, which does not short circuit.
|
||||||
================================================
|
|
||||||
|
|
||||||
def binary\| 5 (LHS RHS) if LHS then 1 else if RHS then 1 else 0
|
def binary\| 5 (LHS RHS) if LHS then 1 else if RHS then 1 else 0
|
||||||
|
|
||||||
Binary logical and, which does not short circuit.
|
# Binary logical and, which does not short circuit.
|
||||||
=================================================
|
|
||||||
|
|
||||||
def binary& 6 (LHS RHS) if !LHS then 0 else !!RHS
|
def binary& 6 (LHS RHS) if !LHS then 0 else !!RHS
|
||||||
|
|
||||||
Define = with slightly lower precedence than relationals.
|
# Define = with slightly lower precedence than relationals.
|
||||||
=========================================================
|
|
||||||
|
|
||||||
def binary = 9 (LHS RHS) !(LHS < RHS \| LHS > RHS)
|
def binary = 9 (LHS RHS) !(LHS < RHS \| LHS > RHS)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Given the previous if/then/else support, we can also define interesting
|
Given the previous if/then/else support, we can also define interesting
|
||||||
functions for I/O. For example, the following prints out a character
|
functions for I/O. For example, the following prints out a character
|
||||||
whose "density" reflects the value passed in: the lower the value, the
|
whose "density" reflects the value passed in: the lower the value, the
|
||||||
denser the character:
|
denser the character:
|
||||||
|
|
||||||
{% highlight python %} ready>
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
ready>
|
||||||
|
|
||||||
extern putchard(char) def printdensity(d) if d > 8 then putchard(32) # '
|
extern putchard(char) def printdensity(d) if d > 8 then putchard(32) # '
|
||||||
' else if d > 4 then putchard(46) # '.' else if d > 2 then putchard(43)
|
' else if d > 4 then putchard(46) # '.' else if d > 2 then putchard(43)
|
||||||
|
|
@ -414,11 +461,11 @@ mandelconverger(real imag iters creal cimag) if iters > 255 \|
|
||||||
mandelconverger(real\ *real - imag*\ imag + creal, 2\ *real*\ imag +
|
mandelconverger(real\ *real - imag*\ imag + creal, 2\ *real*\ imag +
|
||||||
cimag, iters+1, creal, cimag)
|
cimag, iters+1, creal, cimag)
|
||||||
|
|
||||||
return the number of iterations required for the iteration to escape
|
# return the number of iterations required for the iteration to escape
|
||||||
====================================================================
|
|
||||||
|
|
||||||
def mandelconverge(real imag) mandelconverger(real, imag, 0, real, imag)
|
def mandelconverge(real imag) mandelconverger(real, imag, 0, real, imag)
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This "z = z2 + c" function is a beautiful little creature that is the
|
This "z = z2 + c" function is a beautiful little creature that is the
|
||||||
basis for computation of the `Mandelbrot
|
basis for computation of the `Mandelbrot
|
||||||
|
|
@ -430,24 +477,28 @@ two-dimensional plane, you can see the Mandelbrot set. Given that we are
|
||||||
limited to using putchard here, our amazing graphical output is limited,
|
limited to using putchard here, our amazing graphical output is limited,
|
||||||
but we can whip together something using the density plotter above:
|
but we can whip together something using the density plotter above:
|
||||||
|
|
||||||
{% highlight python %} # compute and plot the mandlebrot set with the
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# compute and plot the mandlebrot set with the
|
||||||
specified 2 dimensional range # info. def mandelhelp(xmin xmax xstep
|
specified 2 dimensional range # info. def mandelhelp(xmin xmax xstep
|
||||||
ymin ymax ystep) for y = ymin, y < ymax, ystep in ( (for x = xmin, x <
|
ymin ymax ystep) for y = ymin, y < ymax, ystep in ( (for x = xmin, x <
|
||||||
xmax, xstep in printdensity(mandleconverge(x,y))) : putchard(10) )
|
xmax, xstep in printdensity(mandleconverge(x,y))) : putchard(10) )
|
||||||
|
|
||||||
mandel - This is a convenient helper function for ploting the mandelbrot set
|
# mandel - This is a convenient helper function for ploting the mandelbrot set
|
||||||
============================================================================
|
# from the specified position with the specified Magnification.
|
||||||
|
|
||||||
from the specified position with the specified Magnification.
|
|
||||||
=============================================================
|
|
||||||
|
|
||||||
def mandel(realstart imagstart realmag imagmag) mandelhelp(realstart,
|
def mandel(realstart imagstart realmag imagmag) mandelhelp(realstart,
|
||||||
realstart+realmag\ *78, realmag, imagstart, imagstart+imagmag*\ 40,
|
realstart+realmag\ *78, realmag, imagstart, imagstart+imagmag*\ 40,
|
||||||
imagmag); {% endhighlight %}
|
imagmag);
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Given this, we can try plotting out the mandlebrot set! Lets try it out:
|
Given this, we can try plotting out the mandlebrot set! Lets try it out:
|
||||||
|
|
||||||
{% highlight bash %} ready> mandel(-2.3, -1.3, 0.05, 0.07)
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ready> mandel(-2.3, -1.3, 0.05, 0.07)
|
||||||
\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*
|
\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*
|
||||||
\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*
|
\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*
|
||||||
\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*++++++\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*
|
\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*++++++\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*\*
|
||||||
|
|
@ -598,7 +649,9 @@ Evaluated to: 0.0 ready> mandel(-0.9, -1.4, 0.02, 0.03)
|
||||||
.......... ......+++++++\ **\* .......... .....+++++++**\ \* ..........
|
.......... ......+++++++\ **\* .......... .....+++++++**\ \* ..........
|
||||||
.....++++++\ **\* ......... .+++++++** ........ .+++++++\ *\* ......
|
.....++++++\ **\* ......... .+++++++** ........ .+++++++\ *\* ......
|
||||||
...+++++++* . ....++++++++\* ...++++++++\* ..+++++++++ ..+++++++++
|
...+++++++* . ....++++++++\* ...++++++++\* ..+++++++++ ..+++++++++
|
||||||
Evaluated to: 0.0 ready> ^C {% endhighlight %}
|
Evaluated to: 0.0 ready> ^C
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
At this point, you may be starting to realize that Kaleidoscope is a
|
At this point, you may be starting to realize that Kaleidoscope is a
|
||||||
real and powerful language. It may not be self-similar :), but it can be
|
real and powerful language. It may not be self-similar :), but it can be
|
||||||
|
|
@ -627,58 +680,45 @@ Full Code Listing # {#code}
|
||||||
Here is the complete code listing for our running example, enhanced with
|
Here is the complete code listing for our running example, enhanced with
|
||||||
the if/then/else and for expressions:
|
the if/then/else and for expressions:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
import re from llvm.core import Module, Constant, Type, Function,
|
import re from llvm.core import Module, Constant, Type, Function,
|
||||||
Builder from llvm.ee import ExecutionEngine, TargetData from llvm.passes
|
Builder from llvm.ee import ExecutionEngine, TargetData from llvm.passes
|
||||||
import FunctionPassManager
|
import FunctionPassManager
|
||||||
|
|
||||||
from llvm.core import FCMP\_ULT, FCMP\_ONE from llvm.passes import
|
from llvm.core import FCMP_ULT, FCMP_ONE from llvm.passes import
|
||||||
(PASS\_INSTRUCTION\_COMBINING, PASS\_REASSOCIATE, PASS\_GVN,
|
(PASS_INSTRUCTION_COMBINING, PASS_REASSOCIATE, PASS_GVN,
|
||||||
PASS\_CFG\_SIMPLIFICATION)
|
PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
Globals
|
Globals
|
||||||
-------
|
-------
|
||||||
|
|
||||||
The LLVM module, which holds all the IR code.
|
# The LLVM module, which holds all the IR code.
|
||||||
=============================================
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
|
g_llvm_builder = None
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
=========================================================================
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
|
|
||||||
g\_llvm\_builder = None
|
# The function optimization passes manager.
|
||||||
|
g_llvm_pass_manager = FunctionPassManager.new(g_llvm_module)
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
# The LLVM execution engine.
|
||||||
==============================================================================
|
g_llvm_executor = ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
and what their LLVM representation is.
|
# The binary operator precedence chart.
|
||||||
======================================
|
g_binop_precedence = {}
|
||||||
|
|
||||||
g\_named\_values = {}
|
|
||||||
|
|
||||||
The function optimization passes manager.
|
|
||||||
=========================================
|
|
||||||
|
|
||||||
g\_llvm\_pass\_manager = FunctionPassManager.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
The LLVM execution engine.
|
|
||||||
==========================
|
|
||||||
|
|
||||||
g\_llvm\_executor = ExecutionEngine.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
The binary operator precedence chart.
|
|
||||||
=====================================
|
|
||||||
|
|
||||||
g\_binop\_precedence = {}
|
|
||||||
|
|
||||||
Lexer
|
Lexer
|
||||||
-----
|
-----
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
# The lexer yields one of these types for each token.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class EOFToken(object): pass class DefToken(object): pass class
|
class EOFToken(object): pass class DefToken(object): pass class
|
||||||
ExternToken(object): pass class IfToken(object): pass class
|
ExternToken(object): pass class IfToken(object): pass class
|
||||||
ThenToken(object): pass class ElseToken(object): pass class
|
ThenToken(object): pass class ElseToken(object): pass class
|
||||||
|
|
@ -696,11 +736,9 @@ char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
== other
|
== other
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
=============================================================
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
string[0].isspace(): string = string[1:] continue
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
@ -754,34 +792,26 @@ yield EOFToken()
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
# Expression class for numeric literals like "1.0".
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode):
|
class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
def **init**\ (self, value): self.value = value
|
||||||
|
|
||||||
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
# Expression class for referencing a variable, like "a".
|
||||||
======================================================
|
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
def CodeGen(self): if self.name in g\_named\_values: return
|
def CodeGen(self): if self.name in g_named_values: return
|
||||||
g\_named\_values[self.name] else: raise RuntimeError('Unknown variable
|
g_named_values[self.name] else: raise RuntimeError('Unknown variable
|
||||||
name: ' + self.name)
|
name: ' + self.name)
|
||||||
|
|
||||||
Expression class for a binary operator.
|
# Expression class for a binary operator.
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode):
|
class BinaryOperatorExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, operator, left, right): self.operator = operator
|
def **init**\ (self, operator, left, right): self.operator = operator
|
||||||
|
|
@ -806,16 +836,14 @@ self.right.CodeGen()
|
||||||
function = g_llvm_module.get_function_named('binary' + self.operator)
|
function = g_llvm_module.get_function_named('binary' + self.operator)
|
||||||
return g_llvm_builder.call(function, [left, right], 'binop')
|
return g_llvm_builder.call(function, [left, right], 'binop')
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode):
|
class CallExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, callee, args): self.callee = callee self.args =
|
def **init**\ (self, callee, args): self.callee = callee self.args =
|
||||||
args
|
args
|
||||||
|
|
||||||
def CodeGen(self): # Look up the name in the global module table. callee
|
def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
= g\_llvm\_module.get\_function\_named(self.callee)
|
= g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -827,14 +855,12 @@ def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
Expression class for if/then/else.
|
# Expression class for if/then/else.
|
||||||
==================================
|
|
||||||
|
|
||||||
class IfExpressionNode(ExpressionNode):
|
class IfExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, condition, then\_branch, else\_branch):
|
def **init**\ (self, condition, then_branch, else_branch):
|
||||||
self.condition = condition self.then\_branch = then\_branch
|
self.condition = condition self.then_branch = then_branch
|
||||||
self.else\_branch = else\_branch
|
self.else_branch = else_branch
|
||||||
|
|
||||||
def CodeGen(self): condition = self.condition.CodeGen()
|
def CodeGen(self): condition = self.condition.CodeGen()
|
||||||
|
|
||||||
|
|
@ -880,13 +906,11 @@ def CodeGen(self): condition = self.condition.CodeGen()
|
||||||
|
|
||||||
return phi
|
return phi
|
||||||
|
|
||||||
Expression class for for/in.
|
# Expression class for for/in.
|
||||||
============================
|
|
||||||
|
|
||||||
class ForExpressionNode(ExpressionNode):
|
class ForExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, loop\_variable, start, end, step, body):
|
def **init**\ (self, loop_variable, start, end, step, body):
|
||||||
self.loop\_variable = loop\_variable self.start = start self.end = end
|
self.loop_variable = loop_variable self.start = start self.end = end
|
||||||
self.step = step self.body = body
|
self.step = step self.body = body
|
||||||
|
|
||||||
def CodeGen(self): # Output this as: # ... # start = startexpr # goto
|
def CodeGen(self): # Output this as: # ... # start = startexpr # goto
|
||||||
|
|
@ -961,39 +985,31 @@ endloop # outloop:
|
||||||
# for expr always returns 0.0.
|
# for expr always returns 0.0.
|
||||||
return Constant.real(Type.double(), 0)
|
return Constant.real(Type.double(), 0)
|
||||||
|
|
||||||
Expression class for a unary operator.
|
# Expression class for a unary operator.
|
||||||
======================================
|
|
||||||
|
|
||||||
class UnaryExpressionNode(ExpressionNode):
|
class UnaryExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, operator, operand): self.operator = operator
|
def **init**\ (self, operator, operand): self.operator = operator
|
||||||
self.operand = operand
|
self.operand = operand
|
||||||
|
|
||||||
def CodeGen(self): operand = self.operand.CodeGen() function =
|
def CodeGen(self): operand = self.operand.CodeGen() function =
|
||||||
g\_llvm\_module.get\_function\_named('unary' + self.operator) return
|
g_llvm_module.get_function_named('unary' + self.operator) return
|
||||||
g\_llvm\_builder.call(function, [operand], 'unop')
|
g_llvm_builder.call(function, [operand], 'unop')
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
|
||||||
==============================================================================
|
|
||||||
|
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
|
||||||
============================================================================
|
|
||||||
|
|
||||||
takes), as well as if it is an operator.
|
|
||||||
========================================
|
|
||||||
|
|
||||||
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes), as well as if it is an operator.
|
||||||
class PrototypeNode(object):
|
class PrototypeNode(object):
|
||||||
|
|
||||||
def **init**\ (self, name, args, is\_operator=False, precedence=0):
|
def **init**\ (self, name, args, is_operator=False, precedence=0):
|
||||||
self.name = name self.args = args self.is\_operator = is\_operator
|
self.name = name self.args = args self.is_operator = is_operator
|
||||||
self.precedence = precedence
|
self.precedence = precedence
|
||||||
|
|
||||||
def IsBinaryOp(self): return self.is\_operator and len(self.args) == 2
|
def IsBinaryOp(self): return self.is_operator and len(self.args) == 2
|
||||||
|
|
||||||
def GetOperatorName(self): assert self.is\_operator return self.name[-1]
|
def GetOperatorName(self): assert self.is_operator return self.name[-1]
|
||||||
|
|
||||||
def CodeGen(self): # Make the function type, eg. double(double,double).
|
def CodeGen(self): # Make the function type, eg. double(double,double).
|
||||||
funct\_type = Type.function( Type.double(), [Type.double()] \*
|
funct_type = Type.function( Type.double(), [Type.double()] \*
|
||||||
len(self.args), False)
|
len(self.args), False)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1023,15 +1039,13 @@ len(self.args), False)
|
||||||
|
|
||||||
return function
|
return function
|
||||||
|
|
||||||
This class represents a function definition itself.
|
# This class represents a function definition itself.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class FunctionNode(object):
|
class FunctionNode(object):
|
||||||
|
|
||||||
def **init**\ (self, prototype, body): self.prototype = prototype
|
def **init**\ (self, prototype, body): self.prototype = prototype
|
||||||
self.body = body
|
self.body = body
|
||||||
|
|
||||||
def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
def CodeGen(self): # Clear scope. g_named_values.clear()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -1081,10 +1095,10 @@ self.current = self.tokens.next()
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
isinstance(self.current, CharacterToken): return
|
isinstance(self.current, CharacterToken): return
|
||||||
g\_binop\_precedence.get(self.current.char, -1) else: return -1
|
g_binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
self.Next() # eat identifier.
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1193,7 +1207,7 @@ isinstance(self.current, ForToken): return self.ParseForExpr() elif
|
||||||
self.current == CharacterToken('('): return self.ParseParenExpr() else:
|
self.current == CharacterToken('('): return self.ParseParenExpr() else:
|
||||||
raise RuntimeError('Unknown token when expecting an expression.')
|
raise RuntimeError('Unknown token when expecting an expression.')
|
||||||
|
|
||||||
# unary ::= primary \| unary\_operator unary def ParseUnary(self): # If
|
# unary ::= primary \| unary_operator unary def ParseUnary(self): # If
|
||||||
the current token is not an operator, it must be a primary expression.
|
the current token is not an operator, it must be a primary expression.
|
||||||
if (not isinstance(self.current, CharacterToken) or self.current in
|
if (not isinstance(self.current, CharacterToken) or self.current in
|
||||||
[CharacterToken('('), CharacterToken(',')]): return self.ParsePrimary()
|
[CharacterToken('('), CharacterToken(',')]): return self.ParsePrimary()
|
||||||
|
|
@ -1205,8 +1219,8 @@ if (not isinstance(self.current, CharacterToken) or self.current in
|
||||||
self.Next() # eat the operator.
|
self.Next() # eat the operator.
|
||||||
return UnaryExpressionNode(operator, self.ParseUnary())
|
return UnaryExpressionNode(operator, self.ParseUnary())
|
||||||
|
|
||||||
# binoprhs ::= (binary\_operator unary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (binary_operator unary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1237,14 +1251,14 @@ self.ParseUnary() return self.ParseBinOpRHS(left, 0)
|
||||||
# prototype # ::= id '(' id\* ')' # ::= binary LETTER number? (id, id) #
|
# prototype # ::= id '(' id\* ')' # ::= binary LETTER number? (id, id) #
|
||||||
::= unary LETTER (id) def ParsePrototype(self): precedence = None if
|
::= unary LETTER (id) def ParsePrototype(self): precedence = None if
|
||||||
isinstance(self.current, IdentifierToken): kind = 'normal'
|
isinstance(self.current, IdentifierToken): kind = 'normal'
|
||||||
function\_name = self.current.name self.Next() # eat function name. elif
|
function_name = self.current.name self.Next() # eat function name. elif
|
||||||
isinstance(self.current, UnaryToken): kind = 'unary' self.Next() # eat
|
isinstance(self.current, UnaryToken): kind = 'unary' self.Next() # eat
|
||||||
'unary'. if not isinstance(self.current, CharacterToken): raise
|
'unary'. if not isinstance(self.current, CharacterToken): raise
|
||||||
RuntimeError('Expected an operator after "unary".') function\_name =
|
RuntimeError('Expected an operator after "unary".') function_name =
|
||||||
'unary' + self.current.char self.Next() # eat the operator. elif
|
'unary' + self.current.char self.Next() # eat the operator. elif
|
||||||
isinstance(self.current, BinaryToken): kind = 'binary' self.Next() # eat
|
isinstance(self.current, BinaryToken): kind = 'binary' self.Next() # eat
|
||||||
'binary'. if not isinstance(self.current, CharacterToken): raise
|
'binary'. if not isinstance(self.current, CharacterToken): raise
|
||||||
RuntimeError('Expected an operator after "binary".') function\_name =
|
RuntimeError('Expected an operator after "binary".') function_name =
|
||||||
'binary' + self.current.char self.Next() # eat the operator. if
|
'binary' + self.current.char self.Next() # eat the operator. if
|
||||||
isinstance(self.current, NumberToken): if not 1 <= self.current.value <=
|
isinstance(self.current, NumberToken): if not 1 <= self.current.value <=
|
||||||
100: raise RuntimeError('Invalid precedence: must be in range [1,
|
100: raise RuntimeError('Invalid precedence: must be in range [1,
|
||||||
|
|
@ -1293,8 +1307,8 @@ def HandleExtern(self): self.Handle(self.ParseExtern, 'Read an extern:')
|
||||||
|
|
||||||
def HandleTopLevelExpression(self): try: function =
|
def HandleTopLevelExpression(self): try: function =
|
||||||
self.ParseTopLevelExpr().CodeGen() result =
|
self.ParseTopLevelExpr().CodeGen() result =
|
||||||
g\_llvm\_executor.run\_function(function, []) print 'Evaluated to:',
|
g_llvm_executor.run_function(function, []) print 'Evaluated to:',
|
||||||
result.as\_real(Type.double()) except Exception, e: print 'Error:', e
|
result.as_real(Type.double()) except Exception, e: print 'Error:', e
|
||||||
try: self.Next() # Skip for error recovery. except: pass
|
try: self.Next() # Skip for error recovery. except: pass
|
||||||
|
|
||||||
def Handle(self, function, message): try: print message,
|
def Handle(self, function, message): try: print message,
|
||||||
|
|
@ -1306,23 +1320,23 @@ Main driver code.
|
||||||
|
|
||||||
def main(): # Set up the optimizer pipeline. Start with registering info
|
def main(): # Set up the optimizer pipeline. Start with registering info
|
||||||
about how the # target lays out data structures.
|
about how the # target lays out data structures.
|
||||||
g\_llvm\_pass\_manager.add(g\_llvm\_executor.target\_data) # Do simple
|
g_llvm_pass_manager.add(g_llvm_executor.target_data) # Do simple
|
||||||
"peephole" optimizations and bit-twiddling optzns.
|
"peephole" optimizations and bit-twiddling optzns.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_INSTRUCTION\_COMBINING) # Reassociate
|
g_llvm_pass_manager.add(PASS_INSTRUCTION_COMBINING) # Reassociate
|
||||||
expressions. g\_llvm\_pass\_manager.add(PASS\_REASSOCIATE) # Eliminate
|
expressions. g_llvm_pass_manager.add(PASS_REASSOCIATE) # Eliminate
|
||||||
Common SubExpressions. g\_llvm\_pass\_manager.add(PASS\_GVN) # Simplify
|
Common SubExpressions. g_llvm_pass_manager.add(PASS_GVN) # Simplify
|
||||||
the control flow graph (deleting unreachable blocks, etc).
|
the control flow graph (deleting unreachable blocks, etc).
|
||||||
g\_llvm\_pass\_manager.add(PASS\_CFG\_SIMPLIFICATION)
|
g_llvm_pass_manager.add(PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
g\_llvm\_pass\_manager.initialize()
|
g_llvm_pass_manager.initialize()
|
||||||
|
|
||||||
# Install standard binary operators. # 1 is lowest possible precedence.
|
# Install standard binary operators. # 1 is lowest possible precedence.
|
||||||
40 is the highest. g\_binop\_precedence['<'] = 10
|
40 is the highest. g_binop_precedence['<'] = 10
|
||||||
g\_binop\_precedence['+'] = 20 g\_binop\_precedence['-'] = 20
|
g_binop_precedence['+'] = 20 g_binop_precedence['-'] = 20
|
||||||
g\_binop\_precedence['\*'] = 40
|
g_binop_precedence['\*'] = 40
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
# Run the main "interpreter loop". while True: print 'ready>', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: break
|
= raw_input() except KeyboardInterrupt: break
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -1338,11 +1352,6 @@ g\_binop\_precedence['\*'] = 40
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
# Print out all of the generated code. print '', g\_llvm\_module
|
# Print out all of the generated code. print '', g_llvm_module
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Extending the language: mutable variables / SSA
|
|
||||||
construction <PythonLangImpl7.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -37,20 +37,30 @@ Why is this a hard problem? # {#why}
|
||||||
To understand why mutable variables cause complexities in SSA
|
To understand why mutable variables cause complexities in SSA
|
||||||
construction, consider this extremely simple C example:
|
construction, consider this extremely simple C example:
|
||||||
|
|
||||||
{% highlight python %} int G, H; int test(\_Bool Condition) { int X; if
|
|
||||||
(Condition) X = G; else X = H; return X; } {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
int G, H; int test(_Bool Condition) { int X; if
|
||||||
|
(Condition) X = G; else X = H; return X; }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In this case, we have the variable "X", whose value depends on the path
|
In this case, we have the variable "X", whose value depends on the path
|
||||||
executed in the program. Because there are two different possible values
|
executed in the program. Because there are two different possible values
|
||||||
for X before the return instruction, a PHI node is inserted to merge the
|
for X before the return instruction, a PHI node is inserted to merge the
|
||||||
two values. The LLVM IR that we want for this example looks like this:
|
two values. The LLVM IR that we want for this example looks like this:
|
||||||
|
|
||||||
{% highlight llvm %} @G = weak global i32 0 ; type of @G is i32\* @H =
|
|
||||||
|
.. code-block:: llvm
|
||||||
|
|
||||||
|
@G = weak global i32 0 ; type of @G is i32\* @H =
|
||||||
weak global i32 0 ; type of @H is i32\* define i32 @test(i1 %Condition)
|
weak global i32 0 ; type of @H is i32\* define i32 @test(i1 %Condition)
|
||||||
{ entry: br i1 %Condition, label %cond\_true, label %cond\_false
|
{ entry: br i1 %Condition, label %cond_true, label %cond_false
|
||||||
cond\_true: %X.0 = load i32\* @G br label %cond\_next cond\_false: %X.1
|
cond_true: %X.0 = load i32\* @G br label %cond_next cond_false: %X.1
|
||||||
= load i32\* @H br label %cond\_next cond\_next: %X.2 = phi i32 [ %X.1,
|
= load i32\* @H br label %cond_next cond_next: %X.2 = phi i32 [ %X.1,
|
||||||
%cond\_false ], [ %X.0, %cond\_true ] ret i32 %X.2 } {% endhighlight %}
|
%cond_false ], [ %X.0, %cond_true ] ret i32 %X.2 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
In this example, the loads from the G and H global variables are
|
In this example, the loads from the G and H global variables are
|
||||||
explicit in the LLVM IR, and they live in the then/else branches of the
|
explicit in the LLVM IR, and they live in the then/else branches of the
|
||||||
|
|
@ -98,10 +108,15 @@ work the same way, except that instead of being declared with global
|
||||||
variable definitions, they are declared with the `LLVM alloca
|
variable definitions, they are declared with the `LLVM alloca
|
||||||
instruction <http://www.llvm.org/docs/LangRef.html#i_alloca>`_:
|
instruction <http://www.llvm.org/docs/LangRef.html#i_alloca>`_:
|
||||||
|
|
||||||
{% highlight python %} define i32 @example() { entry: %X = alloca i32 ;
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
define i32 @example() { entry: %X = alloca i32 ;
|
||||||
type of %X is i32\ *. ... %tmp = load i32* %X ; load the stack value %X
|
type of %X is i32\ *. ... %tmp = load i32* %X ; load the stack value %X
|
||||||
from the stack. %tmp2 = add i32 %tmp, 1 ; increment it store i32 %tmp2,
|
from the stack. %tmp2 = add i32 %tmp, 1 ; increment it store i32 %tmp2,
|
||||||
i32\* %X ; store it back ... {% endhighlight %}
|
i32\* %X ; store it back ...
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code shows an example of how you can declare and manipulate a stack
|
This code shows an example of how you can declare and manipulate a stack
|
||||||
variable in the LLVM IR. Stack memory allocated with the alloca
|
variable in the LLVM IR. Stack memory allocated with the alloca
|
||||||
|
|
@ -110,13 +125,16 @@ to functions, you can store it in other variables, etc. In our example
|
||||||
above, we could rewrite the example to use the alloca technique to avoid
|
above, we could rewrite the example to use the alloca technique to avoid
|
||||||
using a PHI node:
|
using a PHI node:
|
||||||
|
|
||||||
{% highlight llvm %} @G = weak global i32 0 ; type of @G is i32\* @H =
|
|
||||||
|
.. code-block:: llvm
|
||||||
|
|
||||||
|
@G = weak global i32 0 ; type of @G is i32\* @H =
|
||||||
weak global i32 0 ; type of @H is i32\* define i32 @test(i1 %Condition)
|
weak global i32 0 ; type of @H is i32\* define i32 @test(i1 %Condition)
|
||||||
{ entry: %X = alloca i32 ; type of %X is i32\ *. br i1 %Condition, label
|
{ entry: %X = alloca i32 ; type of %X is i32\ *. br i1 %Condition, label
|
||||||
%cond\_true, label %cond\_false cond\_true: %X.0 = load i32* @G store
|
%cond_true, label %cond_false cond_true: %X.0 = load i32* @G store
|
||||||
i32 %X.0, i32\* %X ; Update X br label %cond\_next cond\_false: %X.1 =
|
i32 %X.0, i32\* %X ; Update X br label %cond_next cond_false: %X.1 =
|
||||||
load i32\* @H store i32 %X.1, i32\* %X ; Update X br label %cond\_next
|
load i32\* @H store i32 %X.1, i32\* %X ; Update X br label %cond_next
|
||||||
cond\_next: %X.2 = load i32\* %X ; Read X ret i32 %X.2 } {% endhighlight
|
cond_next: %X.2 = load i32\* %X ; Read X ret i32 %X.2 } {% endhighlight
|
||||||
%}
|
%}
|
||||||
|
|
||||||
With this, we have discovered a way to handle arbitrary mutable
|
With this, we have discovered a way to handle arbitrary mutable
|
||||||
|
|
@ -164,14 +182,21 @@ into SSA registers, inserting Phi nodes as appropriate. If you run this
|
||||||
example through the pass, for example, you'll get:
|
example through the pass, for example, you'll get:
|
||||||
|
|
||||||
{% highlight bash %} $ llvm-as < example.ll \| opt -mem2reg \| llvm-dis
|
{% highlight bash %} $ llvm-as < example.ll \| opt -mem2reg \| llvm-dis
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
{% highlight llvm %} @G = weak global i32 0 @H = weak global i32 0
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
.. code-block:: llvm
|
||||||
|
|
||||||
|
@G = weak global i32 0 @H = weak global i32 0
|
||||||
define i32 @test(i1 %Condition) { entry: br i1 %Condition, label
|
define i32 @test(i1 %Condition) { entry: br i1 %Condition, label
|
||||||
%cond\_true, label %cond\_false cond\_true: %X.0 = load i32\* @G br
|
%cond_true, label %cond_false cond_true: %X.0 = load i32\* @G br
|
||||||
label %cond\_next cond\_false: %X.1 = load i32\* @H br label %cond\_next
|
label %cond_next cond_false: %X.1 = load i32\* @H br label %cond_next
|
||||||
cond\_next: %X.01 = phi i32 [ %X.1, %cond\_false ], [ %X.0, %cond\_true
|
cond_next: %X.01 = phi i32 [ %X.1, %cond_false ], [ %X.0, %cond_true
|
||||||
] ret i32 %X.01 } {% endhighlight %}
|
] ret i32 %X.01 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The mem2reg pass implements the standard "iterated dominance frontier"
|
The mem2reg pass implements the standard "iterated dominance frontier"
|
||||||
algorithm for constructing SSA form and has a number of optimizations
|
algorithm for constructing SSA form and has a number of optimizations
|
||||||
|
|
@ -249,25 +274,24 @@ redefining those only goes so far :). Also, the ability to define new
|
||||||
variables is a useful thing regardless of whether you will be mutating
|
variables is a useful thing regardless of whether you will be mutating
|
||||||
them. Here's a motivating example that shows how we could use these:
|
them. Here's a motivating example that shows how we could use these:
|
||||||
|
|
||||||
{% highlight python %} # Define ':' for sequencing: as a low-precedence
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Define ':' for sequencing: as a low-precedence
|
||||||
operator that ignores operands # and just returns the RHS. def binary :
|
operator that ignores operands # and just returns the RHS. def binary :
|
||||||
1 (x y) y;
|
1 (x y) y;
|
||||||
|
|
||||||
Recursive fib, we could do this before.
|
# Recursive fib, we could do this before.
|
||||||
=======================================
|
|
||||||
|
|
||||||
def fib(x) if (x < 3) then 1 else fib(x-1) + fib(x-2)
|
def fib(x) if (x < 3) then 1 else fib(x-1) + fib(x-2)
|
||||||
|
|
||||||
Iterative fib.
|
# Iterative fib.
|
||||||
==============
|
|
||||||
|
|
||||||
def fibi(x) var a = 1, b = 1, c in (for i = 3, i < x in c = a + b : a =
|
def fibi(x) var a = 1, b = 1, c in (for i = 3, i < x in c = a + b : a =
|
||||||
b : b = c) : b
|
b : b = c) : b
|
||||||
|
|
||||||
Call it.
|
# Call it.
|
||||||
========
|
fibi(10)
|
||||||
|
|
||||||
|
|
||||||
fibi(10) {% endhighlight %}
|
|
||||||
|
|
||||||
In order to mutate variables, we have to change our existing variables
|
In order to mutate variables, we have to change our existing variables
|
||||||
to use the "alloca trick". Once we have that, we'll add our new
|
to use the "alloca trick". Once we have that, we'll add our new
|
||||||
|
|
@ -298,12 +322,17 @@ allocas that we will store in ``g_named_values``. We'll use a helper
|
||||||
function that ensures that the allocas are created in the entry block of
|
function that ensures that the allocas are created in the entry block of
|
||||||
the function:
|
the function:
|
||||||
|
|
||||||
{% highlight python %} # Creates an alloca instruction in the entry
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Creates an alloca instruction in the entry
|
||||||
block of the function. This is used # for mutable variables. def
|
block of the function. This is used # for mutable variables. def
|
||||||
CreateEntryBlockAlloca(function, var\_name): entry =
|
CreateEntryBlockAlloca(function, var_name): entry =
|
||||||
function.get\_entry\_basic\_block() builder = Builder.new(entry)
|
function.get_entry_basic_block() builder = Builder.new(entry)
|
||||||
builder.position\_at\_beginning(entry) return
|
builder.position_at_beginning(entry) return
|
||||||
builder.alloca(Type.double(), var\_name) {% endhighlight %}
|
builder.alloca(Type.double(), var_name)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code creates a temporary ``llvm.core.Builder`` that is pointing at
|
This code creates a temporary ``llvm.core.Builder`` that is pointing at
|
||||||
the first instruction of the entry block. It then creates an alloca with
|
the first instruction of the entry block. It then creates an alloca with
|
||||||
|
|
@ -315,9 +344,12 @@ variable references. In our new scheme, variables live on the stack, so
|
||||||
code generating a reference to them actually needs to produce a load
|
code generating a reference to them actually needs to produce a load
|
||||||
from the stack slot:
|
from the stack slot:
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): if self.name in
|
|
||||||
g\_named\_values: return
|
.. code-block:: python
|
||||||
g\_llvm\_builder.load(g\_named\_values[self.name], self.name) else:
|
|
||||||
|
def CodeGen(self): if self.name in
|
||||||
|
g_named_values: return
|
||||||
|
g_llvm_builder.load(g_named_values[self.name], self.name) else:
|
||||||
raise RuntimeError('Unknown variable name: ' + self.name) {%
|
raise RuntimeError('Unknown variable name: ' + self.name) {%
|
||||||
endhighlight %}
|
endhighlight %}
|
||||||
|
|
||||||
|
|
@ -327,7 +359,7 @@ with ``ForExpressionNode.CodeGen`` (see the `full code listing <#code>`_
|
||||||
for the unabridged code):
|
for the unabridged code):
|
||||||
|
|
||||||
{% highlight python %} def CodeGen(self): function =
|
{% highlight python %} def CodeGen(self): function =
|
||||||
g\_llvm\_builder.basic\_block.function
|
g_llvm_builder.basic_block.function
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -354,7 +386,9 @@ g\_llvm\_builder.basic\_block.function
|
||||||
FCMP_ONE, end_condition, Constant.real(Type.double(), 0), 'loopcond')
|
FCMP_ONE, end_condition, Constant.real(Type.double(), 0), 'loopcond')
|
||||||
...
|
...
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This code is virtually identical to the code `before we allowed mutable
|
This code is virtually identical to the code `before we allowed mutable
|
||||||
variables <PythonLangImpl5.html#forcodegen>`_. The big difference is
|
variables <PythonLangImpl5.html#forcodegen>`_. The big difference is
|
||||||
|
|
@ -364,12 +398,17 @@ access the variable as needed.
|
||||||
To support mutable argument variables, we need to also make allocas for
|
To support mutable argument variables, we need to also make allocas for
|
||||||
them. The code for this is also pretty simple:
|
them. The code for this is also pretty simple:
|
||||||
|
|
||||||
{% highlight python %} class PrototypeNode(object): ... # Create an
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
class PrototypeNode(object): ... # Create an
|
||||||
alloca for each argument and register the argument in the symbol # table
|
alloca for each argument and register the argument in the symbol # table
|
||||||
so that references to it will succeed. def CreateArgumentAllocas(self,
|
so that references to it will succeed. def CreateArgumentAllocas(self,
|
||||||
function): for arg\_name, arg in zip(self.args, function.args): alloca =
|
function): for arg_name, arg in zip(self.args, function.args): alloca =
|
||||||
CreateEntryBlockAlloca(function, arg\_name) g\_llvm\_builder.store(arg,
|
CreateEntryBlockAlloca(function, arg_name) g_llvm_builder.store(arg,
|
||||||
alloca) g\_named\_values[arg\_name] = alloca {% endhighlight %}
|
alloca) g_named_values[arg_name] = alloca
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
For each argument, we make an alloca, store the input value to the
|
For each argument, we make an alloca, store the input value to the
|
||||||
function into the alloca, and register the alloca as the memory location
|
function into the alloca, and register the alloca as the memory location
|
||||||
|
|
@ -379,17 +418,20 @@ right after it sets up the entry block for the function.
|
||||||
The final missing piece is adding the mem2reg pass, which allows us to
|
The final missing piece is adding the mem2reg pass, which allows us to
|
||||||
get good codegen once again:
|
get good codegen once again:
|
||||||
|
|
||||||
{% highlight python %} from llvm.passes import
|
|
||||||
(PASS\_PROMOTE\_MEMORY\_TO\_REGISTER, PASS\_INSTRUCTION\_COMBINING,
|
.. code-block:: python
|
||||||
PASS\_REASSOCIATE, PASS\_GVN, PASS\_CFG\_SIMPLIFICATION) ... def main():
|
|
||||||
|
from llvm.passes import
|
||||||
|
(PASS_PROMOTE_MEMORY_TO_REGISTER, PASS_INSTRUCTION_COMBINING,
|
||||||
|
PASS_REASSOCIATE, PASS_GVN, PASS_CFG_SIMPLIFICATION) ... def main():
|
||||||
# Set up the optimizer pipeline. Start with registering info about how
|
# Set up the optimizer pipeline. Start with registering info about how
|
||||||
the # target lays out data structures.
|
the # target lays out data structures.
|
||||||
g\_llvm\_pass\_manager.add(g\_llvm\_executor.target\_data) # Promote
|
g_llvm_pass_manager.add(g_llvm_executor.target_data) # Promote
|
||||||
allocas to registers.
|
allocas to registers.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_PROMOTE\_MEMORY\_TO\_REGISTER) # Do
|
g_llvm_pass_manager.add(PASS_PROMOTE_MEMORY_TO_REGISTER) # Do
|
||||||
simple "peephole" optimizations and bit-twiddling optzns.
|
simple "peephole" optimizations and bit-twiddling optzns.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_INSTRUCTION\_COMBINING) # Reassociate
|
g_llvm_pass_manager.add(PASS_INSTRUCTION_COMBINING) # Reassociate
|
||||||
expressions. g\_llvm\_pass\_manager.add(PASS\_REASSOCIATE) {%
|
expressions. g_llvm_pass_manager.add(PASS_REASSOCIATE) {%
|
||||||
endhighlight %}
|
endhighlight %}
|
||||||
|
|
||||||
It is interesting to see what the code looks like before and after the
|
It is interesting to see what the code looks like before and after the
|
||||||
|
|
@ -428,7 +470,9 @@ fcmp ult double %x, 3.000000e+00 %booltmp = uitofp i1 %cmptmp to double
|
||||||
fsub double %x, 2.000000e+00 %calltmp6 = call double @fib(double
|
fsub double %x, 2.000000e+00 %calltmp6 = call double @fib(double
|
||||||
%subtmp5) %addtmp = fadd double %calltmp, %calltmp6 br label %ifcont
|
%subtmp5) %addtmp = fadd double %calltmp, %calltmp6 br label %ifcont
|
||||||
ifcont: ; preds = %else, %then %iftmp = phi double [ 1.000000e+00, %then
|
ifcont: ; preds = %else, %then %iftmp = phi double [ 1.000000e+00, %then
|
||||||
], [ %addtmp, %else ] ret double %iftmp } {% endhighlight %}
|
], [ %addtmp, %else ] ret double %iftmp }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
This is a trivial case for mem2reg, since there are no redefinitions of
|
This is a trivial case for mem2reg, since there are no redefinitions of
|
||||||
the variable. The point of showing this is to calm your tension about
|
the variable. The point of showing this is to calm your tension about
|
||||||
|
|
@ -436,14 +480,19 @@ inserting such blatent inefficiencies :).
|
||||||
|
|
||||||
After the rest of the optimizers run, we get:
|
After the rest of the optimizers run, we get:
|
||||||
|
|
||||||
{% highlight llvm %} define double @fib(double %x) { entry: %cmptmp =
|
|
||||||
|
.. code-block:: llvm
|
||||||
|
|
||||||
|
define double @fib(double %x) { entry: %cmptmp =
|
||||||
fcmp ult double %x, 3.000000e+00 %booltmp = uitofp i1 %cmptmp to double
|
fcmp ult double %x, 3.000000e+00 %booltmp = uitofp i1 %cmptmp to double
|
||||||
%ifcond = fcmp ueq double %booltmp, 0.000000e+00 br i1 %ifcond, label
|
%ifcond = fcmp ueq double %booltmp, 0.000000e+00 br i1 %ifcond, label
|
||||||
%else, label %ifcont else: %subtmp = fsub double %x, 1.000000e+00
|
%else, label %ifcont else: %subtmp = fsub double %x, 1.000000e+00
|
||||||
%calltmp = call double @fib(double %subtmp) %subtmp5 = fsub double %x,
|
%calltmp = call double @fib(double %subtmp) %subtmp5 = fsub double %x,
|
||||||
2.000000e+00 %calltmp6 = call double @fib(double %subtmp5) %addtmp =
|
2.000000e+00 %calltmp6 = call double @fib(double %subtmp5) %addtmp =
|
||||||
fadd double %calltmp, %calltmp6 ret double %addtmp ifcont: ret double
|
fadd double %calltmp, %calltmp6 ret double %addtmp ifcont: ret double
|
||||||
1.000000e+00 } {% endhighlight %}
|
1.000000e+00 }
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Here we see that the simplifycfg pass decided to clone the return
|
Here we see that the simplifycfg pass decided to clone the return
|
||||||
instruction into the end of the 'else' block. This allowed it to
|
instruction into the end of the 'else' block. This allowed it to
|
||||||
|
|
@ -462,10 +511,13 @@ simple. We will parse it just like any other binary operator, but handle
|
||||||
it internally (instead of allowing the user to define it). The first
|
it internally (instead of allowing the user to define it). The first
|
||||||
step is to set a precedence:
|
step is to set a precedence:
|
||||||
|
|
||||||
{% highlight python %} def main(): ... # Install standard binary
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def main(): ... # Install standard binary
|
||||||
operators. # 1 is lowest possible precedence. 40 is the highest.
|
operators. # 1 is lowest possible precedence. 40 is the highest.
|
||||||
g\_binop\_precedence['='] = 2 g\_binop\_precedence['<'] = 10
|
g_binop_precedence['='] = 2 g_binop_precedence['<'] = 10
|
||||||
g\_binop\_precedence['+'] = 20 g\_binop\_precedence['-'] = 20 {%
|
g_binop_precedence['+'] = 20 g_binop_precedence['-'] = 20 {%
|
||||||
endhighlight %}
|
endhighlight %}
|
||||||
|
|
||||||
Now that the parser knows the precedence of the binary operator, it
|
Now that the parser knows the precedence of the binary operator, it
|
||||||
|
|
@ -500,7 +552,9 @@ allowed.
|
||||||
return value
|
return value
|
||||||
...
|
...
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Once we have the variable, CodeGening the assignment is straightforward:
|
Once we have the variable, CodeGening the assignment is straightforward:
|
||||||
we emit the RHS of the assignment, create a store, and return the
|
we emit the RHS of the assignment, create a store, and return the
|
||||||
|
|
@ -510,19 +564,20 @@ computed value. Returning a value allows for chained assignments like
|
||||||
Now that we have an assignment operator, we can mutate loop variables
|
Now that we have an assignment operator, we can mutate loop variables
|
||||||
and arguments. For example, we can now run code like this:
|
and arguments. For example, we can now run code like this:
|
||||||
|
|
||||||
{% highlight python %} # Function to print a double. extern printd(x)
|
|
||||||
|
|
||||||
Define ':' for sequencing: as a low-precedence operator that ignores operands
|
.. code-block:: python
|
||||||
=============================================================================
|
|
||||||
|
|
||||||
and just returns the RHS.
|
# Function to print a double. extern printd(x)
|
||||||
=========================
|
|
||||||
|
|
||||||
|
# Define ':' for sequencing: as a low-precedence operator that ignores operands
|
||||||
|
# and just returns the RHS.
|
||||||
def binary : 1 (x y) y
|
def binary : 1 (x y) y
|
||||||
|
|
||||||
def test(x) printd(x) : x = 4 : printd(x)
|
def test(x) printd(x) : x = 4 : printd(x)
|
||||||
|
|
||||||
test(123) {% endhighlight %}
|
test(123)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
When run, this example prints "123" and then "4", showing that we did
|
When run, this example prints "123" and then "4", showing that we did
|
||||||
actually mutate the value! Okay, we have now officially implemented our
|
actually mutate the value! Okay, we have now officially implemented our
|
||||||
|
|
@ -541,21 +596,31 @@ generator. The first step for adding our new 'var/in' construct is to
|
||||||
extend the lexer. As before, this is pretty trivial, the code looks like
|
extend the lexer. As before, this is pretty trivial, the code looks like
|
||||||
this:
|
this:
|
||||||
|
|
||||||
{% highlight python %} ... class UnaryToken(object): pass class
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
... class UnaryToken(object): pass class
|
||||||
VarToken(object): pass ... def Tokenize(string): ... elif identifier ==
|
VarToken(object): pass ... def Tokenize(string): ... elif identifier ==
|
||||||
'unary': yield UnaryToken() elif identifier == 'var': yield VarToken()
|
'unary': yield UnaryToken() elif identifier == 'var': yield VarToken()
|
||||||
else: yield IdentifierToken(identifier) {% endhighlight %}
|
else: yield IdentifierToken(identifier)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The next step is to define the AST node that we will construct. For
|
The next step is to define the AST node that we will construct. For
|
||||||
var/in, it looks like this:
|
var/in, it looks like this:
|
||||||
|
|
||||||
{% highlight python %} # Expression class for var/in. class
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Expression class for var/in. class
|
||||||
VarExpressionNode(ExpressionNode):
|
VarExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, variables, body): self.variables = variables
|
def **init**\ (self, variables, body): self.variables = variables
|
||||||
self.body = body
|
self.body = body
|
||||||
|
|
||||||
def CodeGen(self): ... {% endhighlight %}
|
def CodeGen(self): ...
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
var/in allows a list of names to be defined all at once, and each name
|
var/in allows a list of names to be defined all at once, and each name
|
||||||
can optionally have an initializer value. As such, we capture this
|
can optionally have an initializer value. As such, we capture this
|
||||||
|
|
@ -565,7 +630,10 @@ allowed to access the variables defined by the var/in.
|
||||||
With this in place, we can define the parser pieces. The first thing we
|
With this in place, we can define the parser pieces. The first thing we
|
||||||
do is add it as a primary expression:
|
do is add it as a primary expression:
|
||||||
|
|
||||||
{% highlight python %} # primary ::= # dentifierexpr \| numberexpr \|
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# primary ::= # dentifierexpr \| numberexpr \|
|
||||||
parenexpr \| ifexpr \| forexpr \| varexpr def ParsePrimary(self): if
|
parenexpr \| ifexpr \| forexpr \| varexpr def ParsePrimary(self): if
|
||||||
isinstance(self.current, IdentifierToken): return
|
isinstance(self.current, IdentifierToken): return
|
||||||
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
self.ParseIdentifierExpr() elif isinstance(self.current, NumberToken):
|
||||||
|
|
@ -574,11 +642,16 @@ return self.ParseIfExpr() elif isinstance(self.current, ForToken):
|
||||||
return self.ParseForExpr() elif isinstance(self.current, VarToken):
|
return self.ParseForExpr() elif isinstance(self.current, VarToken):
|
||||||
return self.ParseVarExpr() elif self.current == CharacterToken('('):
|
return self.ParseVarExpr() elif self.current == CharacterToken('('):
|
||||||
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
return self.ParseParenExpr() else: raise RuntimeError('Unknown token
|
||||||
when expecting an expression.') {% endhighlight %}
|
when expecting an expression.')
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Next we define ParseVarExpr:
|
Next we define ParseVarExpr:
|
||||||
|
|
||||||
{% highlight python %} # varexpr ::= 'var' (identifier ('='
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# varexpr ::= 'var' (identifier ('='
|
||||||
expression)?)+ 'in' expression def ParseVarExpr(self): self.Next() # eat
|
expression)?)+ 'in' expression def ParseVarExpr(self): self.Next() # eat
|
||||||
'var'.
|
'var'.
|
||||||
|
|
||||||
|
|
@ -590,12 +663,17 @@ expression)?)+ 'in' expression def ParseVarExpr(self): self.Next() # eat
|
||||||
if not isinstance(self.current, IdentifierToken):
|
if not isinstance(self.current, IdentifierToken):
|
||||||
raise RuntimeError('Expected identifier after "var".')
|
raise RuntimeError('Expected identifier after "var".')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The first part of this code parses the list of identifier/expr pairs
|
The first part of this code parses the list of identifier/expr pairs
|
||||||
into the local ``variables`` list.
|
into the local ``variables`` list.
|
||||||
|
|
||||||
{% highlight python %} while True: var\_name = self.current.name
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
while True: var_name = self.current.name
|
||||||
self.Next() # eat the identifier.
|
self.Next() # eat the identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -615,12 +693,17 @@ self.Next() # eat the identifier.
|
||||||
if not isinstance(self.current, IdentifierToken):
|
if not isinstance(self.current, IdentifierToken):
|
||||||
raise RuntimeError('Expected identifier after "," in a var expression.')
|
raise RuntimeError('Expected identifier after "," in a var expression.')
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Once all the variables are parsed, we then parse the body and create the
|
Once all the variables are parsed, we then parse the body and create the
|
||||||
AST node:
|
AST node:
|
||||||
|
|
||||||
{% highlight python %} # At this point, we have to have 'in'. if not
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# At this point, we have to have 'in'. if not
|
||||||
isinstance(self.current, InToken): raise RuntimeError('Expected "in"
|
isinstance(self.current, InToken): raise RuntimeError('Expected "in"
|
||||||
keyword after "var".') self.Next() # eat 'in'.
|
keyword after "var".') self.Next() # eat 'in'.
|
||||||
|
|
||||||
|
|
@ -630,14 +713,19 @@ keyword after "var".') self.Next() # eat 'in'.
|
||||||
|
|
||||||
return VarExpressionNode(variables, body)
|
return VarExpressionNode(variables, body)
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Now that we can parse and represent the code, we need to support
|
Now that we can parse and represent the code, we need to support
|
||||||
emission of LLVM IR for it. This code starts out with:
|
emission of LLVM IR for it. This code starts out with:
|
||||||
|
|
||||||
{% highlight python %} class VarExpressionNode(ExpressionNode): ... def
|
|
||||||
CodeGen(self): old\_bindings = {} function =
|
.. code-block:: python
|
||||||
g\_llvm\_builder.basic\_block.function
|
|
||||||
|
class VarExpressionNode(ExpressionNode): ... def
|
||||||
|
CodeGen(self): old_bindings = {} function =
|
||||||
|
g_llvm_builder.basic_block.function
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -663,7 +751,9 @@ g\_llvm\_builder.basic\_block.function
|
||||||
# Remember this binding.
|
# Remember this binding.
|
||||||
g_named_values[var_name] = alloca
|
g_named_values[var_name] = alloca
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Basically it loops over all the variables, installing them one at a
|
Basically it loops over all the variables, installing them one at a
|
||||||
time. For each variable we put into the symbol table, we remember the
|
time. For each variable we put into the symbol table, we remember the
|
||||||
|
|
@ -674,22 +764,32 @@ the initializer, create the alloca, then update the symbol table to
|
||||||
point to it. Once all the variables are installed in the symbol table,
|
point to it. Once all the variables are installed in the symbol table,
|
||||||
we evaluate the body of the var/in expression:
|
we evaluate the body of the var/in expression:
|
||||||
|
|
||||||
{% highlight python %} # Codegen the body, now that all vars are in
|
|
||||||
scope. body = self.body.CodeGen() {% endhighlight %}
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Codegen the body, now that all vars are in
|
||||||
|
scope. body = self.body.CodeGen()
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Finally, before returning, we restore the previous variable bindings:
|
Finally, before returning, we restore the previous variable bindings:
|
||||||
|
|
||||||
{% highlight python %} # Pop all our variables from scope. for var\_name
|
|
||||||
in self.variables: if old\_bindings[var\_name] is not None:
|
.. code-block:: python
|
||||||
g\_named\_values[var\_name] = old\_bindings[var\_name] else: del
|
|
||||||
g\_named\_values[var\_name]
|
# Pop all our variables from scope. for var_name
|
||||||
|
in self.variables: if old_bindings[var_name] is not None:
|
||||||
|
g_named_values[var_name] = old_bindings[var_name] else: del
|
||||||
|
g_named_values[var_name]
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
# Return the body computation.
|
# Return the body computation.
|
||||||
return body
|
return body
|
||||||
|
|
||||||
{% endhighlight %}
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
The end result of all of this is that we get properly scoped variable
|
The end result of all of this is that we get properly scoped variable
|
||||||
definitions, and we even (trivially) allow mutation of them :).
|
definitions, and we even (trivially) allow mutation of them :).
|
||||||
|
|
@ -708,69 +808,52 @@ Full Code Listing # {#code}
|
||||||
Here is the complete code listing for our running example, enhanced with
|
Here is the complete code listing for our running example, enhanced with
|
||||||
mutable variables and var/in support:
|
mutable variables and var/in support:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
import re from llvm.core import Module, Constant, Type, Function,
|
import re from llvm.core import Module, Constant, Type, Function,
|
||||||
Builder from llvm.ee import ExecutionEngine, TargetData from llvm.passes
|
Builder from llvm.ee import ExecutionEngine, TargetData from llvm.passes
|
||||||
import FunctionPassManager
|
import FunctionPassManager
|
||||||
|
|
||||||
from llvm.core import FCMP\_ULT, FCMP\_ONE from llvm.passes import
|
from llvm.core import FCMP_ULT, FCMP_ONE from llvm.passes import
|
||||||
(PASS\_PROMOTE\_MEMORY\_TO\_REGISTER, PASS\_INSTRUCTION\_COMBINING,
|
(PASS_PROMOTE_MEMORY_TO_REGISTER, PASS_INSTRUCTION_COMBINING,
|
||||||
PASS\_REASSOCIATE, PASS\_GVN, PASS\_CFG\_SIMPLIFICATION)
|
PASS_REASSOCIATE, PASS_GVN, PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
Globals
|
Globals
|
||||||
-------
|
-------
|
||||||
|
|
||||||
The LLVM module, which holds all the IR code.
|
# The LLVM module, which holds all the IR code.
|
||||||
=============================================
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
||||||
g\_llvm\_module = Module.new('my cool jit')
|
# The LLVM instruction builder. Created whenever a new function is entered.
|
||||||
|
g_llvm_builder = None
|
||||||
|
|
||||||
The LLVM instruction builder. Created whenever a new function is entered.
|
# A dictionary that keeps track of which values are defined in the current scope
|
||||||
=========================================================================
|
# and what their LLVM representation is.
|
||||||
|
g_named_values = {}
|
||||||
|
|
||||||
g\_llvm\_builder = None
|
# The function optimization passes manager.
|
||||||
|
g_llvm_pass_manager = FunctionPassManager.new(g_llvm_module)
|
||||||
|
|
||||||
A dictionary that keeps track of which values are defined in the current scope
|
# The LLVM execution engine.
|
||||||
==============================================================================
|
g_llvm_executor = ExecutionEngine.new(g_llvm_module)
|
||||||
|
|
||||||
and what their LLVM representation is.
|
# The binary operator precedence chart.
|
||||||
======================================
|
g_binop_precedence = {}
|
||||||
|
|
||||||
g\_named\_values = {}
|
# Creates an alloca instruction in the entry block of the function. This is used
|
||||||
|
# for mutable variables.
|
||||||
The function optimization passes manager.
|
def CreateEntryBlockAlloca(function, var_name): entry =
|
||||||
=========================================
|
function.get_entry_basic_block() builder = Builder.new(entry)
|
||||||
|
builder.position_at_beginning(entry) return
|
||||||
g\_llvm\_pass\_manager = FunctionPassManager.new(g\_llvm\_module)
|
builder.alloca(Type.double(), var_name)
|
||||||
|
|
||||||
The LLVM execution engine.
|
|
||||||
==========================
|
|
||||||
|
|
||||||
g\_llvm\_executor = ExecutionEngine.new(g\_llvm\_module)
|
|
||||||
|
|
||||||
The binary operator precedence chart.
|
|
||||||
=====================================
|
|
||||||
|
|
||||||
g\_binop\_precedence = {}
|
|
||||||
|
|
||||||
Creates an alloca instruction in the entry block of the function. This is used
|
|
||||||
==============================================================================
|
|
||||||
|
|
||||||
for mutable variables.
|
|
||||||
======================
|
|
||||||
|
|
||||||
def CreateEntryBlockAlloca(function, var\_name): entry =
|
|
||||||
function.get\_entry\_basic\_block() builder = Builder.new(entry)
|
|
||||||
builder.position\_at\_beginning(entry) return
|
|
||||||
builder.alloca(Type.double(), var\_name)
|
|
||||||
|
|
||||||
Lexer
|
Lexer
|
||||||
-----
|
-----
|
||||||
|
|
||||||
The lexer yields one of these types for each token.
|
# The lexer yields one of these types for each token.
|
||||||
===================================================
|
|
||||||
|
|
||||||
class EOFToken(object): pass class DefToken(object): pass class
|
class EOFToken(object): pass class DefToken(object): pass class
|
||||||
ExternToken(object): pass class IfToken(object): pass class
|
ExternToken(object): pass class IfToken(object): pass class
|
||||||
ThenToken(object): pass class ElseToken(object): pass class
|
ThenToken(object): pass class ElseToken(object): pass class
|
||||||
|
|
@ -789,11 +872,9 @@ char def **eq**\ (self, other): return isinstance(other, CharacterToken)
|
||||||
and self.char == other.char def **ne**\ (self, other): return not self
|
and self.char == other.char def **ne**\ (self, other): return not self
|
||||||
== other
|
== other
|
||||||
|
|
||||||
Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
=============================================================
|
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
||||||
|
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
||||||
REGEX\_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX\_IDENTIFIER =
|
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX\_COMMENT = re.compile('#.*')
|
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
def Tokenize(string): while string: # Skip whitespace. if
|
||||||
string[0].isspace(): string = string[1:] continue
|
string[0].isspace(): string = string[1:] continue
|
||||||
|
|
@ -849,34 +930,26 @@ yield EOFToken()
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
====================================
|
|
||||||
|
|
||||||
class ExpressionNode(object): pass
|
class ExpressionNode(object): pass
|
||||||
|
|
||||||
Expression class for numeric literals like "1.0".
|
# Expression class for numeric literals like "1.0".
|
||||||
=================================================
|
|
||||||
|
|
||||||
class NumberExpressionNode(ExpressionNode):
|
class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, value): self.value = value
|
def **init**\ (self, value): self.value = value
|
||||||
|
|
||||||
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
Expression class for referencing a variable, like "a".
|
# Expression class for referencing a variable, like "a".
|
||||||
======================================================
|
|
||||||
|
|
||||||
class VariableExpressionNode(ExpressionNode):
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, name): self.name = name
|
def **init**\ (self, name): self.name = name
|
||||||
|
|
||||||
def CodeGen(self): if self.name in g\_named\_values: return
|
def CodeGen(self): if self.name in g_named_values: return
|
||||||
g\_llvm\_builder.load(g\_named\_values[self.name], self.name) else:
|
g_llvm_builder.load(g_named_values[self.name], self.name) else:
|
||||||
raise RuntimeError('Unknown variable name: ' + self.name)
|
raise RuntimeError('Unknown variable name: ' + self.name)
|
||||||
|
|
||||||
Expression class for a binary operator.
|
# Expression class for a binary operator.
|
||||||
=======================================
|
|
||||||
|
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode):
|
class BinaryOperatorExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, operator, left, right): self.operator = operator
|
def **init**\ (self, operator, left, right): self.operator = operator
|
||||||
|
|
@ -918,16 +991,14 @@ a variable.')
|
||||||
function = g_llvm_module.get_function_named('binary' + self.operator)
|
function = g_llvm_module.get_function_named('binary' + self.operator)
|
||||||
return g_llvm_builder.call(function, [left, right], 'binop')
|
return g_llvm_builder.call(function, [left, right], 'binop')
|
||||||
|
|
||||||
Expression class for function calls.
|
# Expression class for function calls.
|
||||||
====================================
|
|
||||||
|
|
||||||
class CallExpressionNode(ExpressionNode):
|
class CallExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, callee, args): self.callee = callee self.args =
|
def **init**\ (self, callee, args): self.callee = callee self.args =
|
||||||
args
|
args
|
||||||
|
|
||||||
def CodeGen(self): # Look up the name in the global module table. callee
|
def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
= g\_llvm\_module.get\_function\_named(self.callee)
|
= g_llvm_module.get_function_named(self.callee)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -939,14 +1010,12 @@ def CodeGen(self): # Look up the name in the global module table. callee
|
||||||
|
|
||||||
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
return g_llvm_builder.call(callee, arg_values, 'calltmp')
|
||||||
|
|
||||||
Expression class for if/then/else.
|
# Expression class for if/then/else.
|
||||||
==================================
|
|
||||||
|
|
||||||
class IfExpressionNode(ExpressionNode):
|
class IfExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, condition, then\_branch, else\_branch):
|
def **init**\ (self, condition, then_branch, else_branch):
|
||||||
self.condition = condition self.then\_branch = then\_branch
|
self.condition = condition self.then_branch = then_branch
|
||||||
self.else\_branch = else\_branch
|
self.else_branch = else_branch
|
||||||
|
|
||||||
def CodeGen(self): condition = self.condition.CodeGen()
|
def CodeGen(self): condition = self.condition.CodeGen()
|
||||||
|
|
||||||
|
|
@ -992,13 +1061,11 @@ def CodeGen(self): condition = self.condition.CodeGen()
|
||||||
|
|
||||||
return phi
|
return phi
|
||||||
|
|
||||||
Expression class for for/in.
|
# Expression class for for/in.
|
||||||
============================
|
|
||||||
|
|
||||||
class ForExpressionNode(ExpressionNode):
|
class ForExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, loop\_variable, start, end, step, body):
|
def **init**\ (self, loop_variable, start, end, step, body):
|
||||||
self.loop\_variable = loop\_variable self.start = start self.end = end
|
self.loop_variable = loop_variable self.start = start self.end = end
|
||||||
self.step = step self.body = body
|
self.step = step self.body = body
|
||||||
|
|
||||||
def CodeGen(self): # Output this as: # var = alloca double # ... # start
|
def CodeGen(self): # Output this as: # var = alloca double # ... # start
|
||||||
|
|
@ -1076,28 +1143,24 @@ endloop # outloop:
|
||||||
# for expr always returns 0.0.
|
# for expr always returns 0.0.
|
||||||
return Constant.real(Type.double(), 0)
|
return Constant.real(Type.double(), 0)
|
||||||
|
|
||||||
Expression class for a unary operator.
|
# Expression class for a unary operator.
|
||||||
======================================
|
|
||||||
|
|
||||||
class UnaryExpressionNode(ExpressionNode):
|
class UnaryExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, operator, operand): self.operator = operator
|
def **init**\ (self, operator, operand): self.operator = operator
|
||||||
self.operand = operand
|
self.operand = operand
|
||||||
|
|
||||||
def CodeGen(self): operand = self.operand.CodeGen() function =
|
def CodeGen(self): operand = self.operand.CodeGen() function =
|
||||||
g\_llvm\_module.get\_function\_named('unary' + self.operator) return
|
g_llvm_module.get_function_named('unary' + self.operator) return
|
||||||
g\_llvm\_builder.call(function, [operand], 'unop')
|
g_llvm_builder.call(function, [operand], 'unop')
|
||||||
|
|
||||||
Expression class for var/in.
|
|
||||||
============================
|
|
||||||
|
|
||||||
|
# Expression class for var/in.
|
||||||
class VarExpressionNode(ExpressionNode):
|
class VarExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def **init**\ (self, variables, body): self.variables = variables
|
def **init**\ (self, variables, body): self.variables = variables
|
||||||
self.body = body
|
self.body = body
|
||||||
|
|
||||||
def CodeGen(self): old\_bindings = {} function =
|
def CodeGen(self): old_bindings = {} function =
|
||||||
g\_llvm\_builder.basic\_block.function
|
g_llvm_builder.basic_block.function
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -1136,27 +1199,21 @@ g\_llvm\_builder.basic\_block.function
|
||||||
# Return the body computation.
|
# Return the body computation.
|
||||||
return body
|
return body
|
||||||
|
|
||||||
This class represents the "prototype" for a function, which captures its name,
|
# This class represents the "prototype" for a function, which captures its name,
|
||||||
==============================================================================
|
# and its argument names (thus implicitly the number of arguments the function
|
||||||
|
# takes), as well as if it is an operator.
|
||||||
and its argument names (thus implicitly the number of arguments the function
|
|
||||||
============================================================================
|
|
||||||
|
|
||||||
takes), as well as if it is an operator.
|
|
||||||
========================================
|
|
||||||
|
|
||||||
class PrototypeNode(object):
|
class PrototypeNode(object):
|
||||||
|
|
||||||
def **init**\ (self, name, args, is\_operator=False, precedence=0):
|
def **init**\ (self, name, args, is_operator=False, precedence=0):
|
||||||
self.name = name self.args = args self.is\_operator = is\_operator
|
self.name = name self.args = args self.is_operator = is_operator
|
||||||
self.precedence = precedence
|
self.precedence = precedence
|
||||||
|
|
||||||
def IsBinaryOp(self): return self.is\_operator and len(self.args) == 2
|
def IsBinaryOp(self): return self.is_operator and len(self.args) == 2
|
||||||
|
|
||||||
def GetOperatorName(self): assert self.is\_operator return self.name[-1]
|
def GetOperatorName(self): assert self.is_operator return self.name[-1]
|
||||||
|
|
||||||
def CodeGen(self): # Make the function type, eg. double(double,double).
|
def CodeGen(self): # Make the function type, eg. double(double,double).
|
||||||
funct\_type = Type.function( Type.double(), [Type.double()] \*
|
funct_type = Type.function( Type.double(), [Type.double()] \*
|
||||||
len(self.args), False)
|
len(self.args), False)
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1186,20 +1243,18 @@ len(self.args), False)
|
||||||
|
|
||||||
# Create an alloca for each argument and register the argument in the
|
# Create an alloca for each argument and register the argument in the
|
||||||
symbol # table so that references to it will succeed. def
|
symbol # table so that references to it will succeed. def
|
||||||
CreateArgumentAllocas(self, function): for arg\_name, arg in
|
CreateArgumentAllocas(self, function): for arg_name, arg in
|
||||||
zip(self.args, function.args): alloca = CreateEntryBlockAlloca(function,
|
zip(self.args, function.args): alloca = CreateEntryBlockAlloca(function,
|
||||||
arg\_name) g\_llvm\_builder.store(arg, alloca)
|
arg_name) g_llvm_builder.store(arg, alloca)
|
||||||
g\_named\_values[arg\_name] = alloca
|
g_named_values[arg_name] = alloca
|
||||||
|
|
||||||
This class represents a function definition itself.
|
|
||||||
===================================================
|
|
||||||
|
|
||||||
|
# This class represents a function definition itself.
|
||||||
class FunctionNode(object):
|
class FunctionNode(object):
|
||||||
|
|
||||||
def **init**\ (self, prototype, body): self.prototype = prototype
|
def **init**\ (self, prototype, body): self.prototype = prototype
|
||||||
self.body = body
|
self.body = body
|
||||||
|
|
||||||
def CodeGen(self): # Clear scope. g\_named\_values.clear()
|
def CodeGen(self): # Clear scope. g_named_values.clear()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -1252,10 +1307,10 @@ self.current = self.tokens.next()
|
||||||
# Gets the precedence of the current token, or -1 if the token is not a
|
# Gets the precedence of the current token, or -1 if the token is not a
|
||||||
binary # operator. def GetCurrentTokenPrecedence(self): if
|
binary # operator. def GetCurrentTokenPrecedence(self): if
|
||||||
isinstance(self.current, CharacterToken): return
|
isinstance(self.current, CharacterToken): return
|
||||||
g\_binop\_precedence.get(self.current.char, -1) else: return -1
|
g_binop_precedence.get(self.current.char, -1) else: return -1
|
||||||
|
|
||||||
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
# identifierexpr ::= identifier \| identifier '(' expression\* ')' def
|
||||||
ParseIdentifierExpr(self): identifier\_name = self.current.name
|
ParseIdentifierExpr(self): identifier_name = self.current.name
|
||||||
self.Next() # eat identifier.
|
self.Next() # eat identifier.
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1404,7 +1459,7 @@ isinstance(self.current, VarToken): return self.ParseVarExpr() elif
|
||||||
self.current == CharacterToken('('): return self.ParseParenExpr() else:
|
self.current == CharacterToken('('): return self.ParseParenExpr() else:
|
||||||
raise RuntimeError('Unknown token when expecting an expression.')
|
raise RuntimeError('Unknown token when expecting an expression.')
|
||||||
|
|
||||||
# unary ::= primary \| unary\_operator unary def ParseUnary(self): # If
|
# unary ::= primary \| unary_operator unary def ParseUnary(self): # If
|
||||||
the current token is not an operator, it must be a primary expression.
|
the current token is not an operator, it must be a primary expression.
|
||||||
if (not isinstance(self.current, CharacterToken) or self.current in
|
if (not isinstance(self.current, CharacterToken) or self.current in
|
||||||
[CharacterToken('('), CharacterToken(',')]): return self.ParsePrimary()
|
[CharacterToken('('), CharacterToken(',')]): return self.ParsePrimary()
|
||||||
|
|
@ -1416,8 +1471,8 @@ if (not isinstance(self.current, CharacterToken) or self.current in
|
||||||
self.Next() # eat the operator.
|
self.Next() # eat the operator.
|
||||||
return UnaryExpressionNode(operator, self.ParseUnary())
|
return UnaryExpressionNode(operator, self.ParseUnary())
|
||||||
|
|
||||||
# binoprhs ::= (binary\_operator unary)\* def ParseBinOpRHS(self, left,
|
# binoprhs ::= (binary_operator unary)\* def ParseBinOpRHS(self, left,
|
||||||
left\_precedence): # If this is a binary operator, find its precedence.
|
left_precedence): # If this is a binary operator, find its precedence.
|
||||||
while True: precedence = self.GetCurrentTokenPrecedence()
|
while True: precedence = self.GetCurrentTokenPrecedence()
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
@ -1448,14 +1503,14 @@ self.ParseUnary() return self.ParseBinOpRHS(left, 0)
|
||||||
# prototype # ::= id '(' id\* ')' # ::= binary LETTER number? (id, id) #
|
# prototype # ::= id '(' id\* ')' # ::= binary LETTER number? (id, id) #
|
||||||
::= unary LETTER (id) def ParsePrototype(self): precedence = None if
|
::= unary LETTER (id) def ParsePrototype(self): precedence = None if
|
||||||
isinstance(self.current, IdentifierToken): kind = 'normal'
|
isinstance(self.current, IdentifierToken): kind = 'normal'
|
||||||
function\_name = self.current.name self.Next() # eat function name. elif
|
function_name = self.current.name self.Next() # eat function name. elif
|
||||||
isinstance(self.current, UnaryToken): kind = 'unary' self.Next() # eat
|
isinstance(self.current, UnaryToken): kind = 'unary' self.Next() # eat
|
||||||
'unary'. if not isinstance(self.current, CharacterToken): raise
|
'unary'. if not isinstance(self.current, CharacterToken): raise
|
||||||
RuntimeError('Expected an operator after "unary".') function\_name =
|
RuntimeError('Expected an operator after "unary".') function_name =
|
||||||
'unary' + self.current.char self.Next() # eat the operator. elif
|
'unary' + self.current.char self.Next() # eat the operator. elif
|
||||||
isinstance(self.current, BinaryToken): kind = 'binary' self.Next() # eat
|
isinstance(self.current, BinaryToken): kind = 'binary' self.Next() # eat
|
||||||
'binary'. if not isinstance(self.current, CharacterToken): raise
|
'binary'. if not isinstance(self.current, CharacterToken): raise
|
||||||
RuntimeError('Expected an operator after "binary".') function\_name =
|
RuntimeError('Expected an operator after "binary".') function_name =
|
||||||
'binary' + self.current.char self.Next() # eat the operator. if
|
'binary' + self.current.char self.Next() # eat the operator. if
|
||||||
isinstance(self.current, NumberToken): if not 1 <= self.current.value <=
|
isinstance(self.current, NumberToken): if not 1 <= self.current.value <=
|
||||||
100: raise RuntimeError('Invalid precedence: must be in range [1,
|
100: raise RuntimeError('Invalid precedence: must be in range [1,
|
||||||
|
|
@ -1504,8 +1559,8 @@ def HandleExtern(self): self.Handle(self.ParseExtern, 'Read an extern:')
|
||||||
|
|
||||||
def HandleTopLevelExpression(self): try: function =
|
def HandleTopLevelExpression(self): try: function =
|
||||||
self.ParseTopLevelExpr().CodeGen() result =
|
self.ParseTopLevelExpr().CodeGen() result =
|
||||||
g\_llvm\_executor.run\_function(function, []) print 'Evaluated to:',
|
g_llvm_executor.run_function(function, []) print 'Evaluated to:',
|
||||||
result.as\_real(Type.double()) except Exception, e: raise#print
|
result.as_real(Type.double()) except Exception, e: raise#print
|
||||||
'Error:', e try: self.Next() # Skip for error recovery. except: pass
|
'Error:', e try: self.Next() # Skip for error recovery. except: pass
|
||||||
|
|
||||||
def Handle(self, function, message): try: print message,
|
def Handle(self, function, message): try: print message,
|
||||||
|
|
@ -1517,25 +1572,25 @@ Main driver code.
|
||||||
|
|
||||||
def main(): # Set up the optimizer pipeline. Start with registering info
|
def main(): # Set up the optimizer pipeline. Start with registering info
|
||||||
about how the # target lays out data structures.
|
about how the # target lays out data structures.
|
||||||
g\_llvm\_pass\_manager.add(g\_llvm\_executor.target\_data) # Promote
|
g_llvm_pass_manager.add(g_llvm_executor.target_data) # Promote
|
||||||
allocas to registers.
|
allocas to registers.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_PROMOTE\_MEMORY\_TO\_REGISTER) # Do
|
g_llvm_pass_manager.add(PASS_PROMOTE_MEMORY_TO_REGISTER) # Do
|
||||||
simple "peephole" optimizations and bit-twiddling optzns.
|
simple "peephole" optimizations and bit-twiddling optzns.
|
||||||
g\_llvm\_pass\_manager.add(PASS\_INSTRUCTION\_COMBINING) # Reassociate
|
g_llvm_pass_manager.add(PASS_INSTRUCTION_COMBINING) # Reassociate
|
||||||
expressions. g\_llvm\_pass\_manager.add(PASS\_REASSOCIATE) # Eliminate
|
expressions. g_llvm_pass_manager.add(PASS_REASSOCIATE) # Eliminate
|
||||||
Common SubExpressions. g\_llvm\_pass\_manager.add(PASS\_GVN) # Simplify
|
Common SubExpressions. g_llvm_pass_manager.add(PASS_GVN) # Simplify
|
||||||
the control flow graph (deleting unreachable blocks, etc).
|
the control flow graph (deleting unreachable blocks, etc).
|
||||||
g\_llvm\_pass\_manager.add(PASS\_CFG\_SIMPLIFICATION)
|
g_llvm_pass_manager.add(PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
g\_llvm\_pass\_manager.initialize()
|
g_llvm_pass_manager.initialize()
|
||||||
|
|
||||||
# Install standard binary operators. # 1 is lowest possible precedence.
|
# Install standard binary operators. # 1 is lowest possible precedence.
|
||||||
40 is the highest. g\_binop\_precedence['='] = 2
|
40 is the highest. g_binop_precedence['='] = 2
|
||||||
g\_binop\_precedence['<'] = 10 g\_binop\_precedence['+'] = 20
|
g_binop_precedence['<'] = 10 g_binop_precedence['+'] = 20
|
||||||
g\_binop\_precedence['-'] = 20 g\_binop\_precedence['\*'] = 40
|
g_binop_precedence['-'] = 20 g_binop_precedence['\*'] = 40
|
||||||
|
|
||||||
# Run the main "interpreter loop". while True: print 'ready<', try: raw
|
# Run the main "interpreter loop". while True: print 'ready<', try: raw
|
||||||
= raw\_input() except KeyboardInterrupt: break
|
= raw_input() except KeyboardInterrupt: break
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
|
|
@ -1551,11 +1606,6 @@ g\_binop\_precedence['-'] = 20 g\_binop\_precedence['\*'] = 40
|
||||||
else:
|
else:
|
||||||
parser.HandleTopLevelExpression()
|
parser.HandleTopLevelExpression()
|
||||||
|
|
||||||
# Print out all of the generated code. print '', g\_llvm\_module
|
# Print out all of the generated code. print '', g_llvm_module
|
||||||
|
|
||||||
if **name** == '**main**\ ': main() {% endhighlight %}
|
if **name** == '**main**\ ': main()
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
**`Next: Conclusion and other useful LLVM
|
|
||||||
tidbits <PythonLangImpl8.html>`_**
|
|
||||||
|
|
|
||||||
|
|
@ -1,277 +0,0 @@
|
||||||
*****************************************************************
|
|
||||||
Chapter 8: Conclusion and other useful LLVM tidbits
|
|
||||||
*****************************************************************
|
|
||||||
|
|
||||||
Written by `Chris Lattner <mailto:sabre@nondot.org>`_
|
|
||||||
|
|
||||||
|
|
||||||
Tutorial Conclusion # {#conclusion}
|
|
||||||
===================================
|
|
||||||
|
|
||||||
Welcome to the the final chapter of the `Implementing a language with
|
|
||||||
LLVM <http://www.llvm.org/docs/tutorial/index.html>`_ tutorial. In the
|
|
||||||
course of this tutorial, we have grown our little Kaleidoscope language
|
|
||||||
from being a useless toy, to being a semi-interesting (but probably
|
|
||||||
still useless) toy. :)
|
|
||||||
|
|
||||||
It is interesting to see how far we've come, and how little code it has
|
|
||||||
taken. We built the entire lexer, parser, AST, code generator, and an
|
|
||||||
interactive run-loop (with a JIT!) by-hand in under 540 lines of
|
|
||||||
(non-comment/non-blank) code.
|
|
||||||
|
|
||||||
Our little language supports a couple of interesting features: it
|
|
||||||
supports user defined binary and unary operators, it uses JIT
|
|
||||||
compilation for immediate evaluation, and it supports a few control flow
|
|
||||||
constructs with SSA construction.
|
|
||||||
|
|
||||||
Part of the idea of this tutorial was to show you how easy and fun it
|
|
||||||
can be to define, build, and play with languages. Building a compiler
|
|
||||||
need not be a scary or mystical process! Now that you've seen some of
|
|
||||||
the basics, I strongly encourage you to take the code and hack on it.
|
|
||||||
For example, try adding:
|
|
||||||
|
|
||||||
- **global variables** -- While global variables have questional value
|
|
||||||
in modern software engineering, they are often useful when putting
|
|
||||||
together quick little hacks like the Kaleidoscope compiler itself.
|
|
||||||
Fortunately, our current setup makes it very easy to add global
|
|
||||||
variables: just have value lookup check to see if an unresolved
|
|
||||||
variable is in the global variable symbol table before rejecting it.
|
|
||||||
To create a new global variable, make an instance of the LLVM
|
|
||||||
``GlobalVariable`` class.
|
|
||||||
|
|
||||||
- **typed variables** -- Kaleidoscope currently only supports variables
|
|
||||||
of type double. This gives the language a very nice elegance, because
|
|
||||||
only supporting one type means that you never have to specify types.
|
|
||||||
Different languages have different ways of handling this. The easiest
|
|
||||||
way is to require the user to specify types for every variable
|
|
||||||
definition, and record the type of the variable in the symbol table
|
|
||||||
along with its Value\*.
|
|
||||||
|
|
||||||
- **arrays, structs, vectors, etc** -- Once you add types, you can
|
|
||||||
start extending the type system in all sorts of interesting ways.
|
|
||||||
Simple arrays are very easy and are quite useful for many different
|
|
||||||
applications. Adding them is mostly an exercise in learning how the
|
|
||||||
LLVM
|
|
||||||
`getelementptr <http://www.llvm.org/docs/LangRef.html#i_getelementptr>`_
|
|
||||||
instruction works: it is so nifty/unconventional, it `has its own
|
|
||||||
FAQ <http://www.llvm.org/docs/GetElementPtr.html>`_! If you add
|
|
||||||
support for recursive types (e.g. linked lists), make sure to read
|
|
||||||
the `section in the LLVM Programmer's
|
|
||||||
Manual <http://www.llvm.org/docs/ProgrammersManual.html#TypeResolve>`_
|
|
||||||
that describes how to construct them.
|
|
||||||
|
|
||||||
- **standard runtime** -- Our current language allows the user to
|
|
||||||
access arbitrary external functions, and we use it for things like
|
|
||||||
"putchard". As you extend the language to add higher-level
|
|
||||||
constructs, often these constructs make the most sense if they are
|
|
||||||
lowered to calls into a language-supplied runtime. For example, if
|
|
||||||
you add hash tables to the language, it would probably make sense to
|
|
||||||
add the routines to a runtime, instead of inlining them all the way.
|
|
||||||
|
|
||||||
- **memory management** -- Currently we can only access the stack in
|
|
||||||
Kaleidoscope. It would also be useful to be able to allocate heap
|
|
||||||
memory, either with calls to the standard libc malloc/free interface
|
|
||||||
or with a garbage collector. If you would like to use garbage
|
|
||||||
collection, note that LLVM fully supports `Accurate Garbage
|
|
||||||
Collection <http://www.llvm.org/docs/GarbageCollection.html>`_
|
|
||||||
including algorithms that move objects and need to scan/update the
|
|
||||||
stack.
|
|
||||||
|
|
||||||
- **debugger support** -- LLVM supports generation of `DWARF Debug
|
|
||||||
info <http://www.llvm.org/docs/SourceLevelDebugging.html>`_ which is
|
|
||||||
understood by common debuggers like GDB. Adding support for debug
|
|
||||||
info is fairly straightforward. The best way to understand it is to
|
|
||||||
compile some C/C++ code with "``llvm-gcc -g -O0``\ " and taking a
|
|
||||||
look at what it produces.
|
|
||||||
|
|
||||||
- **exception handling support** - LLVM supports generation of `zero
|
|
||||||
cost exceptions <http://www.llvm.org/docs/ExceptionHandling.html>`_
|
|
||||||
which interoperate with code compiled in other languages. You could
|
|
||||||
also generate code by implicitly making every function return an
|
|
||||||
error value and checking it. You could also make explicit use of
|
|
||||||
setjmp/longjmp. There are many different ways to go here.
|
|
||||||
|
|
||||||
- **object orientation, generics, database access, complex numbers,
|
|
||||||
geometric programming, ...** -- Really, there is no end of crazy
|
|
||||||
features that you can add to the language.
|
|
||||||
|
|
||||||
- **unusual domains** -- We've been talking about applying LLVM to a
|
|
||||||
domain that many people are interested in: building a compiler for a
|
|
||||||
specific language. However, there are many other domains that can use
|
|
||||||
compiler technology that are not typically considered. For example,
|
|
||||||
LLVM has been used to implement OpenGL graphics acceleration,
|
|
||||||
translate C++ code to ActionScript, and many other cute and clever
|
|
||||||
things. Maybe you will be the first to JIT compile a regular
|
|
||||||
expression interpreter into native code with LLVM?
|
|
||||||
|
|
||||||
Have fun - try doing something crazy and unusual. Building a language
|
|
||||||
like everyone else always has, is much less fun than trying something a
|
|
||||||
little crazy or off the wall and seeing how it turns out. If you get
|
|
||||||
stuck or want to talk about it, feel free to email the `llvmdev mailing
|
|
||||||
list <http://lists.cs.uiuc.edu/mailman/listinfo/llvmdev>`_: it has lots
|
|
||||||
of people who are interested in languages and are often willing to help
|
|
||||||
out.
|
|
||||||
|
|
||||||
Before we end this tutorial, I want to talk about some "tips and tricks"
|
|
||||||
for generating LLVM IR. These are some of the more subtle things that
|
|
||||||
may not be obvious, but are very useful if you want to take advantage of
|
|
||||||
LLVM's capabilities.
|
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
Properties of the LLVM IR # {#llvmirproperties}
|
|
||||||
===============================================
|
|
||||||
|
|
||||||
We have a couple common questions about code in the LLVM IR form - let's
|
|
||||||
just get these out of the way right now, shall we?
|
|
||||||
|
|
||||||
Target Independence ## {#targetindep}
|
|
||||||
-------------------------------------
|
|
||||||
|
|
||||||
Kaleidoscope is an example of a "portable language": any program written
|
|
||||||
in Kaleidoscope will work the same way on any target that it runs on.
|
|
||||||
Many other languages have this property, e.g. LISP, Java, Haskell,
|
|
||||||
Javascript, Python, etc. (note that while these languages are portable,
|
|
||||||
not all their libraries are).
|
|
||||||
|
|
||||||
One nice aspect of LLVM is that it is often capable of preserving target
|
|
||||||
independence in the IR: you can take the LLVM IR for a
|
|
||||||
Kaleidoscope-compiled program and run it on any target that LLVM
|
|
||||||
supports, even emitting C code and compiling that on targets that LLVM
|
|
||||||
doesn't support natively. You can trivially tell that the Kaleidoscope
|
|
||||||
compiler generates target-independent code because it never queries for
|
|
||||||
any target-specific information when generating code.
|
|
||||||
|
|
||||||
The fact that LLVM provides a compact, target-independent,
|
|
||||||
representation for code gets a lot of people excited. Unfortunately,
|
|
||||||
these people are usually thinking about C or a language from the C
|
|
||||||
family when they are asking questions about language portability. I say
|
|
||||||
"unfortunately", because there is really no way to make (fully general)
|
|
||||||
C code portable, other than shipping the source code around (and of
|
|
||||||
course, C source code is not actually portable in general either - ever
|
|
||||||
port a really old application from 32- to 64-bits?).
|
|
||||||
|
|
||||||
The problem with C (again, in its full generality) is that it is heavily
|
|
||||||
laden with target specific assumptions. As one simple example, the
|
|
||||||
preprocessor often destructively removes target-independence from the
|
|
||||||
code when it processes the input text:
|
|
||||||
|
|
||||||
{% highlight c %} #ifdef **i386** int X = 1; #else int X = 42; #endif {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
While it is possible to engineer more and more complex solutions to
|
|
||||||
problems like this, it cannot be solved in full generality in a way that
|
|
||||||
is better than shipping the actual source code.
|
|
||||||
|
|
||||||
That said, there are interesting subsets of C that can be made portable.
|
|
||||||
If you are willing to fix primitive types to a fixed size (say int =
|
|
||||||
32-bits, and long = 64-bits), don't care about ABI compatibility with
|
|
||||||
existing binaries, and are willing to give up some other minor features,
|
|
||||||
you can have portable code. This can make sense for specialized domains
|
|
||||||
such as an in-kernel language.
|
|
||||||
|
|
||||||
Safety Guarantees ## {#safety}
|
|
||||||
------------------------------
|
|
||||||
|
|
||||||
Many of the languages above are also "safe" languages: it is impossible
|
|
||||||
for a program written in Java to corrupt its address space and crash the
|
|
||||||
process (assuming the JVM has no bugs). Safety is an interesting
|
|
||||||
property that requires a combination of language design, runtime
|
|
||||||
support, and often operating system support.
|
|
||||||
|
|
||||||
It is certainly possible to implement a safe language in LLVM, but LLVM
|
|
||||||
IR does not itself guarantee safety. The LLVM IR allows unsafe pointer
|
|
||||||
casts, use after free bugs, buffer over-runs, and a variety of other
|
|
||||||
problems. Safety needs to be implemented as a layer on top of LLVM and,
|
|
||||||
conveniently, several groups have investigated this. Ask on the `llvmdev
|
|
||||||
mailing list <http://lists.cs.uiuc.edu/mailman/listinfo/llvmdev>`_ if
|
|
||||||
you are interested in more details.
|
|
||||||
|
|
||||||
Language-Specific Optimizations ## {#langspecific}
|
|
||||||
--------------------------------------------------
|
|
||||||
|
|
||||||
One thing about LLVM that turns off many people is that it does not
|
|
||||||
solve all the world's problems in one system (sorry 'world hunger',
|
|
||||||
someone else will have to solve you some other day). One specific
|
|
||||||
complaint is that people perceive LLVM as being incapable of performing
|
|
||||||
high-level language-specific optimization: LLVM "loses too much
|
|
||||||
information".
|
|
||||||
|
|
||||||
Unfortunately, this is really not the place to give you a full and
|
|
||||||
unified version of "Chris Lattner's theory of compiler design". Instead,
|
|
||||||
I'll make a few observations:
|
|
||||||
|
|
||||||
First, you're right that LLVM does lose information. For example, as of
|
|
||||||
this writing, there is no way to distinguish in the LLVM IR whether an
|
|
||||||
SSA-value came from a C "int" or a C "long" on an ILP32 machine (other
|
|
||||||
than debug info). Both get compiled down to an 'i32' value and the
|
|
||||||
information about what it came from is lost. The more general issue
|
|
||||||
here, is that the LLVM type system uses "structural equivalence" instead
|
|
||||||
of "name equivalence". Another place this surprises people is if you
|
|
||||||
have two types in a high-level language that have the same structure
|
|
||||||
(e.g. two different structs that have a single int field): these types
|
|
||||||
will compile down into a single LLVM type and it will be impossible to
|
|
||||||
tell what it came from.
|
|
||||||
|
|
||||||
Second, while LLVM does lose information, LLVM is not a fixed target: we
|
|
||||||
continue to enhance and improve it in many different ways. In addition
|
|
||||||
to adding new features (LLVM did not always support exceptions or debug
|
|
||||||
info), we also extend the IR to capture important information for
|
|
||||||
optimization (e.g. whether an argument is sign or zero extended,
|
|
||||||
information about pointers aliasing, etc). Many of the enhancements are
|
|
||||||
user-driven: people want LLVM to include some specific feature, so they
|
|
||||||
go ahead and extend it.
|
|
||||||
|
|
||||||
Third, it is *possible and easy* to add language-specific optimizations,
|
|
||||||
and you have a number of choices in how to do it. As one trivial
|
|
||||||
example, it is easy to add language-specific optimization passes that
|
|
||||||
"know" things about code compiled for a language. In the case of the C
|
|
||||||
family, there is an optimization pass that "knows" about the standard C
|
|
||||||
library functions. If you call "exit(0)" in main(), it knows that it is
|
|
||||||
safe to optimize that into "return 0;" because C specifies what the
|
|
||||||
'exit' function does.
|
|
||||||
|
|
||||||
In addition to simple library knowledge, it is possible to embed a
|
|
||||||
variety of other language-specific information into the LLVM IR. If you
|
|
||||||
have a specific need and run into a wall, please bring the topic up on
|
|
||||||
the llvmdev list. At the very worst, you can always treat LLVM as if it
|
|
||||||
were a "dumb code generator" and implement the high-level optimizations
|
|
||||||
you desire in your front-end, on the language-specific AST.
|
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
Tips and Tricks # {#tipsandtricks}
|
|
||||||
==================================
|
|
||||||
|
|
||||||
There is a variety of useful tips and tricks that you come to know after
|
|
||||||
working on/with LLVM that aren't obvious at first glance. Instead of
|
|
||||||
letting everyone rediscover them, this section talks about some of these
|
|
||||||
issues.
|
|
||||||
|
|
||||||
Implementing portable offsetof/sizeof ## {#offsetofsizeof}
|
|
||||||
----------------------------------------------------------
|
|
||||||
|
|
||||||
One interesting thing that comes up, if you are trying to keep the code
|
|
||||||
generated by your compiler "target independent", is that you often need
|
|
||||||
to know the size of some LLVM type or the offset of some field in an
|
|
||||||
llvm structure. For example, you might need to pass the size of a type
|
|
||||||
into a function that allocates memory.
|
|
||||||
|
|
||||||
Unfortunately, this can vary widely across targets: for example the
|
|
||||||
width of a pointer is trivially target-specific. However, there is a
|
|
||||||
`clever way to use the getelementptr
|
|
||||||
instruction <http://nondot.org/sabre/LLVMNotes/SizeOf-OffsetOf-VariableSizedStructs.txt>`_
|
|
||||||
that allows you to compute this in a portable way.
|
|
||||||
|
|
||||||
Garbage Collected Stack Frames ## {#gcstack}
|
|
||||||
--------------------------------------------
|
|
||||||
|
|
||||||
Some languages want to explicitly manage their stack frames, often so
|
|
||||||
that they are garbage collected or to allow easy implementation of
|
|
||||||
closures. There are often better ways to implement these features than
|
|
||||||
explicit stack frames, but `LLVM does support
|
|
||||||
them <http://nondot.org/sabre/LLVMNotes/ExplicitlyManagedStackFrames.txt>`_,
|
|
||||||
if you want. It requires your front-end to convert the code into
|
|
||||||
`Continuation Passing
|
|
||||||
Style <http://en.wikipedia.org/wiki/Continuation-passing_style>`_ and
|
|
||||||
the use of tail calls (which LLVM also supports).
|
|
||||||
|
|
@ -11,7 +11,10 @@ created from Python constants. A constant expression is also a constant
|
||||||
etc) can be specified, to yield a new ``Constant`` object. Let's see
|
etc) can be specified, to yield a new ``Constant`` object. Let's see
|
||||||
some examples:
|
some examples:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
ti = Type.int() # a 32-bit int type
|
ti = Type.int() # a 32-bit int type
|
||||||
|
|
||||||
|
|
@ -24,9 +27,7 @@ r1 = Constant.real(tr, "3.141592") # create from a string r2 =
|
||||||
Constant.real(tr, 1.61803399) # create from a Python float {%
|
Constant.real(tr, 1.61803399) # create from a Python float {%
|
||||||
endhighlight %}
|
endhighlight %}
|
||||||
|
|
||||||
llvm.core.Constant
|
# llvm.core.Constant
|
||||||
==================
|
|
||||||
|
|
||||||
- This will become a table of contents (this text will be scraped).
|
- This will become a table of contents (this text will be scraped).
|
||||||
{:toc}
|
{:toc}
|
||||||
|
|
||||||
|
|
@ -324,9 +325,7 @@ Shuffle vector constant ``k`` based on vector constants ``k2`` and
|
||||||
|
|
||||||
--------------
|
--------------
|
||||||
|
|
||||||
Other Constant Classes
|
# Other Constant Classes
|
||||||
======================
|
|
||||||
|
|
||||||
The following subclasses of ``Constant`` do not provide additional
|
The following subclasses of ``Constant`` do not provide additional
|
||||||
methods, **they serve only to provide richer type information.**
|
methods, **they serve only to provide richer type information.**
|
||||||
|
|
||||||
|
|
@ -347,8 +346,8 @@ LLVM IR \|
|
||||||
These types are helpful in ``isinstance`` checks, like so:
|
These types are helpful in ``isinstance`` checks, like so:
|
||||||
|
|
||||||
{% highlight python %} ti = Type.int(32) k1 = Constant.int(ti, 42) #
|
{% highlight python %} ti = Type.int(32) k1 = Constant.int(ti, 42) #
|
||||||
int32\_t k1 = 42; k2 = Constant.array(ti, [k1, k1]) # int32\_t k2[] = {
|
int32_t k1 = 42; k2 = Constant.array(ti, [k1, k1]) # int32_t k2[] = {
|
||||||
k1, k1 };
|
k1, k1 };
|
||||||
|
|
||||||
assert isinstance(k1, ConstantInt) assert isinstance(k2, ConstantArray)
|
assert isinstance(k1, ConstantInt) assert isinstance(k2, ConstantArray)
|
||||||
{% endhighlight %}
|
|
||||||
|
|
|
||||||
|
|
@ -39,14 +39,10 @@ Returns an iterable object that yields `Type <llvm.core.Type.html>`_
|
||||||
objects that represent, in order, the types of the arguments accepted by
|
objects that represent, in order, the types of the arguments accepted by
|
||||||
the function. Used like this:
|
the function. Used like this:
|
||||||
|
|
||||||
{% highlight python %} func\_type = Type.function( Type.int(), [
|
|
||||||
Type.int(), Type.int() ] ) for arg in func\_type.args: assert arg.kind
|
|
||||||
== TYPE\_INTEGER assert arg == Type.int() assert func\_type.arg\_count
|
|
||||||
== len(func\_type.args) {% endhighlight %}
|
|
||||||
|
|
||||||
``arg_count``
|
.. code-block:: python
|
||||||
~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
func_type = Type.function( Type.int(), [
|
||||||
|
Type.int(), Type.int() ] ) for arg in func_type.args: assert arg.kind
|
||||||
The number of arguments. Same as ``len(obj.args)``, but faster.
|
== TYPE_INTEGER assert arg == Type.int() assert func_type.arg_count
|
||||||
|
== len(func_type.args)
|
||||||
|
|
|
||||||
|
|
@ -11,14 +11,15 @@ marked as constants. Global variables can be created either by using the
|
||||||
``add_global_variable`` method of the `Module <llvm.core.Module.html>`_
|
``add_global_variable`` method of the `Module <llvm.core.Module.html>`_
|
||||||
class, or by using the static method ``GlobalVariable.new``.
|
class, or by using the static method ``GlobalVariable.new``.
|
||||||
|
|
||||||
{% highlight python %} # create a global variable using
|
|
||||||
add\_global\_variable method gv1 =
|
|
||||||
module\_obj.add\_global\_variable(Type.int(), "gv1")
|
|
||||||
|
|
||||||
or equivalently, using a static constructor method
|
.. code-block:: python
|
||||||
==================================================
|
|
||||||
|
|
||||||
gv2 = GlobalVariable.new(module\_obj, Type.int(), "gv2") {% endhighlight
|
# create a global variable using
|
||||||
|
add_global_variable method gv1 =
|
||||||
|
module_obj.add_global_variable(Type.int(), "gv1")
|
||||||
|
|
||||||
|
# or equivalently, using a static constructor method
|
||||||
|
gv2 = GlobalVariable.new(module_obj, Type.int(), "gv2") {% endhighlight
|
||||||
%}
|
%}
|
||||||
|
|
||||||
Existing global variables of a module can be accessed by name using
|
Existing global variables of a module can be accessed by name using
|
||||||
|
|
@ -27,77 +28,12 @@ Existing global variables of a module can be accessed by name using
|
||||||
via iterating over the property ``module_obj.global_variables``.
|
via iterating over the property ``module_obj.global_variables``.
|
||||||
|
|
||||||
{% highlight python %} # retrieve a reference to the global variable
|
{% highlight python %} # retrieve a reference to the global variable
|
||||||
gv1, # using the get\_global\_variable\_named method gv1 =
|
gv1, # using the get_global_variable_named method gv1 =
|
||||||
module\_obj.get\_global\_variable\_named("gv1")
|
module_obj.get_global_variable_named("gv1")
|
||||||
|
|
||||||
or equivalently, using the static ``get`` method:
|
# or equivalently, using the static ``get`` method:
|
||||||
=================================================
|
gv2 = GlobalVariable.get(module_obj, "gv2")
|
||||||
|
|
||||||
gv2 = GlobalVariable.get(module\_obj, "gv2")
|
# list all global variables in a module
|
||||||
|
for gv in module_obj.global_variables: print gv.name, "of type",
|
||||||
list all global variables in a module
|
gv.type
|
||||||
=====================================
|
|
||||||
|
|
||||||
for gv in module\_obj.global\_variables: print gv.name, "of type",
|
|
||||||
gv.type {% endhighlight %}
|
|
||||||
|
|
||||||
The initializer for a global variable can be set by assigning to the
|
|
||||||
``initializer`` property of the object. The ``is_global_constant``
|
|
||||||
property can be used to indicate that the variable is a global constant.
|
|
||||||
|
|
||||||
Global variables can be delete using the ``delete`` method. Do not use
|
|
||||||
the object after calling ``delete`` on it.
|
|
||||||
|
|
||||||
{% highlight python %} # add an initializer 10 (32-bit integer)
|
|
||||||
gv.initializer = Constant.int( Type.int(), 10 )
|
|
||||||
|
|
||||||
delete the global
|
|
||||||
=================
|
|
||||||
|
|
||||||
gv.delete() # DO NOT dereference \`gv' beyond this point! gv = None {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
llvm.core.GlobalVariable
|
|
||||||
========================
|
|
||||||
|
|
||||||
Base Class
|
|
||||||
----------
|
|
||||||
|
|
||||||
- `llvm.core.GlobalValue <llvm.core.GlobalValue.html>`_
|
|
||||||
|
|
||||||
Static Constructors
|
|
||||||
-------------------
|
|
||||||
|
|
||||||
``new(module_obj, ty, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Create a global variable named ``name`` of type ``ty`` in the module
|
|
||||||
``module_obj`` and return a ``GlobalVariable`` object that represents
|
|
||||||
it.
|
|
||||||
|
|
||||||
``get(module_obj, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Return a ``GlobalVariable`` object to represent the global variable
|
|
||||||
named ``name`` in the module ``module_obj`` or raise ``LLVMException``
|
|
||||||
if such a variable does not exist.
|
|
||||||
|
|
||||||
Properties
|
|
||||||
----------
|
|
||||||
|
|
||||||
``initializer``
|
|
||||||
~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
The intializer of the variable. Set to
|
|
||||||
`llvm.core.Constant <llvm.core.Constant.html>`_ (or derived). Gets the
|
|
||||||
initializer constant, or ``None`` if none exists. ``global_constant``
|
|
||||||
``True`` if the variable is a global constant, ``False`` otherwise.
|
|
||||||
|
|
||||||
Methods
|
|
||||||
-------
|
|
||||||
|
|
||||||
``delete()``
|
|
||||||
~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Deletes the global variable from it's module. **Do not hold any
|
|
||||||
references to this object after calling ``delete`` on it.**
|
|
||||||
|
|
|
||||||
|
|
@ -8,226 +8,12 @@ Modules are top-level container objects. You need to create a module
|
||||||
object first, before you can add global variables, aliases or functions.
|
object first, before you can add global variables, aliases or functions.
|
||||||
Modules are created using the static method ``Module.new``:
|
Modules are created using the static method ``Module.new``:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
from llvm import \* from llvm.core import \*
|
from llvm import \* from llvm.core import \*
|
||||||
|
|
||||||
create a module
|
# create a module
|
||||||
===============
|
my_module = Module.new('my_module')
|
||||||
|
|
||||||
my\_module = Module.new('my\_module') {% endhighlight %}
|
|
||||||
|
|
||||||
The constructor of the Module class should *not* be used to instantiate
|
|
||||||
a Module object. This is a common feature for all llvmpy classes.
|
|
||||||
|
|
||||||
**Convention**
|
|
||||||
|
|
||||||
*All* llvmpy objects are instantiated using static methods of
|
|
||||||
corresponding classes. Constructors *should not* be used.
|
|
||||||
|
|
||||||
The argument ``my_module`` is a module identifier (a plain string).
|
|
||||||
A module can also be constructed via deserialization from a bit code
|
|
||||||
file, using the static method ``from_bitcode``. This method takes a
|
|
||||||
file-like object as argument, i.e., it should have a ``read()``
|
|
||||||
method that returns the entire data in a single call, as is the case
|
|
||||||
with the builtin file object. Here is an example:
|
|
||||||
|
|
||||||
{% highlight python %} # create a module from a bit code file bcfile =
|
|
||||||
file("test.bc") my\_module = Module.from\_bitcode(bcfile) {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
There is corresponding serialization method also, called ``to_bitcode``:
|
|
||||||
|
|
||||||
{% highlight python %} # write out a bit code file from the module
|
|
||||||
bcfile = file("test.bc", "w") my\_module.to\_bitcode(bcfile) {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
Modules can also be constructed from LLVM assembly files (``.ll``
|
|
||||||
files). The static method ``from_assembly`` can be used for this.
|
|
||||||
Similar to the ``from_bitcode`` method, this one also takes a file-like
|
|
||||||
object as argument:
|
|
||||||
|
|
||||||
{% highlight python %} # create a module from an assembly file llfile =
|
|
||||||
file("test.ll") my\_module = Module.from\_assembly(llfile) {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
Modules can be converted into their assembly representation by
|
|
||||||
stringifying them (see below).
|
|
||||||
|
|
||||||
--------------
|
|
||||||
|
|
||||||
llvm.core.Module
|
|
||||||
================
|
|
||||||
|
|
||||||
- This will become a table of contents (this text will be scraped).
|
|
||||||
{:toc}
|
|
||||||
|
|
||||||
Static Constructors
|
|
||||||
-------------------
|
|
||||||
|
|
||||||
``new(module_id)``
|
|
||||||
~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Create a new ``Module`` instance with given ``module_id``. The
|
|
||||||
``module_id`` should be a string.
|
|
||||||
|
|
||||||
``from_bitcode(fileobj)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Create a new ``Module`` instance by deserializing the bitcode file
|
|
||||||
represented by the file-like object ``fileobj``.
|
|
||||||
|
|
||||||
``from_assembly(fileobj)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Create a new ``Module`` instance by parsing the LLVM assembly file
|
|
||||||
represented by the file-like object ``fileobj``.
|
|
||||||
|
|
||||||
Properties
|
|
||||||
----------
|
|
||||||
|
|
||||||
``data_layout``
|
|
||||||
~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
A string representing the ABI of the platform.
|
|
||||||
|
|
||||||
``target``
|
|
||||||
~~~~~~~~~~
|
|
||||||
|
|
||||||
A string like ``i386-pc-linux-gnu`` or ``i386-pc-solaris2.8``.
|
|
||||||
|
|
||||||
``pointer_size``
|
|
||||||
~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
The size in bits of pointers, of the target platform. A value of zero
|
|
||||||
represents ``llvm::Module::AnyPointerSize``.
|
|
||||||
|
|
||||||
``global_variables``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
An iterable that yields
|
|
||||||
`GlobalVariable <llvm.core.GlobalVariable.html>`_ objects, that
|
|
||||||
represent the global variables of the module.
|
|
||||||
|
|
||||||
``functions``
|
|
||||||
~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
An iterable that yields `Function <llvm.core.Function.html>`_ objects,
|
|
||||||
that represent functions in the module.
|
|
||||||
|
|
||||||
``id``
|
|
||||||
~~~~~~
|
|
||||||
|
|
||||||
A string that represents the module identifier (name).
|
|
||||||
|
|
||||||
Methods
|
|
||||||
-------
|
|
||||||
|
|
||||||
``get_type_named(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Return a `StructType <llvm.core.StructType.html>`_ object for the given
|
|
||||||
name.
|
|
||||||
|
|
||||||
The definition of this method was changed to work with LLVM 3.0+, in
|
|
||||||
which the type system was rewritten. See `LLVM
|
|
||||||
Blog <http://blog.llvm.org/2011/11/llvm-30-type-system-rewrite.html>`_.
|
|
||||||
|
|
||||||
{% comment %} ++++++++REMOVED+++++++++++ ### ``add_type_name(name, ty)``
|
|
||||||
|
|
||||||
Add an alias (typedef) for the type ``ty`` with the name ``name``.
|
|
||||||
|
|
||||||
``delete_type_name(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Delete an alias with the name ``name``. ++++++++END-REMOVED+++++++++++
|
|
||||||
{% endcomment %}
|
|
||||||
|
|
||||||
``add_global_variable(ty, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Add a global variable of the type ``ty`` with the name ``name``. Returns
|
|
||||||
a `GlobalVariable <llvm.core.GlobalVariable.html>`_ object.
|
|
||||||
|
|
||||||
``get_global_variable_named(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Get a `GlobalVariable <llvm.core.GlobalVariable.html>`_ object
|
|
||||||
corresponding to the global variable with the name ``name``. Raises
|
|
||||||
``LLVMException`` if such a variable does not exist.
|
|
||||||
|
|
||||||
``add_library(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Add a dependent library to the Module. This only adds a name to a list
|
|
||||||
of dependent library. **No linking is performed**.
|
|
||||||
|
|
||||||
``add_function(ty, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Add a function named ``name`` with the function type ``ty``. ``ty`` must
|
|
||||||
of an object of type `FunctionType <llvm.core.FunctionType.html>`_.
|
|
||||||
|
|
||||||
``get_function_named(name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Get a `Function <llvm.core.Function.html>`_ object corresponding to the
|
|
||||||
function with the name ``name``. Raises ``LLVMException`` if such a
|
|
||||||
function does not exist.
|
|
||||||
|
|
||||||
``get_or_insert_function(ty, name)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Like ``get_function_named``, but adds the function first, if not present
|
|
||||||
(like ``add_function``).
|
|
||||||
|
|
||||||
``verify()``
|
|
||||||
~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Verify the correctness of the module. Raises ``LLVMException`` on
|
|
||||||
errors.
|
|
||||||
|
|
||||||
``to_bitcode(fileobj)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Write the bitcode representation of the module to the file-like object
|
|
||||||
``fileobj``.
|
|
||||||
|
|
||||||
``link_in(other)``
|
|
||||||
~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Link in another module ``other`` into this module. Global variables,
|
|
||||||
functions etc. are matched and resolved. The ``other`` module is no
|
|
||||||
longer valid and should not be used after this operation. This API might
|
|
||||||
be replaced with a full-fledged Linker class in the future.
|
|
||||||
|
|
||||||
Special Methods
|
|
||||||
---------------
|
|
||||||
|
|
||||||
``__str__``
|
|
||||||
~~~~~~~~~~~
|
|
||||||
|
|
||||||
``Module`` objects can be stringified into it's LLVM assembly language
|
|
||||||
representation.
|
|
||||||
|
|
||||||
``__eq__``
|
|
||||||
~~~~~~~~~~
|
|
||||||
|
|
||||||
``Module`` objects can be compared for equality. Internally, this
|
|
||||||
converts both arguments into their LLVM assembly representations and
|
|
||||||
compares the resultant strings.
|
|
||||||
|
|
||||||
**Convention**
|
|
||||||
|
|
||||||
*All* llvmpy objects (where it makes sense), when stringified,
|
|
||||||
return the LLVM assembly representation. ``print module_obj`` for
|
|
||||||
example, prints the LLVM assembly form of the entire module.
|
|
||||||
|
|
||||||
Such objects, when compared for equality, internally compare these
|
|
||||||
string representations.
|
|
||||||
|
|
|
||||||
|
|
@ -1,84 +0,0 @@
|
||||||
+---------------------------------+
|
|
||||||
| layout: page |
|
|
||||||
+---------------------------------+
|
|
||||||
| title: StructType (llvm.core) |
|
|
||||||
+---------------------------------+
|
|
||||||
|
|
||||||
llvm.core.StructType
|
|
||||||
====================
|
|
||||||
|
|
||||||
Base Class
|
|
||||||
----------
|
|
||||||
|
|
||||||
- `llvm.core.Type <llvm.core.Type.html>`_
|
|
||||||
|
|
||||||
Methods
|
|
||||||
-------
|
|
||||||
|
|
||||||
``set_body(self, elems, packed=False)``
|
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
Define the body for opaque identified structure.
|
|
||||||
|
|
||||||
``elems`` is an iterable of `llvm.core.Type <llvm.core.Type.html>`_ If
|
|
||||||
``packed`` is ``True``, creates a packed structure.
|
|
||||||
|
|
||||||
Properties
|
|
||||||
----------
|
|
||||||
|
|
||||||
``is_identified``
|
|
||||||
~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
``True`` if this is an identified structure.
|
|
||||||
|
|
||||||
``is_literal``
|
|
||||||
~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
``True`` if this is a literal structure.
|
|
||||||
|
|
||||||
``is_opaque``
|
|
||||||
~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
``True`` if this is an opaque structure. Only identified structure can
|
|
||||||
be opaque.
|
|
||||||
|
|
||||||
``packed``
|
|
||||||
~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
``True`` if the structure is packed (no padding between elements).
|
|
||||||
|
|
||||||
``name``
|
|
||||||
~~~~~~~~
|
|
||||||
|
|
||||||
Use in identified structure. If set to empty, the identified structure
|
|
||||||
is removed from the global context.
|
|
||||||
|
|
||||||
``elements``
|
|
||||||
~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
Returns an iterable object that yields `Type <llvm.core.Type.html>`_
|
|
||||||
objects that represent, in order, the types of the elements of the
|
|
||||||
structure. Used like this:
|
|
||||||
|
|
||||||
{% highlight python %} struct\_type = Type.struct( [ Type.int(),
|
|
||||||
Type.int() ] ) for elem in struct\_type.elements: assert elem.kind ==
|
|
||||||
TYPE\_INTEGER assert elem == Type.int() assert
|
|
||||||
struct\_type.element\_count == len(struct\_type.elements) {%
|
|
||||||
endhighlight %}
|
|
||||||
|
|
||||||
``element_count``
|
|
||||||
~~~~~~~~~~~~~~~~~
|
|
||||||
|
|
||||||
[read-only]
|
|
||||||
|
|
||||||
The number of elements. Same as ``len(obj.elements)``, but faster.
|
|
||||||
|
|
@ -106,40 +106,23 @@ Properties
|
||||||
A value (enum) representing the "type" of the object. It will be one of
|
A value (enum) representing the "type" of the object. It will be one of
|
||||||
the following constants defined in ``llvm.core``:
|
the following constants defined in ``llvm.core``:
|
||||||
|
|
||||||
{% highlight python %} # Warning: do not rely on actual numerical
|
|
||||||
values! TYPE\_VOID = 0 TYPE\_FLOAT = 1 TYPE\_DOUBLE = 2 TYPE\_X86\_FP80
|
.. code-block:: python
|
||||||
= 3 TYPE\_FP128 = 4 TYPE\_PPC\_FP128 = 5 TYPE\_LABEL = 6 TYPE\_INTEGER =
|
|
||||||
7 TYPE\_FUNCTION = 8 TYPE\_STRUCT = 9 TYPE\_ARRAY = 10 TYPE\_POINTER =
|
# Warning: do not rely on actual numerical
|
||||||
11 TYPE\_OPAQUE = 12 TYPE\_VECTOR = 13 TYPE\_METADATA = 14 TYPE\_UNION =
|
values! TYPE_VOID = 0 TYPE_FLOAT = 1 TYPE_DOUBLE = 2 TYPE_X86_FP80
|
||||||
15 {% endhighlight %}
|
= 3 TYPE_FP128 = 4 TYPE_PPC_FP128 = 5 TYPE_LABEL = 6 TYPE_INTEGER =
|
||||||
|
7 TYPE_FUNCTION = 8 TYPE_STRUCT = 9 TYPE_ARRAY = 10 TYPE_POINTER =
|
||||||
|
11 TYPE_OPAQUE = 12 TYPE_VECTOR = 13 TYPE_METADATA = 14 TYPE_UNION =
|
||||||
|
15
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Example:
|
Example:
|
||||||
^^^^^^^^
|
^^^^^^^^
|
||||||
|
|
||||||
{% highlight python %} assert Type.int().kind == TYPE\_INTEGER assert
|
|
||||||
Type.void().kind == TYPE\_VOID {% endhighlight %}
|
|
||||||
|
|
||||||
Methods
|
.. code-block:: python
|
||||||
-------
|
|
||||||
|
|
||||||
``refine``
|
assert Type.int().kind == TYPE_INTEGER assert
|
||||||
~~~~~~~~~~
|
Type.void().kind == TYPE_VOID
|
||||||
|
|
||||||
Used for constructing self-referencing types. See the documentation of
|
|
||||||
`TypeHandle <llvm.core.TypeHandle.html>`_ objects.
|
|
||||||
|
|
||||||
Special Methods
|
|
||||||
---------------
|
|
||||||
|
|
||||||
``__str__``
|
|
||||||
~~~~~~~~~~~
|
|
||||||
|
|
||||||
``Type`` objects can be stringified into it's LLVM assembly language
|
|
||||||
representation.
|
|
||||||
|
|
||||||
``__eq__``
|
|
||||||
~~~~~~~~~~
|
|
||||||
|
|
||||||
``Type`` objects can be compared for equality. Internally, this converts
|
|
||||||
both arguments into their LLVM assembly representations and compares the
|
|
||||||
resultant strings.
|
|
||||||
|
|
|
||||||
|
|
@ -80,15 +80,8 @@ Pythonically, modules are imported with the statement
|
||||||
``import llvm.core``. However, you might find it more convenient to
|
``import llvm.core``. However, you might find it more convenient to
|
||||||
import llvmpy modules thus:
|
import llvmpy modules thus:
|
||||||
|
|
||||||
{% highlight python %} from llvm import \* from llvm.core import \* from
|
|
||||||
llvm.ee import \* from llvm.passes import \* {% endhighlight %}
|
|
||||||
|
|
||||||
This avoids quite some typing. Both conventions work, however.
|
.. code-block:: python
|
||||||
|
|
||||||
**Tip**
|
from llvm import \* from llvm.core import \* from
|
||||||
|
llvm.ee import \* from llvm.passes import \*
|
||||||
Python-style documentation strings (``__doc__``) are present in
|
|
||||||
llvmpy. You can use the ``help()`` of the interactive Python
|
|
||||||
interpreter or the ``object?`` of
|
|
||||||
`IPython <http://ipython.scipy.org/moin/>`_ to get online help.
|
|
||||||
(Note: not complete yet!)
|
|
||||||
|
|
|
||||||
|
|
@ -60,50 +60,43 @@ An Example
|
||||||
|
|
||||||
Here is an example that demonstrates the creation of types:
|
Here is an example that demonstrates the creation of types:
|
||||||
|
|
||||||
{% highlight python %} #!/usr/bin/env python
|
|
||||||
|
|
||||||
integers
|
.. code-block:: python
|
||||||
========
|
|
||||||
|
|
||||||
int\_ty = Type.int() bool\_ty = Type.int(1) int\_64bit = Type.int(64)
|
#!/usr/bin/env python
|
||||||
|
|
||||||
floats
|
# integers
|
||||||
======
|
int_ty = Type.int() bool_ty = Type.int(1) int_64bit = Type.int(64)
|
||||||
|
|
||||||
sprec\_real = Type.float() dprec\_real = Type.double()
|
# floats
|
||||||
|
sprec_real = Type.float() dprec_real = Type.double()
|
||||||
|
|
||||||
arrays and vectors
|
# arrays and vectors
|
||||||
==================
|
intar_ty = Type.array( int_ty, 10 ) # "typedef int intar_ty[10];"
|
||||||
|
twodim = Type.array( intar_ty , 10 ) # "typedef int twodim[10][10];"
|
||||||
|
vec = Type.array( int_ty, 10 )
|
||||||
|
|
||||||
intar\_ty = Type.array( int\_ty, 10 ) # "typedef int intar\_ty[10];"
|
# structures
|
||||||
twodim = Type.array( intar\_ty , 10 ) # "typedef int twodim[10][10];"
|
s1_ty = Type.struct( [ int_ty, sprec_real ] ) # "struct s1_ty { int
|
||||||
vec = Type.array( int\_ty, 10 )
|
|
||||||
|
|
||||||
structures
|
|
||||||
==========
|
|
||||||
|
|
||||||
s1\_ty = Type.struct( [ int\_ty, sprec\_real ] ) # "struct s1\_ty { int
|
|
||||||
v1; float v2; };"
|
v1; float v2; };"
|
||||||
|
|
||||||
pointers
|
# pointers
|
||||||
========
|
intptr_ty = Type.pointer(int_ty) # "typedef int \*intptr_ty;"
|
||||||
|
|
||||||
intptr\_ty = Type.pointer(int\_ty) # "typedef int \*intptr\_ty;"
|
# functions
|
||||||
|
f1 = Type.function( int_ty, [ int_ty ] ) # functions that take 1
|
||||||
|
int_ty and return 1 int_ty
|
||||||
|
|
||||||
functions
|
f2 = Type.function( Type.void(), [ int_ty, int_ty ] ) # functions that
|
||||||
=========
|
take 2 int_tys and return nothing
|
||||||
|
|
||||||
f1 = Type.function( int\_ty, [ int\_ty ] ) # functions that take 1
|
f3 = Type.function( Type.void(), ( int_ty, int_ty ) ) # same as f2;
|
||||||
int\_ty and return 1 int\_ty
|
|
||||||
|
|
||||||
f2 = Type.function( Type.void(), [ int\_ty, int\_ty ] ) # functions that
|
|
||||||
take 2 int\_tys and return nothing
|
|
||||||
|
|
||||||
f3 = Type.function( Type.void(), ( int\_ty, int\_ty ) ) # same as f2;
|
|
||||||
any iterable can be used
|
any iterable can be used
|
||||||
|
|
||||||
fnargs = [ Type.pointer( Type.int(8) ) ] printf = Type.function(
|
fnargs = [ Type.pointer( Type.int(8) ) ] printf = Type.function(
|
||||||
Type.int(), fnargs, True ) # variadic function {% endhighlight %}
|
Type.int(), fnargs, True ) # variadic function
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
--------------
|
--------------
|
||||||
|
|
||||||
|
|
@ -123,16 +116,8 @@ The following code defines a opaque structure, named "mystruct". The
|
||||||
body is defined after the construction using ``StructType.set_body``.
|
body is defined after the construction using ``StructType.set_body``.
|
||||||
The second subtype is a pointer to a "mystruct" type.
|
The second subtype is a pointer to a "mystruct" type.
|
||||||
|
|
||||||
{% highlight python %} ts = Type.opaque('mystruct')
|
|
||||||
ts.set\_body([Type.int(), Type.pointer(ts)]) {% endhighlight %}
|
|
||||||
|
|
||||||
--------------
|
.. code-block:: python
|
||||||
|
|
||||||
**Related Links** `llvm.core.Type <llvm.core.Type.html>`_,
|
ts = Type.opaque('mystruct')
|
||||||
`llvm.core.IntegerType <llvm.core.IntegerType.html>`_,
|
ts.set_body([Type.int(), Type.pointer(ts)])
|
||||||
`llvm.core.FunctionType <llvm.core.FunctionType.html>`_,
|
|
||||||
`llvm.core.StructType <llvm.core.StructType.html>`_,
|
|
||||||
`llvm.core.ArrayType <llvm.core.ArrayType.html>`_,
|
|
||||||
`llvm.core.PointerType <llvm.core.PointerType.html>`_,
|
|
||||||
`llvm.core.VectorType <llvm.core.VectorType.html>`_,
|
|
||||||
`llvm.core.TypeHandle <llvm.core.TypeHandle.html>`_
|
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue