Additional doc fixes.
This commit is contained in:
parent
67ee2792bf
commit
c7cbebab7b
5 changed files with 176 additions and 150 deletions
|
|
@ -267,55 +267,46 @@ ignore the captured match:
|
||||||
|
|
||||||
.. code-block:: python
|
.. code-block:: python
|
||||||
|
|
||||||
# Check if any of the regexes matched and yield
|
# Check if any of the regexes matched and yield
|
||||||
# the appropriate result.
|
# the appropriate result.
|
||||||
if comment_match:
|
if comment_match:
|
||||||
comment = comment_match.group(0)
|
comment = comment_match.group(0)
|
||||||
string = string[len(comment):]
|
string = string[len(comment):]
|
||||||
|
|
||||||
For numbers, we yield the captured match, converted to a float and
|
|
||||||
tagged with the appropriate token type:
|
|
||||||
|
|
||||||
.. code-block:: python
|
|
||||||
|
|
||||||
elif number_match:
|
# For numbers, we yield the captured match, converted to a float and
|
||||||
number = number_match.group(0)
|
# tagged with the appropriate token type:
|
||||||
yield NumberToken(float(number))
|
|
||||||
string = string[len(number):]
|
|
||||||
|
|
||||||
The identifier case is a little more complex. We have to check for
|
elif number_match:
|
||||||
keywords to decide whether we have captured an identifier or a keyword:
|
number = number_match.group(0)
|
||||||
|
yield NumberToken(float(number))
|
||||||
|
string = string[len(number):]
|
||||||
|
|
||||||
.. code-block:: python
|
# The identifier case is a little more complex. We have to check for
|
||||||
|
# keywords to decide whether we have captured an identifier or a keyword:
|
||||||
|
|
||||||
elif identifier_match:
|
elif identifier_match:
|
||||||
identifier = identifier_match.group(0)
|
identifier = identifier_match.group(0)
|
||||||
# Check if we matched a keyword.
|
# Check if we matched a keyword.
|
||||||
if identifier == 'def':
|
if identifier == 'def':
|
||||||
yield DefToken()
|
yield DefToken()
|
||||||
elif identifier == 'extern':
|
elif identifier == 'extern':
|
||||||
yield ExternToken()
|
yield ExternToken()
|
||||||
else:
|
else:
|
||||||
yield IdentifierToken(identifier)
|
yield IdentifierToken(identifier)
|
||||||
string = string[len(identifier):]
|
string = string[len(identifier):]
|
||||||
|
|
||||||
|
|
||||||
Finally, if we haven't recognized a comment, a number of an identifier,
|
# Finally, if we haven't recognized a comment, a number of an identifier,
|
||||||
we yield the current character as an "unknown character" token. This is
|
# we yield the current character as an "unknown character" token. This is
|
||||||
used, for example, for operators like ``+`` or ``*``:
|
# used, for example, for operators like ``+`` or ``*``:
|
||||||
|
|
||||||
|
else: # Yield the unknown character.
|
||||||
|
yield CharacterToken(string[0])
|
||||||
|
string = string[1:]
|
||||||
|
|
||||||
|
|
||||||
.. code-block:: python
|
# Once we're done with the loop, we return a final end-of-file token:
|
||||||
|
|
||||||
else: # Yield the unknown character.
|
|
||||||
yield CharacterToken(string[0])
|
|
||||||
string = string[1:]
|
|
||||||
|
|
||||||
|
|
||||||
Once we're done with the loop, we return a final end-of-file token:
|
yield EOFToken()
|
||||||
|
|
||||||
|
|
||||||
.. code-block:: python
|
|
||||||
|
|
||||||
yield EOFToken()
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -688,8 +688,8 @@ Lexer
|
||||||
return not self == other
|
return not self == other
|
||||||
|
|
||||||
# Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?')
|
REGEX_NUMBER = re.compile('[0-9]+(?:\.[0-9]+)?')
|
||||||
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]\ *')
|
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]*')
|
||||||
REGEX_COMMENT = re.compile('#.*')
|
REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
def Tokenize(string):
|
def Tokenize(string):
|
||||||
|
|
|
||||||
|
|
@ -512,8 +512,8 @@ Lexer
|
||||||
return not self == other
|
return not self == other
|
||||||
|
|
||||||
# Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?')
|
REGEX_NUMBER = re.compile('[0-9]+(?:\.[0-9]+)?')
|
||||||
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]\ *')
|
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]*')
|
||||||
REGEX_COMMENT = re.compile('#.*')
|
REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
def Tokenize(string):
|
def Tokenize(string):
|
||||||
|
|
|
||||||
|
|
@ -909,8 +909,8 @@ Lexer
|
||||||
return not self == other
|
return not self == other
|
||||||
|
|
||||||
# Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?')
|
REGEX_NUMBER = re.compile('[0-9]+(?:\.[0-9]+)?')
|
||||||
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]\ *')
|
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]*')
|
||||||
REGEX_COMMENT = re.compile('#.*')
|
REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
def Tokenize(string):
|
def Tokenize(string):
|
||||||
|
|
|
||||||
|
|
@ -764,16 +764,21 @@ the if/then/else and for expressions:
|
||||||
|
|
||||||
#!/usr/bin/env python
|
#!/usr/bin/env python
|
||||||
|
|
||||||
import re from llvm.core import Module, Constant, Type, Function,
|
import re
|
||||||
Builder from llvm.ee import ExecutionEngine, TargetData from llvm.passes
|
from llvm.core import Module, Constant, Type, Function, Builder
|
||||||
import FunctionPassManager
|
from llvm.ee import ExecutionEngine, TargetData
|
||||||
|
from llvm.passes import FunctionPassManager
|
||||||
|
|
||||||
from llvm.core import FCMP_ULT, FCMP_ONE from llvm.passes import
|
from llvm.core import FCMP_ULT, FCMP_ONE
|
||||||
(PASS_INSTRUCTION_COMBINING, PASS_REASSOCIATE, PASS_GVN,
|
from llvm.passes import (PASS_INSTRUCTION_COMBINING,
|
||||||
PASS_CFG_SIMPLIFICATION)
|
PASS_REASSOCIATE,
|
||||||
|
PASS_GVN,
|
||||||
|
PASS_CFG_SIMPLIFICATION)
|
||||||
|
|
||||||
Globals
|
Globals
|
||||||
-------
|
-------
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
# The LLVM module, which holds all the IR code.
|
# The LLVM module, which holds all the IR code.
|
||||||
g_llvm_module = Module.new('my cool jit')
|
g_llvm_module = Module.new('my cool jit')
|
||||||
|
|
@ -794,126 +799,156 @@ the if/then/else and for expressions:
|
||||||
# The binary operator precedence chart.
|
# The binary operator precedence chart.
|
||||||
g_binop_precedence = {}
|
g_binop_precedence = {}
|
||||||
|
|
||||||
Lexer
|
Lexer
|
||||||
-----
|
-----
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
# The lexer yields one of these types for each token.
|
# The lexer yields one of these types for each token.
|
||||||
class EOFToken(object): pass class DefToken(object): pass class
|
class EOFToken(object):
|
||||||
ExternToken(object): pass class IfToken(object): pass class
|
pass
|
||||||
ThenToken(object): pass class ElseToken(object): pass class
|
class DefToken(object):
|
||||||
ForToken(object): pass class InToken(object): pass class
|
pass
|
||||||
BinaryToken(object): pass class UnaryToken(object): pass
|
class ExternToken(object):
|
||||||
|
pass
|
||||||
|
class IfToken(object):
|
||||||
|
pass
|
||||||
|
class ThenToken(object):
|
||||||
|
pass
|
||||||
|
class ElseToken(object):
|
||||||
|
pass
|
||||||
|
class ForToken(object):
|
||||||
|
pass
|
||||||
|
class InToken(object):
|
||||||
|
pass
|
||||||
|
class BinaryToken(object):
|
||||||
|
pass
|
||||||
|
class UnaryToken(object):
|
||||||
|
pass
|
||||||
|
|
||||||
class IdentifierToken(object): def __init__(self, name): self.name =
|
class IdentifierToken(object): def __init__(self, name):
|
||||||
name
|
self.name = name
|
||||||
|
|
||||||
class NumberToken(object): def __init__(self, value): self.value =
|
class NumberToken(object): def __init__(self, value):
|
||||||
value
|
self.value = value
|
||||||
|
|
||||||
class CharacterToken(object): def __init__(self, char): self.char =
|
class CharacterToken(object):
|
||||||
char def __eq__(self, other): return isinstance(other, CharacterToken)
|
def __init__(self, char):
|
||||||
and self.char == other.char def __ne__(self, other): return not self
|
self.char = char
|
||||||
== other
|
def __eq__(self, other):
|
||||||
|
return isinstance(other, CharacterToken) and self.char == other.char
|
||||||
|
def __ne__(self, other):
|
||||||
|
return not self == other
|
||||||
|
|
||||||
# Regular expressions that tokens and comments of our language.
|
# Regular expressions that tokens and comments of our language.
|
||||||
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER =
|
REGEX_NUMBER = re.compile('[0-9]+(?:\.[0-9]+)?')
|
||||||
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*')
|
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9] *')
|
||||||
|
REGEX_COMMENT = re.compile('#.*')
|
||||||
|
|
||||||
def Tokenize(string): while string: # Skip whitespace. if
|
def Tokenize(string):
|
||||||
string[0].isspace(): string = string[1:] continue
|
while string:
|
||||||
|
# Skip whitespace.
|
||||||
::
|
if string[0].isspace():
|
||||||
|
string = string[1:]
|
||||||
# Run regexes.
|
continue
|
||||||
comment_match = REGEX_COMMENT.match(string)
|
|
||||||
number_match = REGEX_NUMBER.match(string)
|
# Run regexes.
|
||||||
identifier_match = REGEX_IDENTIFIER.match(string)
|
comment_match = REGEX_COMMENT.match(string)
|
||||||
|
number_match = REGEX_NUMBER.match(string)
|
||||||
# Check if any of the regexes matched and yield the appropriate result.
|
identifier_match = REGEX_IDENTIFIER.match(string)
|
||||||
if comment_match:
|
|
||||||
comment = comment_match.group(0)
|
# Check if any of the regexes matched and yield the appropriate result.
|
||||||
string = string[len(comment):]
|
if comment_match:
|
||||||
elif number_match:
|
comment = comment_match.group(0)
|
||||||
number = number_match.group(0)
|
string = string[len(comment):]
|
||||||
yield NumberToken(float(number))
|
elif number_match:
|
||||||
string = string[len(number):]
|
number = number_match.group(0)
|
||||||
elif identifier_match:
|
yield NumberToken(float(number))
|
||||||
identifier = identifier_match.group(0)
|
string = string[len(number):]
|
||||||
# Check if we matched a keyword.
|
elif identifier_match:
|
||||||
if identifier == 'def':
|
identifier = identifier_match.group(0)
|
||||||
yield DefToken()
|
# Check if we matched a keyword.
|
||||||
elif identifier == 'extern':
|
if identifier == 'def':
|
||||||
yield ExternToken()
|
yield DefToken()
|
||||||
elif identifier == 'if':
|
elif identifier == 'extern':
|
||||||
yield IfToken()
|
yield ExternToken()
|
||||||
elif identifier == 'then':
|
elif identifier == 'if':
|
||||||
yield ThenToken()
|
yield IfToken()
|
||||||
elif identifier == 'else':
|
elif identifier == 'then':
|
||||||
yield ElseToken()
|
yield ThenToken()
|
||||||
elif identifier == 'for':
|
elif identifier == 'else':
|
||||||
yield ForToken()
|
yield ElseToken()
|
||||||
elif identifier == 'in':
|
elif identifier == 'for':
|
||||||
yield InToken()
|
yield ForToken()
|
||||||
elif identifier == 'binary':
|
elif identifier == 'in':
|
||||||
yield BinaryToken()
|
yield InToken()
|
||||||
elif identifier == 'unary':
|
elif identifier == 'binary':
|
||||||
yield UnaryToken()
|
yield BinaryToken()
|
||||||
else:
|
elif identifier == 'unary':
|
||||||
yield IdentifierToken(identifier)
|
yield UnaryToken()
|
||||||
string = string[len(identifier):]
|
else:
|
||||||
else:
|
yield IdentifierToken(identifier)
|
||||||
# Yield the ASCII value of the unknown character.
|
string = string[len(identifier):]
|
||||||
yield CharacterToken(string[0])
|
else:
|
||||||
string = string[1:]
|
# Yield the ASCII value of the unknown character.
|
||||||
|
yield CharacterToken(string[0])
|
||||||
|
string = string[1:]
|
||||||
|
|
||||||
yield EOFToken()
|
yield EOFToken()
|
||||||
|
|
||||||
Abstract Syntax Tree (aka Parse Tree)
|
Abstract Syntax Tree (aka Parse Tree)
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
# Base class for all expression nodes.
|
# Base class for all expression nodes.
|
||||||
class ExpressionNode(object): pass
|
class ExpressionNode(object):
|
||||||
|
pass
|
||||||
|
|
||||||
# Expression class for numeric literals like "1.0".
|
# Expression class for numeric literals like "1.0".
|
||||||
class NumberExpressionNode(ExpressionNode):
|
class NumberExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def __init__(self, value): self.value = value
|
def __init__(self, value):
|
||||||
|
self.value = value
|
||||||
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
|
||||||
|
def CodeGen(self): return Constant.real(Type.double(), self.value)
|
||||||
|
|
||||||
# Expression class for referencing a variable, like "a".
|
# Expression class for referencing a variable, like "a".
|
||||||
class VariableExpressionNode(ExpressionNode):
|
class VariableExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def __init__(self, name): self.name = name
|
def __init__(self, name):
|
||||||
|
self.name = name
|
||||||
def CodeGen(self): if self.name in g_named_values: return
|
|
||||||
g_named_values[self.name] else: raise RuntimeError('Unknown variable
|
def CodeGen(self):
|
||||||
name: ' + self.name)
|
if self.name in g_named_values:
|
||||||
|
return g_named_values[self.name]
|
||||||
|
else:
|
||||||
|
raise RuntimeError('Unknown variable name: ' + self.name)
|
||||||
|
|
||||||
# Expression class for a binary operator.
|
# Expression class for a binary operator.
|
||||||
class BinaryOperatorExpressionNode(ExpressionNode):
|
class BinaryOperatorExpressionNode(ExpressionNode):
|
||||||
|
|
||||||
def __init__(self, operator, left, right): self.operator = operator
|
def __init__(self, operator, left, right):
|
||||||
self.left = left self.right = right
|
self.operator = operator
|
||||||
|
self.left = left
|
||||||
def CodeGen(self): left = self.left.CodeGen() right =
|
self.right = right
|
||||||
self.right.CodeGen()
|
|
||||||
|
def CodeGen(self): left = self.left.CodeGen()
|
||||||
::
|
right = self.right.CodeGen()
|
||||||
|
|
||||||
if self.operator == '+':
|
if self.operator == '+':
|
||||||
return g_llvm_builder.fadd(left, right, 'addtmp')
|
return g_llvm_builder.fadd(left, right, 'addtmp')
|
||||||
elif self.operator == '-':
|
elif self.operator == '-':
|
||||||
return g_llvm_builder.fsub(left, right, 'subtmp')
|
return g_llvm_builder.fsub(left, right, 'subtmp')
|
||||||
elif self.operator == '*':
|
elif self.operator == '*':
|
||||||
return g_llvm_builder.fmul(left, right, 'multmp')
|
return g_llvm_builder.fmul(left, right, 'multmp')
|
||||||
elif self.operator == '<':
|
elif self.operator == '<':
|
||||||
result = g_llvm_builder.fcmp(FCMP_ULT, left, right, 'cmptmp')
|
result = g_llvm_builder.fcmp(FCMP_ULT, left, right, 'cmptmp')
|
||||||
# Convert bool 0 or 1 to double 0.0 or 1.0.
|
# Convert bool 0 or 1 to double 0.0 or 1.0.
|
||||||
return g_llvm_builder.uitofp(result, Type.double(), 'booltmp')
|
return g_llvm_builder.uitofp(result, Type.double(), 'booltmp')
|
||||||
else:
|
else:
|
||||||
function = g_llvm_module.get_function_named('binary' + self.operator)
|
function = g_llvm_module.get_function_named('binary' + self.operator)
|
||||||
return g_llvm_builder.call(function, [left, right], 'binop')
|
return g_llvm_builder.call(function, [left, right], 'binop')
|
||||||
|
|
||||||
# Expression class for function calls.
|
# Expression class for function calls.
|
||||||
class CallExpressionNode(ExpressionNode):
|
class CallExpressionNode(ExpressionNode):
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue