Additional doc fixes.

This commit is contained in:
Maggie Mari 2012-08-20 17:11:26 -05:00
commit c7cbebab7b
5 changed files with 176 additions and 150 deletions

View file

@ -273,20 +273,16 @@ ignore the captured match:
comment = comment_match.group(0) comment = comment_match.group(0)
string = string[len(comment):] string = string[len(comment):]
For numbers, we yield the captured match, converted to a float and # For numbers, we yield the captured match, converted to a float and
tagged with the appropriate token type: # tagged with the appropriate token type:
.. code-block:: python
elif number_match: elif number_match:
number = number_match.group(0) number = number_match.group(0)
yield NumberToken(float(number)) yield NumberToken(float(number))
string = string[len(number):] string = string[len(number):]
The identifier case is a little more complex. We have to check for # The identifier case is a little more complex. We have to check for
keywords to decide whether we have captured an identifier or a keyword: # keywords to decide whether we have captured an identifier or a keyword:
.. code-block:: python
elif identifier_match: elif identifier_match:
identifier = identifier_match.group(0) identifier = identifier_match.group(0)
@ -300,22 +296,17 @@ keywords to decide whether we have captured an identifier or a keyword:
string = string[len(identifier):] string = string[len(identifier):]
Finally, if we haven't recognized a comment, a number of an identifier, # Finally, if we haven't recognized a comment, a number of an identifier,
we yield the current character as an "unknown character" token. This is # we yield the current character as an "unknown character" token. This is
used, for example, for operators like ``+`` or ``*``: # used, for example, for operators like ``+`` or ``*``:
.. code-block:: python
else: # Yield the unknown character. else: # Yield the unknown character.
yield CharacterToken(string[0]) yield CharacterToken(string[0])
string = string[1:] string = string[1:]
Once we're done with the loop, we return a final end-of-file token: # Once we're done with the loop, we return a final end-of-file token:
.. code-block:: python
yield EOFToken() yield EOFToken()

View file

@ -688,8 +688,8 @@ Lexer
return not self == other return not self == other
# Regular expressions that tokens and comments of our language. # Regular expressions that tokens and comments of our language.
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_NUMBER = re.compile('[0-9]+(?:\.[0-9]+)?')
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]*')
REGEX_COMMENT = re.compile('#.*') REGEX_COMMENT = re.compile('#.*')
def Tokenize(string): def Tokenize(string):

View file

@ -512,8 +512,8 @@ Lexer
return not self == other return not self == other
# Regular expressions that tokens and comments of our language. # Regular expressions that tokens and comments of our language.
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_NUMBER = re.compile('[0-9]+(?:\.[0-9]+)?')
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]*')
REGEX_COMMENT = re.compile('#.*') REGEX_COMMENT = re.compile('#.*')
def Tokenize(string): def Tokenize(string):

View file

@ -909,8 +909,8 @@ Lexer
return not self == other return not self == other
# Regular expressions that tokens and comments of our language. # Regular expressions that tokens and comments of our language.
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_NUMBER = re.compile('[0-9]+(?:\.[0-9]+)?')
REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9]*')
REGEX_COMMENT = re.compile('#.*') REGEX_COMMENT = re.compile('#.*')
def Tokenize(string): def Tokenize(string):

View file

@ -764,17 +764,22 @@ the if/then/else and for expressions:
#!/usr/bin/env python #!/usr/bin/env python
import re from llvm.core import Module, Constant, Type, Function, import re
Builder from llvm.ee import ExecutionEngine, TargetData from llvm.passes from llvm.core import Module, Constant, Type, Function, Builder
import FunctionPassManager from llvm.ee import ExecutionEngine, TargetData
from llvm.passes import FunctionPassManager
from llvm.core import FCMP_ULT, FCMP_ONE from llvm.passes import from llvm.core import FCMP_ULT, FCMP_ONE
(PASS_INSTRUCTION_COMBINING, PASS_REASSOCIATE, PASS_GVN, from llvm.passes import (PASS_INSTRUCTION_COMBINING,
PASS_REASSOCIATE,
PASS_GVN,
PASS_CFG_SIMPLIFICATION) PASS_CFG_SIMPLIFICATION)
Globals Globals
------- -------
.. code-block:: python
# The LLVM module, which holds all the IR code. # The LLVM module, which holds all the IR code.
g_llvm_module = Module.new('my cool jit') g_llvm_module = Module.new('my cool jit')
@ -797,32 +802,55 @@ the if/then/else and for expressions:
Lexer Lexer
----- -----
.. code-block:: python
# The lexer yields one of these types for each token. # The lexer yields one of these types for each token.
class EOFToken(object): pass class DefToken(object): pass class class EOFToken(object):
ExternToken(object): pass class IfToken(object): pass class pass
ThenToken(object): pass class ElseToken(object): pass class class DefToken(object):
ForToken(object): pass class InToken(object): pass class pass
BinaryToken(object): pass class UnaryToken(object): pass class ExternToken(object):
pass
class IfToken(object):
pass
class ThenToken(object):
pass
class ElseToken(object):
pass
class ForToken(object):
pass
class InToken(object):
pass
class BinaryToken(object):
pass
class UnaryToken(object):
pass
class IdentifierToken(object): def __init__(self, name): self.name = class IdentifierToken(object): def __init__(self, name):
name self.name = name
class NumberToken(object): def __init__(self, value): self.value = class NumberToken(object): def __init__(self, value):
value self.value = value
class CharacterToken(object): def __init__(self, char): self.char = class CharacterToken(object):
char def __eq__(self, other): return isinstance(other, CharacterToken) def __init__(self, char):
and self.char == other.char def __ne__(self, other): return not self self.char = char
== other def __eq__(self, other):
return isinstance(other, CharacterToken) and self.char == other.char
def __ne__(self, other):
return not self == other
# Regular expressions that tokens and comments of our language. # Regular expressions that tokens and comments of our language.
REGEX_NUMBER = re.compile('[0-9]+(?:.[0-9]+)?') REGEX_IDENTIFIER = REGEX_NUMBER = re.compile('[0-9]+(?:\.[0-9]+)?')
re.compile('[a-zA-Z][a-zA-Z0-9]\ *') REGEX_COMMENT = re.compile('#.*') REGEX_IDENTIFIER = re.compile('[a-zA-Z][a-zA-Z0-9] *')
REGEX_COMMENT = re.compile('#.*')
def Tokenize(string): while string: # Skip whitespace. if def Tokenize(string):
string[0].isspace(): string = string[1:] continue while string:
# Skip whitespace.
:: if string[0].isspace():
string = string[1:]
continue
# Run regexes. # Run regexes.
comment_match = REGEX_COMMENT.match(string) comment_match = REGEX_COMMENT.match(string)
@ -871,35 +899,42 @@ the if/then/else and for expressions:
Abstract Syntax Tree (aka Parse Tree) Abstract Syntax Tree (aka Parse Tree)
------------------------------------- -------------------------------------
.. code-block:: python
# Base class for all expression nodes. # Base class for all expression nodes.
class ExpressionNode(object): pass class ExpressionNode(object):
pass
# Expression class for numeric literals like "1.0". # Expression class for numeric literals like "1.0".
class NumberExpressionNode(ExpressionNode): class NumberExpressionNode(ExpressionNode):
def __init__(self, value): self.value = value def __init__(self, value):
self.value = value
def CodeGen(self): return Constant.real(Type.double(), self.value) def CodeGen(self): return Constant.real(Type.double(), self.value)
# Expression class for referencing a variable, like "a". # Expression class for referencing a variable, like "a".
class VariableExpressionNode(ExpressionNode): class VariableExpressionNode(ExpressionNode):
def __init__(self, name): self.name = name def __init__(self, name):
self.name = name
def CodeGen(self): if self.name in g_named_values: return def CodeGen(self):
g_named_values[self.name] else: raise RuntimeError('Unknown variable if self.name in g_named_values:
name: ' + self.name) return g_named_values[self.name]
else:
raise RuntimeError('Unknown variable name: ' + self.name)
# Expression class for a binary operator. # Expression class for a binary operator.
class BinaryOperatorExpressionNode(ExpressionNode): class BinaryOperatorExpressionNode(ExpressionNode):
def __init__(self, operator, left, right): self.operator = operator def __init__(self, operator, left, right):
self.left = left self.right = right self.operator = operator
self.left = left
self.right = right
def CodeGen(self): left = self.left.CodeGen() right = def CodeGen(self): left = self.left.CodeGen()
self.right.CodeGen() right = self.right.CodeGen()
::
if self.operator == '+': if self.operator == '+':
return g_llvm_builder.fadd(left, right, 'addtmp') return g_llvm_builder.fadd(left, right, 'addtmp')