Refactor lexer

This commit is contained in:
Joey Payne 2017-01-23 20:52:43 -07:00
commit 17f7292510

View file

@ -102,20 +102,7 @@ class Lexer(object):
return token return token
def next_token(self): def handle_number(self, cur_line, cur_col):
cur_col = self.col
cur_line = self.line
if self.cur_char:
if self.pos == 0:
indent = self.get_indent()
if indent is not None:
return indent
while self.cur_char.isspace() and self.cur_char:
self.advance()
if self.cur_char.isdigit() or self.cur_char == '.':
num_str = '' num_str = ''
while ((self.cur_char.isdigit() or self.cur_char == '.') and while ((self.cur_char.isdigit() or self.cur_char == '.') and
@ -125,7 +112,11 @@ class Lexer(object):
return Token(TokenType.NUMBER, num_str, cur_line, cur_col) return Token(TokenType.NUMBER, num_str, cur_line, cur_col)
elif self.cur_char == '#': def skip_whitespace(self):
while self.cur_char.isspace() and self.cur_char:
self.advance()
def handle_comments(self, cur_line, cur_col):
self.advance() self.advance()
comment_str = '' comment_str = ''
@ -136,13 +127,13 @@ class Lexer(object):
return Token(TokenType.COMMENT, comment_str, cur_line, cur_col) return Token(TokenType.COMMENT, comment_str, cur_line, cur_col)
elif self.cur_char in Token.keyword_map: def handle_single_keyword(self, cur_line, cur_col):
last = self.cur_char last = self.cur_char
self.advance() self.advance()
return Token(Token.keyword_map[last], last, return Token(Token.keyword_map[last], last,
cur_line, cur_col) cur_line, cur_col)
elif not self.cur_char.isspace(): def handle_ident(self, cur_line, cur_col):
id_str = '' id_str = ''
while (not self.cur_char.isspace() and while (not self.cur_char.isspace() and
@ -164,8 +155,7 @@ class Lexer(object):
else: else:
return Token(TokenType.IDENT, id_str, cur_line, cur_col) return Token(TokenType.IDENT, id_str, cur_line, cur_col)
def get_trailing_dedent(self, cur_line):
if self.indent_stack:
# Extra indents that have not been taken care of (ie: at the # Extra indents that have not been taken care of (ie: at the
# end of an indented file) # end of an indented file)
indent = self.indent_stack.pop() indent = self.indent_stack.pop()
@ -178,6 +168,36 @@ class Lexer(object):
return Token(TokenType.DEDENT, content, cur_line+1, 1) return Token(TokenType.DEDENT, content, cur_line+1, 1)
def get_token(self, cur_col, cur_line):
if self.cur_char.isdigit() or self.cur_char == '.':
return self.handle_number(cur_line, cur_col)
elif self.cur_char == '#':
return self.handle_comments(cur_line, cur_col)
elif self.cur_char in Token.keyword_map:
return self.handle_single_keyword(cur_line, cur_col)
elif not self.cur_char.isspace():
return self.handle_ident(cur_line, cur_col)
def next_token(self):
cur_col = self.col
cur_line = self.line
if self.cur_char:
if self.pos == 0:
indent = self.get_indent()
if indent is not None:
return indent
self.skip_whitespace()
token = self.get_token(cur_line, cur_col)
if token:
return token
if self.indent_stack:
return self.get_trailing_dedent(cur_line)
return Token(TokenType.EOF, '', cur_line+1, 1) return Token(TokenType.EOF, '', cur_line+1, 1)
def tokens(self): def tokens(self):