Files
calc_engine/tokenizer.py
T

138 lines
4.8 KiB
Python

import ptoken as token
from collections import deque
def get_tokens_from_expression_string(expression_string):
# Takes user input (in the format of a string) of an actuarial formula and parses the formula to its components.
# Returns the list of Token objects that represent the string expression.
tokens = deque()
tmp = ""
parsing_number = False
symbols_dic = {} # Dictionary to keep track of the no. of times each variable appears.
variable_is_negative = False
for i in range(0, len(expression_string)):
c = expression_string[i]
if c == '.' or c.isdigit():
if c == '.' and tmp.count('.') == 1:
raise SyntaxError("Invalid grammar, to many periods in number. Found at position " + str (i) + ".")
if len(tokens) > 1:
lookBehind = tokens[-1]
if lookBehind.type == token.TokenType.exp_end:
tokens.append(token.Token("^", token.TokenType.power))
if (i + 1) == len(expression_string):
tokens.append(token.Token(tmp + c, token.TokenType.constant))
parsing_number = True
tmp = tmp + c
continue
elif c == '-' and not parsing_number:
if expression_string[i + 1] == '.' or expression_string[i + 1].isdigit():
if token.TokenType.get_operator(tokens[-1].value) != token.TokenType.unknown:#tokens[-1].type == token.TokenType
#print(tokens[-1].value)
#tokens.append(token.Token('+', token.TokenType.add))
parsing_number = True
tmp = "-"
elif tokens[-1].type == token.TokenType.exp_start:
parsing_number = True
tmp = "-"
elif tokens[-1].type == token.TokenType.exp_end:
tokens.append(token.Token(c, token.TokenType.get_operator(c)))
parsing_number = False
tmp = ""
elif expression_string[i + 1].isalpha():
# Must be a variable.
if tokens[-1].type == token.TokenType.exp_start:
variable_is_negative = True
elif token.TokenType.get_operator(tokens[-1].value) != token.TokenType.unknown:
variable_is_negative = True
elif tokens[-1].type == token.TokenType.exp_end:
tokens.append(token.Token(c, token.TokenType.get_operator(c)))
parsing_number = False
tmp = ""
else:
tokens.append(token.Token(c, token.TokenType.get_operator(c)))
parsing_number = False
tmp = ""
elif c == '(':
if parsing_number:
tokens.append(token.Token(tmp, token.TokenType.constant))
tokens.append(token.Token("*", token.TokenType.multiply))
tokens.append(token.Token("(", token.TokenType.exp_start))
parsing_number = False
tmp = ""
continue
elif c == ')':
if parsing_number:
tokens.append(token.Token(tmp, token.TokenType.constant))
tokens.append(token.Token(")", token.TokenType.exp_end))
parsing_number = False
tmp = ""
continue
elif token.TokenType.get_operator(c) != token.TokenType.unknown:
if parsing_number:
tokens.append(token.Token(tmp, token.TokenType.constant))
tokens.append(token.Token(c, token.TokenType.get_operator(c)))
parsing_number = False
tmp = ""
continue
elif c != ' ':
if len(tokens) > 1:
lookBehind = tokens[-1]
if lookBehind.type == token.TokenType.exp_end:
tokens.append(token.Token("^", token.TokenType.power))
if (i + 1) < len(expression_string):
lookAHead = expression_string[i + 1]
if lookAHead == '(':
tokens.append(token.Token(c, token.TokenType.variable))
tokens.append(token.Token("*", token.TokenType.multiply))
tmp = ""
parsing_number = False
continue
if parsing_number:
tokens.append(token.Token(tmp, token.TokenType.constant))
tokens.append(token.Token("*", token.TokenType.multiply))
# Check to see if the encountered variable, which is effectively a single character,
# has been seen before. If it has been then simply increment the number that represents how many times it's been seen
# else, add a new entry for it in the dictonary 'symbols_dic'.
if c in symbols_dic.keys():
symbols_dic[c] += 1
else:
symbols_dic[c] = 1
# Afterwards, add the variable with the count of the times it's been seen as a token object to the list of tokens.
tokens.append(token.Token(c + str(symbols_dic[c]), token.TokenType.variable, variable_is_negative))
variable_is_negative = False
parsing_number = False
tmp = ""
return tokens
def peek_list(tokens):
# Returns the last Token object in the list, else throws an error.
if tokens:
return tokens[-1]
else:
raise IndexError("The token list is empty.", tokens)
def print_token_list(tokens):
# Prints the list of Tokens to standard out.
for tk in tokens:
if tk.type == token.TokenType.variable:
if tk.is_negative_variable:
print("-%s " %(tk.value), end = "")
else:
print("%s " %(tk.value), end ="")
else:
print("%s " %(tk.value), end = "")
print()
def stringify_token_list(tokens):
# Returns a string that represents the list of Tokens.
text = ''
for tk in tokens:
text += str(tk.value)
text += '\n'
return text