i'm writing an Cool language compiler in phyton and i got an syntax error on the parser. The compiler got this files:
lexer.py
from distutils.debug import DEBUG
import ply.lex as lex
from ply.yacc import yacc
# --- Tokenizer
# All tokens must be named in advance.
# D E F I N I T I O N S =========================
# Reserved keywords
reserved_keywords = {
"case": "CASE",
"class": "CLASS",
"else": "ELSE",
"esac": "ESAC",
"fi": "FI",
"if": "IF",
"in": "IN",
"inherits": "INHERITS",
"isvoid": "ISVOID",
"let": "LET",
"loop": "LOOP",
"new": "NEW",
"of": "OF",
"pool": "POOL",
"self": "SELF",
"then": "THEN",
"while": "WHILE"
}
reserved_keywords_values = reserved_keywords.values()
tokens = (
'ASSIGN','NOT', 'EQUAL', 'LESS_THAN', 'LESS_THAN_OR_EQUAL',
'PLUS', 'MINUS', 'MULTIPLY', 'DIVIDE', 'INT_COMP',
'AT', 'DOT', 'ACTION',
'BLOCK_INIT', 'BLOCK_END', 'PARENTESIS_INIT', 'PARENTESIS_END',
'COLON', 'COMMA', 'SEMICOLON',
'ID', 'TYPE',
'INTEGER', 'STRING',
)+ tuple(reserved_keywords_values)
# R U L E S =========================
# A regular expression rule with some action code
def t_INTEGER(t):
r'[0-9]+'
t.value = int(t.value)
return t
def t_STRING(t):
r'"[^"]*"'
t.value = t.value[1:-1]
return t
def t_SINGLECOMMENT(t):
r"\-\-[^\n]*"
pass
def t_NOT(t):
r'[nN][oO][tT]'
return t
def t_TYPE(t):
r'[A-Z][A-Za-z0-9_]*'
return t
def t_ID(t):
r'[a-z][A-Za-z0-9_]*'
lower_type = t.value.lower();
default = 'ID'
t.type = reserved_keywords.get(lower_type, default)
return t
def t_COMMENT(t):
#r'(?:\(\*(?:(?!\*\))(.|\n|\r\n))*\*\))'
r'\(\*'
t.lexer.code_start = t.lexer.lexpos
t.lexer.comment_count = 1
t.lexer.begin('COMMENT')
def t_BOOL(t):
r'true|false'
t.value = True if t.value == 'true' else False
return t
# Define a rule so we can track line numbers
def t_newline(t):
r'\n+'
t.lexer.lineno += len(t.value)
# Error handling rule
def t_error(t):
print("Illegal character '{}'" % t.value[0])
t.lexer.skip(1)
t_ignore = ' \t' + ' ' + '\n'
t_ACTION = r'\=\>'
t_AT = r'\@'
t_BLOCK_INIT = r'\{'
t_BLOCK_END = r'\}'
t_PARENTESIS_INIT = r'\('
t_PARENTESIS_END = r'\)'
t_COLON= r'\:'
t_COMMA=r'\,'
t_DOT=r'\.'
t_SEMICOLON=r'\;'
t_MINUS = r'\-'
t_MULTIPLY = r'\*'
t_PLUS = r'\+'
t_DIVIDE = r'\/'
t_EQUAL = r'\='
t_LESS_THAN = r'\<'
t_LESS_THAN_OR_EQUAL = r'\<\='
t_ASSIGN = r'\<\-'
t_INT_COMP = r'\~'
# L E X E R =========================
lexer = lex.lex()
# data = '''
# class Main inherits IO {
# main(): SELF_TYPE {
# out_string("Hello, World.")
# };
# };
# '''
# lexer.input(data)
# # Build the lexer object
# lex.lex()
###### PROCESS INPUT ######
if __name__ == '__main__':
import sys
sourcefile = sys.argv[1]
with open(sourcefile, 'r') as source:
lex.input(source.read())
# Read tokens
# Tokenize
while True:
tok = lexer.token()
if not tok:
break
print(tok)
parser.py
import ply.yacc as yacc
# Get the token map from the lexer. This is required.
import lexer
import ast_helper as ast
tokens = lexer.tokens
# The precedence of infix binary and prefix unary operations, from highest to lowest, is given by the
# following table:
# .
# @
# ~
# isvoid
# * /
# + -
# <= < =
# not
# <-
# All binary operations are left-associative, with the exception of assignment, which is right-associative,
# and the three comparison operations, which do not associate
def p_program(p):
"""program : classes"""
p[0] = p[1]
def p_classes(p):
"""classes : class
| class classes"""
if len(p) == 2:
p[0] = (p[1],)
elif len(p) == 3:
p[0] = (p[1],) + p[2]
else:
raise SyntaxError('Invalid number of symbols')
def p_class(p):
"""class : CLASS TYPE inheritance BLOCK_INIT features_opt BLOCK_END SEMICOLON"""
p[0] = ast.Type(name=p[2], inherits=p[3], features=p[5])
def p_inheritance(p):
"""inheritance : INHERITS TYPE
| empty"""
if len(p) == 2:
p[0] = p[1]
elif len(p) == 3:
p[0] = p[2]
else:
raise SyntaxError('Invalid number of symbols')
def p_features_opt(p):
"""features_opt : features
| empty"""
if p.slice[1].type == 'empty':
p[0] = []
else:
p[0] = p[1]
def p_features(p):
"""features : feature
| feature features"""
if len(p) == 2:
p[0] = [p[1]]
elif len(p) == 3:
p[0] = [p[1]] + p[2]
else:
raise SyntaxError('Invalid number of symbols')
def p_feature(p):
"""feature : ID PARENTESIS_INIT formals_opt PARENTESIS_END COLON TYPE '{' expr '}' SEMICOLON
| attr_def SEMICOLON"""
if len(p) == 11:
p[0] = ast.Method(ident=ast.Ident(p[1]), type=p[6], formals=p[3], expr=p[8])
elif len(p) == 3:
p[0] = p[1]
else:
raise SyntaxError('Invalid number of symbols')
def p_attr_defs(p):
"""attr_defs : attr_def
| attr_def COMMA attr_defs"""
if len(p) == 2:
p[0] = (p[1],)
elif len(p) == 4:
p[0] = (p[1],) + p[3]
else:
raise SyntaxError('Invalid number of symbols')
def p_attr_def(p):
"""attr_def : ID COLON TYPE assign_opt"""
p[0] = ast.Attribute(ident=ast.Ident(p[1]), type=p[3], expr=p[4])
def p_assign_opt(p):
"""assign_opt : assign
| empty"""
p[0] = p[1]
def p_assign(p):
"""assign : ASSIGN expr"""
p[0] = p[2]
def p_formals_opt(p):
"""formals_opt : formals
| empty"""
if p.slice[1].type == 'empty':
p[0] = tuple()
else:
p[0] = p[1]
def p_formals(p):
"""formals : formal
| formal COMMA formals"""
if len(p) == 2:
p[0] = (p[1],)
elif len(p) == 4:
p[0] = (p[1],) + p[3]
else:
raise SyntaxError('Invalid number of symbols')
def p_formal(p):
"""formal : ID COLON TYPE"""
p[0] = ast.Formal(ident=ast.Ident(p[1]), type=p[3])
def p_params_opt(p):
"""params_opt : params
| empty"""
if p.slice[1].type == 'empty':
p[0] = tuple()
else:
p[0] = p[1]
def p_params(p):
"""params : expr
| expr COMMA params"""
if len(p) == 2:
p[0] = (p[1],)
elif len(p) == 4:
p[0] = (p[1],) + p[3]
else:
raise SyntaxError('Invalid number of symbols')
def p_block(p):
"""block : blockelements"""
p[0] = ast.Block(p[1])
def p_blockelements(p):
"""blockelements : expr SEMICOLON
| expr SEMICOLON blockelements"""
if len(p) == 3:
p[0] = (p[1],)
elif len(p) == 4:
p[0] = (p[1],) + p[3]
else:
raise SyntaxError('Invalid number of symbols')
def p_typeactions(p):
"""typeactions : typeaction
| typeaction typeactions"""
if len(p) == 2:
p[0] = (p[1],)
elif len(p) == 3:
p[0] = (p[1],) + p[2]
else:
raise SyntaxError('Invalid number of symbols')
def p_typeaction(p):
"""typeaction : ID COLON TYPE ACTION expr SEMICOLON"""
p[0] = ast.TypeAction(ident=ast.Ident(p[1]), type=p[3], expr=p[5])
def p_function_call(p):
"""function_call : ID PARENTESIS_INIT params_opt PARENTESIS_END"""
p[0] = ast.FunctionCall(ident=ast.Ident(p[1]), params=p[3])
def p_targettype_opt(p):
"""targettype_opt : targettype
| empty"""
p[0] = p[1]
def p_targettype(p):
"""targettype : AT TYPE"""
p[0] = p[2]
def p_expr_self(p):
"""expr : SELF"""
p[0] = ast.Self(ident=ast.Ident(p[1]))
def p_expr(p):
"""expr : ID assign
| expr targettype_opt DOT function_call
| function_call
| IF expr THEN expr ELSE expr FI
| WHILE expr LOOP expr POOL
| LET attr_defs IN expr
| CASE expr OF typeactions ESAC
| NEW TYPE
| INT_COMP expr
| NOT expr
| ISVOID expr
| expr PLUS expr
| expr MINUS expr
| expr MULTIPLY expr
| expr DIVIDE expr
| expr LESS_THAN expr
| expr LESS_THAN_OR_EQUAL expr
| expr EQUAL expr
| BLOCK_INIT block BLOCK_END
| PARENTESIS_INIT expr PARENTESIS_END
| ID
| INTEGER
| STRING
"""
first_token = p.slice[1].type
second_token = p.slice[2].type if len(p) > 2 else None
third_token = p.slice[3].type if len(p) > 3 else None
if first_token == 'ID':
if second_token is None:
p[0] = ast.Ident(p[1])
elif second_token == 'assign':
p[0] = ast.Assign(ident=ast.Ident(p[1]), expr=p[2])
elif first_token == 'expr':
if len(p) == 4 and third_token == 'expr':
p[0] = ast.BinOp(operator=p[2], left=p[1], right=p[3])
elif third_token == 'DOT':
p[0] = ast.MethodCall(object=p[1], targettype=p[2], method=p[4])
elif first_token == 'function_call':
p[0] = p[1]
elif first_token == 'IF':
p[0] = ast.If(condition=p[2], true=p[4], false=p[6])
elif first_token == 'WHILE':
p[0] = ast.While(condition=p[2], action=p[4])
elif first_token == 'LET':
p[0] = ast.Let(assignments=p[2], expr=p[4])
elif first_token == 'CASE':
p[0] = ast.Case(expr=p[2])
elif first_token == 'NEW':
p[0] = ast.New(p[2])
elif first_token in ['ISVOID', 'INT_COMP', 'NOT']:
p[0] = ast.UnaryOp(operator=p[1], right=p[2])
elif first_token in ['BLOCK_INIT', 'PARENTESIS_INIT']:
p[0] = p[2]
elif first_token in ['INTEGER', 'STRING', 'BOOL']:
p[0] = p[1]
def p_empty(p):
"""empty :"""
p[0] = None
def p_error(p):
if p == None:
token = "end of file"
else:
token = f"{p.type}({p.value}) on line {p.lineno}"
print(f"Syntax error: Unexpected {token}")
# Create parser
parser = yacc.yacc(start='program', debug=True)
# yacc.yacc(debug=True)
def get_ast(sourcefile):
with open(sourcefile, 'r') as source:
t = yacc.parse(source.read())
return t
precedence = (
('right', 'ASSIGN'),
('left', 'NOT'), # WIP: eh realmente left?
('nonassoc', 'LESS_THAN_OR_EQUAL', 'LESS_THAN', 'EQUAL' ), # Nonassociative operators
('left', 'PLUS', 'MINUS'),
('left', 'MULTIPLY', 'DIVIDE'),
('left', 'ISVOID'),
('left', 'INT_COMP'),
('left', 'AT'),
('left', 'DOT')
)
if __name__ == '__main__':
import sys
if len(sys.argv) > 1:
data = open(sys.argv[1], 'r').read()
result = yacc.parse(data)
print(result)
else:
while True:
try:
s = input('coolp> ')
except EOFError:
break
if not s: continue
result = yacc.parse(s)
print(result)
When i run the parser to read an Cool file, the result = yacc.parse(data) statement returns None and an syntax error pops on the console
Cool file:
class Main inherits IO {
main(): SELF_TYPE {
out_string("Hello, World.")
};
};
Error:
Syntax error: Unexpected BLOCK_INIT({) on line 1
Does anyone knows why this is hapenning?