Parser using yacc in python generating syntax error

Viewed 38

i'm writing an Cool language compiler in phyton and i got an syntax error on the parser. The compiler got this files:

lexer.py

from distutils.debug import DEBUG
import ply.lex as lex
from ply.yacc import yacc

# --- Tokenizer

# All tokens must be named in advance.

# D E F I N I T I O N S =========================

# Reserved keywords

reserved_keywords = {
            "case": "CASE",
            "class": "CLASS",
            "else": "ELSE",
            "esac": "ESAC",
            "fi": "FI",
            "if": "IF",
            "in": "IN",
            "inherits": "INHERITS",
            "isvoid": "ISVOID",
            "let": "LET",
            "loop": "LOOP",
            "new": "NEW",
            "of": "OF",
            "pool": "POOL",
            "self": "SELF",
            "then": "THEN",
            "while": "WHILE"
        }

reserved_keywords_values = reserved_keywords.values()

tokens = ( 
    'ASSIGN','NOT', 'EQUAL', 'LESS_THAN', 'LESS_THAN_OR_EQUAL',
    'PLUS', 'MINUS', 'MULTIPLY', 'DIVIDE', 'INT_COMP',
    'AT', 'DOT', 'ACTION',
    'BLOCK_INIT', 'BLOCK_END', 'PARENTESIS_INIT', 'PARENTESIS_END',
    'COLON', 'COMMA', 'SEMICOLON',
    'ID', 'TYPE',
    'INTEGER', 'STRING',
)+ tuple(reserved_keywords_values)



# R U L E S ========================= 

 # A regular expression rule with some action code

def t_INTEGER(t):
    r'[0-9]+'
    t.value = int(t.value)    
    return t

def t_STRING(t):
    r'"[^"]*"'
    t.value = t.value[1:-1]
    return t

def t_SINGLECOMMENT(t):
    r"\-\-[^\n]*"
    pass

def t_NOT(t):
    r'[nN][oO][tT]'
    return t

def t_TYPE(t):
    r'[A-Z][A-Za-z0-9_]*'
    return t

def t_ID(t):
    r'[a-z][A-Za-z0-9_]*'
    lower_type = t.value.lower();
    default = 'ID'
    t.type = reserved_keywords.get(lower_type, default)
    return t

def t_COMMENT(t):
  #r'(?:\(\*(?:(?!\*\))(.|\n|\r\n))*\*\))'
  r'\(\*'
  t.lexer.code_start = t.lexer.lexpos
  t.lexer.comment_count = 1
  t.lexer.begin('COMMENT')

def t_BOOL(t):
    r'true|false'
    t.value = True if t.value == 'true' else False
    return t

# Define a rule so we can track line numbers
def t_newline(t):
    r'\n+'
    t.lexer.lineno += len(t.value)


# Error handling rule


def t_error(t):
    print("Illegal character '{}'" % t.value[0])
    t.lexer.skip(1)

t_ignore  = ' \t' + ' ' + '\n'

t_ACTION = r'\=\>' 
t_AT = r'\@'
t_BLOCK_INIT = r'\{' 
t_BLOCK_END = r'\}'
t_PARENTESIS_INIT = r'\('
t_PARENTESIS_END = r'\)'
t_COLON= r'\:'
t_COMMA=r'\,'
t_DOT=r'\.'
t_SEMICOLON=r'\;'

t_MINUS = r'\-'         
t_MULTIPLY = r'\*'      
t_PLUS = r'\+'
t_DIVIDE = r'\/'
t_EQUAL = r'\=' 
t_LESS_THAN = r'\<'            
t_LESS_THAN_OR_EQUAL = r'\<\='  
t_ASSIGN = r'\<\-'
t_INT_COMP = r'\~'


# L E X E R ========================= 

lexer = lex.lex()

# data = '''
# class Main inherits IO {
#    main(): SELF_TYPE {
#   out_string("Hello, World.")
#    };
# };
# '''

# lexer.input(data)

# # Build the lexer object
# lex.lex()

###### PROCESS INPUT ######


if __name__ == '__main__':

    import sys
    sourcefile = sys.argv[1]

    with open(sourcefile, 'r') as source:
        lex.input(source.read())

    # Read tokens
    # Tokenize
    while True:
        tok = lexer.token()
        if not tok: 
            break      
        print(tok)

parser.py

import ply.yacc as yacc


# Get the token map from the lexer.  This is required.
import lexer

import ast_helper as ast


tokens = lexer.tokens

# The precedence of infix binary and prefix unary operations, from highest to lowest, is given by the
# following table:
# .
# @
# ~
# isvoid
# * /
# + -
# <= < =
# not
# <-
# All binary operations are left-associative, with the exception of assignment, which is right-associative,
# and the three comparison operations, which do not associate

def p_program(p):
    """program : classes"""
    
    p[0] = p[1]

def p_classes(p):
    """classes : class
               | class classes"""
               
    if len(p) == 2:
        p[0] = (p[1],)
    elif len(p) == 3:
        p[0] = (p[1],) + p[2]
    else:
        raise SyntaxError('Invalid number of symbols')

def p_class(p):
    """class : CLASS TYPE inheritance BLOCK_INIT features_opt BLOCK_END SEMICOLON"""
    
    p[0] = ast.Type(name=p[2], inherits=p[3], features=p[5])

def p_inheritance(p):
    """inheritance : INHERITS TYPE
                   | empty"""
    if len(p) == 2:
        p[0] = p[1]
    elif len(p) == 3:
        p[0] = p[2]
    else:
        raise SyntaxError('Invalid number of symbols')

def p_features_opt(p):
    """features_opt : features
                    | empty"""
    if p.slice[1].type == 'empty':
        p[0] = []
    else:
        p[0] = p[1]

def p_features(p):
    """features : feature
                | feature features"""
    if len(p) == 2:
        p[0] = [p[1]]
    elif len(p) == 3:
        p[0] = [p[1]] + p[2]
    else:
        raise SyntaxError('Invalid number of symbols')

def p_feature(p):
    """feature : ID PARENTESIS_INIT formals_opt PARENTESIS_END COLON TYPE '{' expr '}' SEMICOLON
               | attr_def SEMICOLON"""
    if len(p) == 11:
        p[0] = ast.Method(ident=ast.Ident(p[1]), type=p[6], formals=p[3], expr=p[8])
    elif len(p) == 3:
        p[0] = p[1]
    else:
        raise SyntaxError('Invalid number of symbols')

def p_attr_defs(p):
    """attr_defs : attr_def
                 | attr_def COMMA attr_defs"""
    if len(p) == 2:
        p[0] = (p[1],)
    elif len(p) == 4:
        p[0] = (p[1],) + p[3]
    else:
        raise SyntaxError('Invalid number of symbols')

def p_attr_def(p):
    """attr_def : ID COLON TYPE assign_opt"""
    p[0] = ast.Attribute(ident=ast.Ident(p[1]), type=p[3], expr=p[4])

def p_assign_opt(p):
    """assign_opt : assign
                  | empty"""
    p[0] = p[1]

def p_assign(p):
    """assign : ASSIGN expr"""
    p[0] = p[2]

def p_formals_opt(p):
    """formals_opt : formals
                   | empty"""
    if p.slice[1].type == 'empty':
        p[0] = tuple()
    else:
        p[0] = p[1]

def p_formals(p):
    """formals : formal
               | formal COMMA formals"""
    if len(p) == 2:
        p[0] = (p[1],)
    elif len(p) == 4:
        p[0] = (p[1],) + p[3]
    else:
        raise SyntaxError('Invalid number of symbols')

def p_formal(p):
    """formal : ID COLON TYPE"""
    p[0] = ast.Formal(ident=ast.Ident(p[1]), type=p[3])

def p_params_opt(p):
    """params_opt : params
                  | empty"""
    if p.slice[1].type == 'empty':
        p[0] = tuple()
    else:
        p[0] = p[1]

def p_params(p):
    """params : expr
              | expr COMMA params"""
    if len(p) == 2:
        p[0] = (p[1],)
    elif len(p) == 4:
        p[0] = (p[1],) + p[3]
    else:
        raise SyntaxError('Invalid number of symbols')    

def p_block(p):
    """block : blockelements"""
    p[0] = ast.Block(p[1])

def p_blockelements(p):
    """blockelements : expr SEMICOLON
                     | expr SEMICOLON blockelements"""
    if len(p) == 3:
        p[0] = (p[1],)
    elif len(p) == 4:
        p[0] = (p[1],) + p[3]
    else:
        raise SyntaxError('Invalid number of symbols')    
def p_typeactions(p):
    """typeactions : typeaction
                   | typeaction typeactions"""
    if len(p) == 2:
        p[0] = (p[1],)
    elif len(p) == 3:
        p[0] = (p[1],) + p[2]
    else:
        raise SyntaxError('Invalid number of symbols')

def p_typeaction(p):
    """typeaction : ID COLON TYPE ACTION expr SEMICOLON"""
    p[0] = ast.TypeAction(ident=ast.Ident(p[1]), type=p[3], expr=p[5])

def p_function_call(p):
    """function_call : ID PARENTESIS_INIT params_opt PARENTESIS_END"""
    p[0] = ast.FunctionCall(ident=ast.Ident(p[1]), params=p[3])

def p_targettype_opt(p):
    """targettype_opt : targettype
                      | empty"""
    p[0] = p[1]

def p_targettype(p):
    """targettype : AT TYPE"""
    p[0] = p[2]

def p_expr_self(p):
    """expr  : SELF"""
    p[0] = ast.Self(ident=ast.Ident(p[1]))
    
def p_expr(p):
    """expr : ID assign
            | expr targettype_opt DOT function_call
            | function_call
            | IF expr THEN expr ELSE expr FI
            | WHILE expr LOOP expr POOL
            | LET attr_defs IN expr
            | CASE expr OF typeactions ESAC
            | NEW TYPE
            | INT_COMP expr
            | NOT expr
            | ISVOID expr
            | expr PLUS expr
            | expr MINUS expr
            | expr MULTIPLY expr
            | expr DIVIDE expr
            | expr LESS_THAN expr
            | expr LESS_THAN_OR_EQUAL expr
            | expr EQUAL expr
            | BLOCK_INIT block BLOCK_END
            | PARENTESIS_INIT expr PARENTESIS_END
            | ID
            | INTEGER
            | STRING
    """
    
    first_token = p.slice[1].type
    second_token = p.slice[2].type if len(p) > 2 else None
    third_token = p.slice[3].type if len(p) > 3 else None

    if first_token == 'ID':

        if second_token is None:
            p[0] = ast.Ident(p[1])
        
        elif second_token == 'assign':
             p[0] = ast.Assign(ident=ast.Ident(p[1]), expr=p[2])

    elif first_token == 'expr':
        
        if len(p) == 4 and third_token == 'expr':
            p[0] = ast.BinOp(operator=p[2], left=p[1], right=p[3])
        
        elif third_token == 'DOT':
            p[0] = ast.MethodCall(object=p[1], targettype=p[2], method=p[4])

    elif first_token == 'function_call':
        p[0] = p[1]

    elif first_token == 'IF':
        p[0] = ast.If(condition=p[2], true=p[4], false=p[6])
    
    elif first_token == 'WHILE':
        p[0] = ast.While(condition=p[2], action=p[4])
    
    elif first_token == 'LET':
        p[0] = ast.Let(assignments=p[2], expr=p[4])
    
    elif first_token == 'CASE':
        p[0] = ast.Case(expr=p[2])
    
    elif first_token == 'NEW':
        p[0] = ast.New(p[2])
    
    elif first_token in ['ISVOID', 'INT_COMP', 'NOT']:
        p[0] = ast.UnaryOp(operator=p[1], right=p[2])
    
    elif first_token in ['BLOCK_INIT', 'PARENTESIS_INIT']:
        p[0] = p[2]
    
    elif first_token in ['INTEGER', 'STRING', 'BOOL']:
        p[0] = p[1]

def p_empty(p):
    """empty :"""
    p[0] = None

def p_error(p):
    if p == None:
        token = "end of file"
    else:
        token = f"{p.type}({p.value}) on line {p.lineno}"
    
    print(f"Syntax error: Unexpected {token}")
# Create parser
parser = yacc.yacc(start='program', debug=True)
# yacc.yacc(debug=True)


def get_ast(sourcefile):
    with open(sourcefile, 'r') as source:
        t = yacc.parse(source.read())
    return t

precedence = (
    ('right', 'ASSIGN'),
    ('left', 'NOT'), # WIP: eh realmente left?
    ('nonassoc', 'LESS_THAN_OR_EQUAL', 'LESS_THAN', 'EQUAL' ),  # Nonassociative operators
    ('left', 'PLUS', 'MINUS'),
    ('left', 'MULTIPLY', 'DIVIDE'),
    ('left', 'ISVOID'), 
    ('left', 'INT_COMP'),
    ('left', 'AT'),
    ('left', 'DOT')  
)

if __name__ == '__main__':
    import sys
    if len(sys.argv) > 1:
        data = open(sys.argv[1], 'r').read()
        result = yacc.parse(data)
        print(result)

    else:
        while True:
            try:
                s = input('coolp> ')
            except EOFError:
                break
            if not s: continue
            result = yacc.parse(s)
            print(result)

When i run the parser to read an Cool file, the result = yacc.parse(data) statement returns None and an syntax error pops on the console

Cool file:

class Main inherits IO {
   main(): SELF_TYPE {
    out_string("Hello, World.")
   };
}; 

Error:

Syntax error: Unexpected BLOCK_INIT({) on line 1

Does anyone knows why this is hapenning?

0 Answers
Related