I'm trying to load BERT "tfbert-large-uncased" but i got an error "Can't load config.json file"

Viewed 455

I'm trying to load the pre-train BERT model but I'm getting an error while loading tokenized it says config.json is not found. If anyone knows how to solve these issues please help me

Model and path configure

model_name = 'bert_v13'

data_dir = Path('../input/commonlitreadabilityprize/')
train_file = data_dir / 'train.csv'
test_file = data_dir / 'test.csv'
sample_file = data_dir / 'sample_submission.csv'

build_dir = Path('./build/')
output_dir = build_dir / model_name

trn_encode_file = output_dir / 'trn.enc.joblib'
val_predict_file = output_dir / f'{model_name}.val.txt'

submission_file = 'submission.csv'

pretrained_dir = '../tmp/input/tfbert-large-uncased'

id_col = 'id'
target_col = 'target'
text_col = 'excerpt'

max_len = 205
n_fold = 5
n_est = 2
n_stop = 2
batch_size = 8
seed = 42

Load Tokenizer and Model

# Tokenization using "Transformers"

# load tokenizer
def load_tokenizer():
    if not os.path.exists(pretrained_dir + '/vocab.txt'):
        Path(pretrained_dir).mkdir(parents=True, exist_ok=True)
        tokenizer = BertTokenizerFast.from_pretrained("bert-large-uncased")
        tokenizer.save_pretrained(pretrained_dir)
    else:
        print('loading the saved pretrained tokenizer')
        tokenizer = BertTokenizerFast.from_pretrained(pretrained_dir)
        
    model_config = BertConfig.from_pretrained(pretrained_dir)
    model_config.output_hidden_states = True
    return tokenizer, model_config

# load bert model
def load_bert(config):
    if not os.path.exists(pretrained_dir + '/tf_model.h5'):
        Path(pretrained_dir).mkdir(parents=True, exist_ok=True)
        bert_model = TFBertModel.from_pretrained("bert-large-uncased", config=config)
        bert_model.save_pretrained(pretrained_dir)
    else:
        print('loading the saved pretrained model')
        bert_model = TFBertModel.from_pretrained(pretrained_dir, config=config)
    return bert_model

loading encoder

def bert_encode(texts, tokenizer, max_len=max_len):
    input_ids = []
    token_type_ids = []
    attention_mask = []
    
    for text in texts:
        token = tokenizer(text, max_lenght = max_len,truncation=True, padding='max_length',add_special_tokens = True)
        input_ids.append(token['input_ids'])
        
        token_type_ids.append(token['token_type_ids'])
        attention_mask.append(token['attention_mask'])
        
    return np.array(input_ids), np.array(token_type_ids),np.array(attention_mask)

this function gives an error

tokenizer, bert_cofig = load_tokenizer()

X = bert_encode(trn[text_col].values, tokenizer, 
                max_len=max_len)
X_tst = bert_encode(tst[text_col].values, tokenizer, 
                    max_len = max_len)
y = trn[target_col].values

print(X[0].shape, X_tst[0].shape, y.shape)

Error

file ../tmp/input/tfbert-large-uncased/config.json not found

0 Answers
Related