Finetuning BERT with LSTM via PyTorch and transformers library. Metrics remain the same with hyperparameter changes

Viewed 1074

I know for a fact that changing hyperparameters of an LSTM model or selecting different BERT layers causes changes in the classification result. I have tested this out using TensorFlow and Keras. I recently switched to Pytorch to do the same design, but no matter what I change, the result remains the same. Below is the code. Am I doing anything wrong?


def pad_sents(sents, pad_token):  #Pad list of sentences according to the longest sentence in the batch.
    sents_padded = []
    max_len = max(len(s) for s in sents)
    batch_size = len(sents)

    for s in sents:
        padded = [pad_token] * max_len
        padded[:len(s)] = s
        sents_padded.append(padded)

    return sents_padded

def sents_to_tensor(tokenizer, sents, device):
    tokens_list = [tokenizer.tokenize(str(sent)) for sent in sents]
    sents_lengths = [len(tokens) for tokens in tokens_list]
    tokens_list_padded = pad_sents(tokens_list, '[PAD]')
    sents_lengths = torch.tensor(sents_lengths, device=device)

    masks = []
    for tokens in tokens_list_padded:
        mask = [0 if token=='[PAD]' else 1 for token in tokens]
        masks.append(mask)
    masks_tensor = torch.tensor(masks, dtype=torch.long, device=device)
    tokens_id_list = [tokenizer.convert_tokens_to_ids(tokens) for tokens in tokens_list_padded]
    sents_tensor = torch.tensor(tokens_id_list, dtype=torch.long, device=device)

    return sents_tensor, masks_tensor, sents_lengths 


class BERT_LSTM_Model(nn.Module):

    def __init__(self, device, dropout_rate, n_class, lstm_hidden_size=None):
        super(BERT_LSTM_Model, self).__init__()

        self.bert_config = BertConfig.from_pretrained('bert-base-uncased', output_hidden_states=True)
        self.bert = BertModel.from_pretrained('bert-base-uncased',config =self.bert_config)
        self.tokenizer = BertTokenizer.from_pretrained('bert-base-uncased',config =self.bert_config)

        if not lstm_hidden_size:
            self.lstm_hidden_size = self.bert.config.hidden_size
        else:
            self.lstm_hidden_size = lstm_hidden_size
        self.n_class = n_class
        self.dropout_rate = dropout_rate
        self.lstm = nn.LSTM(self.bert.config.hidden_size, self.lstm_hidden_size, bidirectional=True)
        self.hidden_to_softmax = nn.Linear(self.lstm_hidden_size * 2, n_class, bias=True)
        self.dropout = nn.Dropout(p=self.dropout_rate)
        self.device = device

    def forward(self, sents):
        sents_tensor, masks_tensor, sents_lengths = sents_to_tensor(self.tokenizer, sents, self.device)
        encoded_layers = self.bert(input_ids=sents_tensor, attention_mask=masks_tensor)[2] #,output_all_encoded_layers=False)   #output_hidden_states output_hidden_states=True
        bert_hidden_layer = encoded_layers[12]
        bert_hidden_layer = bert_hidden_layer.permute(1, 0, 2)   #permute rotates the tensor. if tensor.shape = 3,4,5  tensor.permute(1,0,2), then tensor,shape= 4,3,5  (batch_size, sequence_length, hidden_size)

        enc_hiddens, (last_hidden, last_cell) = self.lstm(pack_padded_sequence(bert_hidden_layer, sents_lengths, enforce_sorted=False)) #enforce_sorted=False  #pack_padded_sequence(data and batch_sizes
        output_hidden = torch.cat((last_hidden[0], last_hidden[1]), dim=1)  # (batch_size, 2*hidden_size)
        output_hidden = self.dropout(output_hidden)
        pre_softmax = self.hidden_to_softmax(output_hidden)

        return pre_softmax


def batch_iter(data, batch_size, shuffle=False, bert=None):
    batch_num = math.ceil(data.shape[0] / batch_size)
    index_array = list(range(data.shape[0]))

    if shuffle:
        data = data.sample(frac=1)

    for i in range(batch_num):
        indices = index_array[i * batch_size: (i + 1) * batch_size]
        examples = data.iloc[indices] 
        targets = list(examples.train_label.values)
        yield sents, targets  # list[list[str]] if not bert else list[str], list[int]


def validation(model, df_val, loss_func, device):
    was_training = model.training
    model.eval()
    train_BERT_tweet = list(df_val.train_BERT_tweet)
    train_label = list(df_val.train_label)
    val_batch_size = 16

    n_batch = int(np.ceil(df_val.shape[0]/val_batch_size))

    total_loss = 0.

    with torch.no_grad():
        for i in range(n_batch):
            sents =  train_BERT_tweet[i*val_batch_size: (i+1)*val_batch_size]
            targets = torch.tensor(train_label[i*val_batch_size: (i+1)*val_batch_size],
                                   dtype=torch.long, device=device)
            batch_size = len(sents)
            pre_softmax = model(sents)
            batch_loss = loss_func(pre_softmax, targets)
            total_loss += batch_loss.item()*batch_size

    if was_training:
        model.train()

    return total_loss/df_val.shape[0]

def train():
    label_name = ['Yes', 'Maybe', 'No']
    if torch.cuda.is_available():
        device = torch.device("cuda")
    else:
        device = torch.device("cpu")

    start_time = time.time()
    print('Importing data...', file=sys.stderr)
    df_train = pd.read_csv('trainn.csv') #, index_col=0)
    df_val = pd.read_csv('valn.csv')   #, index_col=0)
    train_label = dict(df_train.train_label.value_counts())

    label_max = float(max(train_label.values()))

    train_label_weight = torch.tensor([label_max/train_label[i] for i in range(len(train_label))], device=device)

    print('Done! time elapsed %.2f sec' % (time.time() - start_time), file=sys.stderr)
    print('-' * 80, file=sys.stderr)

    start_time = time.time()
    print('Set up model...', file=sys.stderr)

    model = BERT_LSTM_Model(device=device, dropout_rate=0.2, n_class=len(label_name),lstm_hidden_size=768)
    optimizer = AdamW(model.parameters(), lr=1e-3, correct_bias=False)
    scheduler = get_linear_schedule_with_warmup(optimizer, warmup_steps=100, t_total=1000)  #changed the last 2 arguments to old ones

    model = model.to(device)
    print('Use device: %s' % device, file=sys.stderr)
    print('Done! time elapsed %.2f sec' % (time.time() - start_time), file=sys.stderr)
    print('-' * 80, file=sys.stderr)

    model.train()

    cn_loss = torch.nn.CrossEntropyLoss(weight=train_label_weight, reduction='mean')
    torch.save(cn_loss, 'loss_func3')  # for later testing

    train_batch_size =16
    valid_niter = 500
    log_every = 10
    model_save_path = 'NonLinear_bert_uncased_model.bin'

    num_trial = 0
    train_iter = patience = cum_loss = report_loss = 0
    cum_examples = report_examples = epoch = 0
    hist_valid_scores = []
    train_time = begin_time = time.time()
    print('Begin Maximum Likelihood training...')

    for epoch in range(20):

        for sents, targets in batch_iter(df_train, batch_size=train_batch_size, shuffle=True):  # for each epoch
            train_iter += 1
            optimizer.zero_grad()
            batch_size = len(sents)
            pre_softmax = model(sents)
            loss = cn_loss(pre_softmax, torch.tensor(targets, dtype=torch.long, device=device))
            loss.backward()
            optimizer.step()
            scheduler.step()
            batch_losses_val = loss.item() * batch_size
            report_loss += batch_losses_val
            cum_loss += batch_losses_val
            report_examples += batch_size
            cum_examples += batch_size

            if train_iter % log_every == 0:
                print('epoch %d, iter %d, avg. loss %.2f, '
                      'cum. examples %d, speed %.2f examples/sec, '
                      'time elapsed %.2f sec' % (epoch, train_iter,
                                                                                         report_loss / report_examples,
                                                                                         cum_examples,
                                                                                         report_examples / (time.time() - train_time),
                                                                                         time.time() - begin_time), file=sys.stderr)

                train_time = time.time()
                report_loss = report_examples = 0.

    #torch.save(model.state_dict(), 'LSTM_bert_uncased_model.bin')

            # perform validation
            if train_iter % valid_niter == 0:
                print('epoch %d, iter %d, cum. loss %.2f, cum. examples %d' % (epoch, train_iter,
                                                                                         cum_loss / cum_examples,
                                                                                         cum_examples), file=sys.stderr)
                cum_loss = cum_examples = 0.

                print('begin validation ...', file=sys.stderr)

                validation_loss = validation(model, df_val, cn_loss, device=device)   # dev batch size can be a bit larger

                print('validation: iter %d, loss %f' % (train_iter, validation_loss), file=sys.stderr)

                is_better = len(hist_valid_scores) == 0 or validation_loss < min(hist_valid_scores)
                hist_valid_scores.append(validation_loss)

                if is_better:
                    patience = 0
                    print('save currently the best model to [%s]' % model_save_path, file=sys.stderr)


                    torch.save(model.state_dict(), 'LSTM_bert_uncased_model.bin')
                    # also save the optimizers' state
                    torch.save(optimizer.state_dict(), model_save_path + '.optim')

                elif patience < 5:
                    patience += 1
                    print('hit patience %d' % patience, file=sys.stderr)

                    if patience == 20:
                        num_trial += 1
                        print('hit #%d trial' % num_trial, file=sys.stderr)
                        if num_trial == 3:
                            print('early stop!', file=sys.stderr)
                            exit(0)

                        # decay lr, and restore from previously best checkpoint
                        print('load previously best model and decay learning rate to %f%%' %
                              (0.1*100), file=sys.stderr)

                        # load model                                       model.load_state_dict(torch.load('LSTM_bert_uncased_model.bin'))
                        model = model.to(device)

                        print('restore parameters of the optimizers', file=sys.stderr)
                        optimizer.load_state_dict(torch.load(model_save_path + '.optim'))

                        # set new lr
                        for param_group in optimizer.param_groups:
                            param_group['lr'] *= 0.5

                        # reset patience
                        patience = 0

                if epoch == 100:
                    print('reached maximum number of epochs!', file=sys.stderr)
                    exit(0)

def test():
    label_name = ['Yes', 'Maybe', 'No']
    if torch.cuda.is_available():
        device = torch.device("cuda")
    else:
        device = torch.device("cpu")
    model = BERT_LSTM_Model(device=device, dropout_rate=0.3, n_class=len(label_name), lstm_hidden_size=768)

    model.load_state_dict(torch.load('LSTM_bert_uncased_model.bin'))
    model.to(device)
    model.eval()
    df_test = pd.read_csv('testn.csv')
    test_batch_size = 16
    n_batch = int(np.ceil(df_test.shape[0]/test_batch_size))
    cn_loss = torch.load('loss_func3', map_location=lambda storage, loc: storage).to(device)
    train_BERT_tweet = list(df_test.train_BERT_tweet)
    train_label = list(df_test.train_label)
    test_loss = 0.
    prediction = []
    prob = []
    softmax = torch.nn.Softmax(dim=1)

    with torch.no_grad():
        for i in range(n_batch):
            sents = train_BERT_tweet[i*test_batch_size: (i+1)*test_batch_size]
            targets = torch.tensor(train_label[i * test_batch_size: (i + 1) * test_batch_size],
                                   dtype=torch.long, device=device)
            batch_size = len(sents)

            pre_softmax = model(sents)
            batch_loss = cn_loss(pre_softmax, targets)
            test_loss += batch_loss.item()*batch_size
            prob_batch = softmax(pre_softmax)
            prob.append(prob_batch)

            prediction.extend([t.item() for t in list(torch.argmax(prob_batch, dim=1))])


    accuracy = accuracy_score(df_test.train_label.values, prediction)
    matthews = matthews_corrcoef(df_test.train_label.values, prediction)
    f1_macro = f1_score(df_test.train_label.values, prediction, average='macro')
    print('accuracy: %.2f' % accuracy)
    print('matthews coef: %.2f' % matthews)
    print('f1_macro: %.2f' % f1_macro)

TrainingModel = train()
TestingModel = test()

The data can be accessed from https://github.com/Kosisochi/DataSnippet I didnt know how else to create a synthetic data.

Also, the training and validation loss remains quite high with the lowest being around 0.93. I also tried a CNN and the same issue remained. Is there something I'm over looking? thanks for your help.

0 Answers
Related