I know for a fact that changing hyperparameters of an LSTM model or selecting different BERT layers causes changes in the classification result. I have tested this out using TensorFlow and Keras. I recently switched to Pytorch to do the same design, but no matter what I change, the result remains the same. Below is the code. Am I doing anything wrong?
def pad_sents(sents, pad_token): #Pad list of sentences according to the longest sentence in the batch.
sents_padded = []
max_len = max(len(s) for s in sents)
batch_size = len(sents)
for s in sents:
padded = [pad_token] * max_len
padded[:len(s)] = s
sents_padded.append(padded)
return sents_padded
def sents_to_tensor(tokenizer, sents, device):
tokens_list = [tokenizer.tokenize(str(sent)) for sent in sents]
sents_lengths = [len(tokens) for tokens in tokens_list]
tokens_list_padded = pad_sents(tokens_list, '[PAD]')
sents_lengths = torch.tensor(sents_lengths, device=device)
masks = []
for tokens in tokens_list_padded:
mask = [0 if token=='[PAD]' else 1 for token in tokens]
masks.append(mask)
masks_tensor = torch.tensor(masks, dtype=torch.long, device=device)
tokens_id_list = [tokenizer.convert_tokens_to_ids(tokens) for tokens in tokens_list_padded]
sents_tensor = torch.tensor(tokens_id_list, dtype=torch.long, device=device)
return sents_tensor, masks_tensor, sents_lengths
class BERT_LSTM_Model(nn.Module):
def __init__(self, device, dropout_rate, n_class, lstm_hidden_size=None):
super(BERT_LSTM_Model, self).__init__()
self.bert_config = BertConfig.from_pretrained('bert-base-uncased', output_hidden_states=True)
self.bert = BertModel.from_pretrained('bert-base-uncased',config =self.bert_config)
self.tokenizer = BertTokenizer.from_pretrained('bert-base-uncased',config =self.bert_config)
if not lstm_hidden_size:
self.lstm_hidden_size = self.bert.config.hidden_size
else:
self.lstm_hidden_size = lstm_hidden_size
self.n_class = n_class
self.dropout_rate = dropout_rate
self.lstm = nn.LSTM(self.bert.config.hidden_size, self.lstm_hidden_size, bidirectional=True)
self.hidden_to_softmax = nn.Linear(self.lstm_hidden_size * 2, n_class, bias=True)
self.dropout = nn.Dropout(p=self.dropout_rate)
self.device = device
def forward(self, sents):
sents_tensor, masks_tensor, sents_lengths = sents_to_tensor(self.tokenizer, sents, self.device)
encoded_layers = self.bert(input_ids=sents_tensor, attention_mask=masks_tensor)[2] #,output_all_encoded_layers=False) #output_hidden_states output_hidden_states=True
bert_hidden_layer = encoded_layers[12]
bert_hidden_layer = bert_hidden_layer.permute(1, 0, 2) #permute rotates the tensor. if tensor.shape = 3,4,5 tensor.permute(1,0,2), then tensor,shape= 4,3,5 (batch_size, sequence_length, hidden_size)
enc_hiddens, (last_hidden, last_cell) = self.lstm(pack_padded_sequence(bert_hidden_layer, sents_lengths, enforce_sorted=False)) #enforce_sorted=False #pack_padded_sequence(data and batch_sizes
output_hidden = torch.cat((last_hidden[0], last_hidden[1]), dim=1) # (batch_size, 2*hidden_size)
output_hidden = self.dropout(output_hidden)
pre_softmax = self.hidden_to_softmax(output_hidden)
return pre_softmax
def batch_iter(data, batch_size, shuffle=False, bert=None):
batch_num = math.ceil(data.shape[0] / batch_size)
index_array = list(range(data.shape[0]))
if shuffle:
data = data.sample(frac=1)
for i in range(batch_num):
indices = index_array[i * batch_size: (i + 1) * batch_size]
examples = data.iloc[indices]
targets = list(examples.train_label.values)
yield sents, targets # list[list[str]] if not bert else list[str], list[int]
def validation(model, df_val, loss_func, device):
was_training = model.training
model.eval()
train_BERT_tweet = list(df_val.train_BERT_tweet)
train_label = list(df_val.train_label)
val_batch_size = 16
n_batch = int(np.ceil(df_val.shape[0]/val_batch_size))
total_loss = 0.
with torch.no_grad():
for i in range(n_batch):
sents = train_BERT_tweet[i*val_batch_size: (i+1)*val_batch_size]
targets = torch.tensor(train_label[i*val_batch_size: (i+1)*val_batch_size],
dtype=torch.long, device=device)
batch_size = len(sents)
pre_softmax = model(sents)
batch_loss = loss_func(pre_softmax, targets)
total_loss += batch_loss.item()*batch_size
if was_training:
model.train()
return total_loss/df_val.shape[0]
def train():
label_name = ['Yes', 'Maybe', 'No']
if torch.cuda.is_available():
device = torch.device("cuda")
else:
device = torch.device("cpu")
start_time = time.time()
print('Importing data...', file=sys.stderr)
df_train = pd.read_csv('trainn.csv') #, index_col=0)
df_val = pd.read_csv('valn.csv') #, index_col=0)
train_label = dict(df_train.train_label.value_counts())
label_max = float(max(train_label.values()))
train_label_weight = torch.tensor([label_max/train_label[i] for i in range(len(train_label))], device=device)
print('Done! time elapsed %.2f sec' % (time.time() - start_time), file=sys.stderr)
print('-' * 80, file=sys.stderr)
start_time = time.time()
print('Set up model...', file=sys.stderr)
model = BERT_LSTM_Model(device=device, dropout_rate=0.2, n_class=len(label_name),lstm_hidden_size=768)
optimizer = AdamW(model.parameters(), lr=1e-3, correct_bias=False)
scheduler = get_linear_schedule_with_warmup(optimizer, warmup_steps=100, t_total=1000) #changed the last 2 arguments to old ones
model = model.to(device)
print('Use device: %s' % device, file=sys.stderr)
print('Done! time elapsed %.2f sec' % (time.time() - start_time), file=sys.stderr)
print('-' * 80, file=sys.stderr)
model.train()
cn_loss = torch.nn.CrossEntropyLoss(weight=train_label_weight, reduction='mean')
torch.save(cn_loss, 'loss_func3') # for later testing
train_batch_size =16
valid_niter = 500
log_every = 10
model_save_path = 'NonLinear_bert_uncased_model.bin'
num_trial = 0
train_iter = patience = cum_loss = report_loss = 0
cum_examples = report_examples = epoch = 0
hist_valid_scores = []
train_time = begin_time = time.time()
print('Begin Maximum Likelihood training...')
for epoch in range(20):
for sents, targets in batch_iter(df_train, batch_size=train_batch_size, shuffle=True): # for each epoch
train_iter += 1
optimizer.zero_grad()
batch_size = len(sents)
pre_softmax = model(sents)
loss = cn_loss(pre_softmax, torch.tensor(targets, dtype=torch.long, device=device))
loss.backward()
optimizer.step()
scheduler.step()
batch_losses_val = loss.item() * batch_size
report_loss += batch_losses_val
cum_loss += batch_losses_val
report_examples += batch_size
cum_examples += batch_size
if train_iter % log_every == 0:
print('epoch %d, iter %d, avg. loss %.2f, '
'cum. examples %d, speed %.2f examples/sec, '
'time elapsed %.2f sec' % (epoch, train_iter,
report_loss / report_examples,
cum_examples,
report_examples / (time.time() - train_time),
time.time() - begin_time), file=sys.stderr)
train_time = time.time()
report_loss = report_examples = 0.
#torch.save(model.state_dict(), 'LSTM_bert_uncased_model.bin')
# perform validation
if train_iter % valid_niter == 0:
print('epoch %d, iter %d, cum. loss %.2f, cum. examples %d' % (epoch, train_iter,
cum_loss / cum_examples,
cum_examples), file=sys.stderr)
cum_loss = cum_examples = 0.
print('begin validation ...', file=sys.stderr)
validation_loss = validation(model, df_val, cn_loss, device=device) # dev batch size can be a bit larger
print('validation: iter %d, loss %f' % (train_iter, validation_loss), file=sys.stderr)
is_better = len(hist_valid_scores) == 0 or validation_loss < min(hist_valid_scores)
hist_valid_scores.append(validation_loss)
if is_better:
patience = 0
print('save currently the best model to [%s]' % model_save_path, file=sys.stderr)
torch.save(model.state_dict(), 'LSTM_bert_uncased_model.bin')
# also save the optimizers' state
torch.save(optimizer.state_dict(), model_save_path + '.optim')
elif patience < 5:
patience += 1
print('hit patience %d' % patience, file=sys.stderr)
if patience == 20:
num_trial += 1
print('hit #%d trial' % num_trial, file=sys.stderr)
if num_trial == 3:
print('early stop!', file=sys.stderr)
exit(0)
# decay lr, and restore from previously best checkpoint
print('load previously best model and decay learning rate to %f%%' %
(0.1*100), file=sys.stderr)
# load model model.load_state_dict(torch.load('LSTM_bert_uncased_model.bin'))
model = model.to(device)
print('restore parameters of the optimizers', file=sys.stderr)
optimizer.load_state_dict(torch.load(model_save_path + '.optim'))
# set new lr
for param_group in optimizer.param_groups:
param_group['lr'] *= 0.5
# reset patience
patience = 0
if epoch == 100:
print('reached maximum number of epochs!', file=sys.stderr)
exit(0)
def test():
label_name = ['Yes', 'Maybe', 'No']
if torch.cuda.is_available():
device = torch.device("cuda")
else:
device = torch.device("cpu")
model = BERT_LSTM_Model(device=device, dropout_rate=0.3, n_class=len(label_name), lstm_hidden_size=768)
model.load_state_dict(torch.load('LSTM_bert_uncased_model.bin'))
model.to(device)
model.eval()
df_test = pd.read_csv('testn.csv')
test_batch_size = 16
n_batch = int(np.ceil(df_test.shape[0]/test_batch_size))
cn_loss = torch.load('loss_func3', map_location=lambda storage, loc: storage).to(device)
train_BERT_tweet = list(df_test.train_BERT_tweet)
train_label = list(df_test.train_label)
test_loss = 0.
prediction = []
prob = []
softmax = torch.nn.Softmax(dim=1)
with torch.no_grad():
for i in range(n_batch):
sents = train_BERT_tweet[i*test_batch_size: (i+1)*test_batch_size]
targets = torch.tensor(train_label[i * test_batch_size: (i + 1) * test_batch_size],
dtype=torch.long, device=device)
batch_size = len(sents)
pre_softmax = model(sents)
batch_loss = cn_loss(pre_softmax, targets)
test_loss += batch_loss.item()*batch_size
prob_batch = softmax(pre_softmax)
prob.append(prob_batch)
prediction.extend([t.item() for t in list(torch.argmax(prob_batch, dim=1))])
accuracy = accuracy_score(df_test.train_label.values, prediction)
matthews = matthews_corrcoef(df_test.train_label.values, prediction)
f1_macro = f1_score(df_test.train_label.values, prediction, average='macro')
print('accuracy: %.2f' % accuracy)
print('matthews coef: %.2f' % matthews)
print('f1_macro: %.2f' % f1_macro)
TrainingModel = train()
TestingModel = test()
The data can be accessed from https://github.com/Kosisochi/DataSnippet I didnt know how else to create a synthetic data.
Also, the training and validation loss remains quite high with the lowest being around 0.93. I also tried a CNN and the same issue remained. Is there something I'm over looking? thanks for your help.