My LSTM model does not learn , the weights do not get update

Viewed 561

My LSTM model in pytorch does not learn and it does not get any update while training .... after each epoch I print the sum of weights for different each layer but still it does not get any update ...

The y array is (n,3) , first column is keeping the actual size , second is the label (0 or 1) and the last one is the weight for penalizing loss function.

Apparently the optimizer.step() does not work and does not apply gradients to weights. On a separate note ; I have tried the model with different learning rates and mini batches and no different result. The result is shown is from a dummy variables randomly generated and y label is an imbalanced dataset ~ 3% ! I have tried different weights to overcome the imbalanced ration but I guess something is wrong with the model.

Also, if I run the model with my own dataset , the gradients for the lstm layer (all four params ) would be just zero ! but in the linear layer there are gradients .

import torch.nn as nn
from torch.nn.utils.rnn import pack_padded_sequence

class LSTMClassifier(nn.Module):
    """
    This is the simple RNN model we will be using to perform Sentiment Analysis.
    """

    def __init__(self, feature_size, hidden_dim , layer_dim = 1):
        """
        Initialize the model by settingg up the various layers.
        """
        super(LSTMClassifier, self).__init__()
        
        self.hidden_dim = hidden_dim
        self.layer_dim = layer_dim
        self.lstm = nn.LSTM(feature_size, hidden_dim , layer_dim,  batch_first = True)
        self.dense = nn.Linear(in_features=hidden_dim, out_features=1)
        self.sig = nn.Sigmoid()
        
        
    def init_hidden(self, x):
        h0 = torch.zeros(self.layer_dim, x.size(0), self.hidden_dim)
        c0 = torch.zeros(self.layer_dim, x.size(0), self.hidden_dim)
        return [t for t in (h0, c0)]
        
        
    
    def forward(self, x , y):
        #import pdb; pdb.set_trace()
        """
        Perform a forward pass of our model on some input.
        """
        #h0, c0 = self.init_hidden(x)
        x_seq = y[:,0]
        x = pack_padded_sequence(x, x_seq, batch_first=True , enforce_sorted = False)
        lstm_out, _ = self.lstm(x)
        lstm_out, _ = torch.nn.utils.rnn.pad_packed_sequence(lstm_out, batch_first=True)
        lstm_out = lstm_out.contiguous()
        out = self.dense(lstm_out)[:,-1,:]
        #out = out[range(len(x_seq)), (x_seq - 1)]
        return self.sig(out.squeeze())
def _get_train_data_loader(batch_size, X , y):
    print("Get train/test data loader.")

   
    train_y = torch.from_numpy(y).long()
    try :
        train_X = torch.from_numpy(X).float()
    except :
        train_X = X

    train_ds = torch.utils.data.TensorDataset(train_X, train_y)

    return torch.utils.data.DataLoader(train_ds, batch_size=batch_size , shuffle=True)
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")


print ('train dist' , sum(y_train[:,1])/y_train.shape[0] , '\n ****\n' , 
      'test dist' , sum(y_test[:,1])/y_test.shape[0])
optimizer = optim.SGD(model.parameters() , lr=.01 )
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")

train_loader = _get_train_data_loader(batch_size = 128, X = X_ , y = y_weighted)
test_loader = _get_train_data_loader(batch_size = 512, X = X_test , y = y_test)

epochs = 2
model = LSTMClassifier (87 , 5)

loss_fn = torch.nn.BCELoss(reduction='none')
train_loss = []
test_loss = []
for epoch in range(1, epochs + 1):
    print ('epoch = ' , epoch)
    model.train()
    total_loss = 0
    for batch in train_loader:         
        batch_X, batch_y = batch

        #batch_X = batch_X.to(device)
        #batch_y = batch_y.to(device)

        # TODO: Complete this train method to train the model provided.
        optimizer.zero_grad()
        
        
         # Forward pass
        outputs = model(batch_X , batch_y)
        y_ = batch_y[:,1].float()
        loss = loss_fn(outputs, y_)
        weight=batch_y[:,2].float()
        loss = (loss * weight).mean()
        

        # Backward and optimize
        loss.backward()
        optimizer.step()
        


        total_loss += loss.data.item()
        
    for p in model.parameters():
           print(torch.sum(p.grad))
        
    print ('total loss' , total_loss)
    print ('lstm weight' , torch.sum(model.lstm.weight_hh_l0.data) , 'dense_weight' , torch.sum(model.dense.weight.data))
    
    train_loss.append(total_loss)
    with torch.no_grad():
        n_correct = 0
        n_samples = 0
        for test, labels in test_loader:
            #labels = labels.to(device)
            outputs = model(test , labels)
            # max returns (value ,index)
            predicted = torch.round(outputs)
            n_samples += labels.size(0)
            n_correct += (predicted == labels[:,1]).sum().item()

        acc = 100.0 * n_correct / n_samples
        print(f'Accuracy of the network on the 10000 test images: {acc} %') 
        test_loss.append(acc)

******************************************
epoch =  1
tensor(6518.8760)
tensor(-236.9392)
tensor(149.6967)
tensor(149.6966)
tensor(-1551.1709)
tensor(5021.9199)
total loss 1871.447255373001
lstm weight tensor(3.0054) dense_weight tensor(-0.5395)
Accuracy of the network on the 10000 test images: 96.92037099752012 %
epoch =  2
tensor(7822.5503)
tensor(-284.3271)
tensor(179.6338)
tensor(179.6338)
tensor(-1861.4037)
tensor(6026.2920)
total loss 1871.465574145317
lstm weight tensor(3.0054) dense_weight tensor(-0.5395)
Accuracy of the network on the 10000 test images: 96.92037099752012 %
***********************************************

Another Question is how to use weight in BCEloss(weight ) while we load data with data loader ? The only way left is instantiating a loss_fn within each loop ?

Any help is much appreciated !

0 Answers
Related