I am trying the example presented in https://pytorch.org/tutorials/intermediate/char_rnn_classification_tutorial.html but I am using a LSTM model instead of a RNN. The dataset is composed by different names (of different sizes) and their corresponding language (total number of languages is 18), and the objective is to train a model that given a certain name outputs the language it belongs to.
My problems right now are:
- How to deal with variable size names, i.e. Hector and Kim, in the LSTM
- A whole name (secuence of character) is processed every time in the LSTM so the output of the softmax function has shape
(#characters of name, #target classes)but I would like just to obtain(1,#target of classes)in order to decide each name to which class does it correspond to. I have tried to just get the last row but results are very bad.
class LSTM(nn.Module):
def __init__(self, embedding_dim, hidden_dim, vocab_size, tagset_size):
super(LSTM, self).__init__()
self.hidden_dim = hidden_dim
self.word_embeddings = nn.Embedding(vocab_size, embedding_dim)
# The LSTM takes word embeddings as inputs, and outputs hidden states
# with dimensionality hidden_dim.
self.lstm = nn.LSTM(embedding_dim, hidden_dim)
# The linear layer that maps from hidden state space to tag space
self.hidden2tag = nn.Linear(hidden_dim, tagset_size)
self.softmax = nn.LogSoftmax(dim = 1)
def forward(self, word):
embeds = self.word_embeddings(word)
lstm_out, _ = self.lstm(embeds.view(len(word), 1, -1))
tag_space = self.hidden2tag(lstm_out.view(len(word), -1))
tag_scores = self.softmax(tag_space)
return tag_scores
def initHidden(self):
return Variable(torch.zeros(1, self.hidden_dim))
lstm = LSTM(n_embedding_dim,n_hidden,n_characters,n_categories)
optimizer = torch.optim.SGD(lstm.parameters(), lr=learning_rate)
criterion = nn.NLLLoss()
def train(category_tensor, line_tensor):
# i.e. line_tensor = tensor([37, 4, 14, 13, 19, 0, 17, 0, 10, 8, 18]) and category_tensor = tensor([11])
optimizer.zero_grad()
output = lstm(line_tensor)
loss = criterion(output[-1:], category_tensor) # VERY BAD
loss.backward()
optimizer.step()
return output, loss.data.item()
Where line_tensor is of variable size (depending the size of each name) and is a mapping between character and their index in the dictionary