How can fix learning rate with text classification on tensorflow?

Viewed 549

I have been coding sentiment analysis model with tensorflow keras. I am using csv dataset which has labels(pos:1, neg:0) in row 1 and English texts in row 2. The results I expect is to show number between 0 and 1 when I input some texts through txt file. However, though I set model, the loss rate is keeping negative score and accuracy rate is not increasing, including validation rates. I do not know what the matter is. Thus, I attached my codes. Thank you very much.

import csv
import tensorflow as tf
import numpy as np
from tensorflow.keras.preprocessing.text import Tokenizer
from tensorflow.keras.preprocessing.sequence import pad_sequences

vocab_size = 20000
embedding_dim = 16
max_length = 120
trunc_type='post'
padding_type='post'
oov_tok = "<OOV>"
training_portion = .8

sentences = []
labels = []
stopwords = [ "a", "about", "above", "after", "again", "against", "all", "am", "an", "any", "are", 
"as", "at", "be", "because", "been", "before", "being", "below", "between", "both", "but", "by", 
"could", "did", "do", "does", "doing", "down", "during", "each", "few", "for", "from", "further", 
"had", "has", "have", "having", "he", "he'd", "he'll", "he's", "her", "here", "here's", "hers", 
"herself", "him", "himself", "his", "how", "how's", "i", "i'd", "i'll", "i'm", "i've", "if", "in", 
"into", "is", "it", "it's", "its", "itself", "let's", "me", "more", "most", "my", "myself", "nor", 
"of", "on", "once", "only", "or", "other", "ought", "our", "ours", "ourselves", "out", "over", "own", 
"same", "she", "she'd", "she'll", "she's", "should", "so", "some", "such", "than", "that", "that's", 
"the", "their", "theirs", "them", "themselves", "then", "there", "there's", "these", "they", 
"they'd", "they'll", "they're", "they've", "this", "those", "through", "to", "too", "under", "until", 
"up", "very", "was", "we", "we'd", "we'll", "we're", "we've", "were", "what", "what's", "when", 
"when's", "where", "where's", "which", "while", "who", "who's", "whom", "why", "why's", "with", 
"would", "you", "you'd", "you'll", "you're", "you've", "your", "yours", "yourself", "yourselves" ]
print(len(stopwords)

with open("/train.csv", 'r', encoding='latin1') as csvfile:
reader = csv.reader(csvfile, delimiter=',')
next(reader)
for row in reader:
    labels.append(row[0])
    sentence = row[1]
    for word in stopwords:
        token = " " + word + " "
        sentence = sentence.replace(token, " ")
    sentences.append(sentence)

train_size = int(len(sentences) * training_portion)

train_sentences = sentences[:train_size]
train_labels = labels[:train_size]

validation_sentences = sentences[train_size:]
validation_labels = labels[train_size:]

tokenizer = Tokenizer(num_words = vocab_size, oov_token=oov_tok)
tokenizer.fit_on_texts(train_sentences)
word_index = tokenizer.word_index

train_sequences = tokenizer.texts_to_sequences(train_sentences)
train_padded = pad_sequences(train_sequences, padding=padding_type, maxlen=max_length)

validation_sequences = tokenizer.texts_to_sequences(validation_sentences)
validation_padded = pad_sequences(validation_sequences, padding=padding_type, maxlen=max_length)

label_tokenizer = Tokenizer()
label_tokenizer.fit_on_texts(labels)

training_label_seq = np.array(label_tokenizer.texts_to_sequences(train_labels))
validation_label_seq = np.array(label_tokenizer.texts_to_sequences(validation_labels))

model = tf.keras.Sequential([
tf.keras.layers.Embedding(vocab_size, 64),
tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(128,  return_sequences=True)),
tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(32)),
tf.keras.layers.Dense(64, activation='relu'),
tf.keras.layers.Dropout(0.5),
tf.keras.layers.Dense(16, activation='relu'),
tf.keras.layers.Dropout(0.5),
tf.keras.layers.Dense(1)
])
model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),
          optimizer=tf.keras.optimizers.Adam(1e-4),
          metrics=['accuracy'])
model.summary()

history = model.fit(train_padded, training_label_seq, epochs=6,
                validation_data=(validation_padded, validation_label_seq),
                validation_steps=30)

enter image description here

1 Answers

The issue is that you are applying tokenizer on labels as well which will convert the labels 0 and 1 to 1 and 2 which confused the classifier, since tf.keras Tokenizer word.index starts from index 1(not 0).

Below was the target labels causing negative loss by confusing the classifier.

label_tokenizer.word_index

{'0': 1, '1': 2}.

Below is the modified code on sample data which is giving positive loss values as expected.

import csv
import tensorflow as tf
import numpy as np
from tensorflow.keras.preprocessing.text import Tokenizer
from tensorflow.keras.preprocessing.sequence import pad_sequences
import pandas as pd 

vocab_size = 20000
embedding_dim = 16
max_length = 120
trunc_type='post'
padding_type='post'
oov_tok = "<OOV>"
training_portion = .8

sentences = []
labels = []
stopwords = [ "a", "about", "above", "after", "again", "against", "all", "am", "an", "any", "are", 
"as", "at", "be", "because", "been", "before", "being", "below", "between", "both", "but", "by", 
"could", "did", "do", "does", "doing", "down", "during", "each", "few", "for", "from", "further", 
"had", "has", "have", "having", "he", "he'd", "he'll", "he's", "her", "here", "here's", "hers", 
"herself", "him", "himself", "his", "how", "how's", "i", "i'd", "i'll", "i'm", "i've", "if", "in", 
"into", "is", "it", "it's", "its", "itself", "let's", "me", "more", "most", "my", "myself", "nor", 
"of", "on", "once", "only", "or", "other", "ought", "our", "ours", "ourselves", "out", "over", "own", 
"same", "she", "she'd", "she'll", "she's", "should", "so", "some", "such", "than", "that", "that's", 
"the", "their", "theirs", "them", "themselves", "then", "there", "there's", "these", "they", 
"they'd", "they'll", "they're", "they've", "this", "those", "through", "to", "too", "under", "until", 
"up", "very", "was", "we", "we'd", "we'll", "we're", "we've", "were", "what", "what's", "when", 
"when's", "where", "where's", "which", "while", "who", "who's", "whom", "why", "why's", "with", 
"would", "you", "you'd", "you'll", "you're", "you've", "your", "yours", "yourself", "yourselves" ]
print(len(stopwords)) 

with open("train.csv", 'r', encoding='latin1') as csvfile:
  reader = csv.reader(csvfile, delimiter=',')
  next(reader)
  for row in reader:
      labels.append(int(row[0]))
      sentence = row[1]
      for word in stopwords:
          token = " " + word + " "
          sentence = sentence.replace(token, " ")
      sentences.append(sentence) 

train_size = int(len(sentences) * training_portion)

train_sentences = sentences[:train_size]
train_labels = np.array(labels[:train_size])

validation_sentences = sentences[train_size:]
validation_labels = np.array(labels[train_size:])

tokenizer = Tokenizer(num_words = vocab_size, oov_token=oov_tok)
tokenizer.fit_on_texts(train_sentences)
word_index = tokenizer.word_index

train_sequences = tokenizer.texts_to_sequences(train_sentences)
train_padded = pad_sequences(train_sequences, padding=padding_type, maxlen=max_length)

validation_sequences = tokenizer.texts_to_sequences(validation_sentences)
validation_padded = pad_sequences(validation_sequences, padding=padding_type, maxlen=max_length) 


model = tf.keras.Sequential([
tf.keras.layers.Embedding(vocab_size, 64),
tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(128,  return_sequences=True)),
tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(32)),
tf.keras.layers.Dense(64, activation='relu'),
tf.keras.layers.Dropout(0.2),
tf.keras.layers.Dense(16, activation='relu'),
tf.keras.layers.Dropout(0.2),
tf.keras.layers.Dense(1,activation="sigmoid")
])
model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),
          optimizer=tf.keras.optimizers.Adam(1e-4),
          metrics=['accuracy'])
model.summary()

history = model.fit(train_padded, train_labels, epochs=6,
                validation_data=(validation_padded, validation_labels),
                validation_steps=30)

Output:

Model: "sequential_2"
_________________________________________________________________
Layer (type)                 Output Shape              Param #   
=================================================================
embedding_2 (Embedding)      (None, None, 64)          1280000   
_________________________________________________________________
bidirectional_4 (Bidirection (None, None, 256)         197632    
_________________________________________________________________
bidirectional_5 (Bidirection (None, 64)                73984     
_________________________________________________________________
dense_6 (Dense)              (None, 64)                4160      
_________________________________________________________________
dropout_4 (Dropout)          (None, 64)                0         
_________________________________________________________________
dense_7 (Dense)              (None, 16)                1040      
_________________________________________________________________
dropout_5 (Dropout)          (None, 16)                0         
_________________________________________________________________
dense_8 (Dense)              (None, 1)                 17        
=================================================================
Total params: 1,556,833
Trainable params: 1,556,833
Non-trainable params: 0
_________________________________________________________________
Epoch 1/6
191/191 [==============================] - 71s 371ms/step - loss: 0.7135 - accuracy: 0.5783 - val_loss: 0.6934 - val_accuracy: 0.5345
Epoch 2/6
191/191 [==============================] - 71s 369ms/step - loss: 0.6943 - accuracy: 0.5793 - val_loss: 0.6932 - val_accuracy: 0.5345
Epoch 3/6
191/191 [==============================] - 71s 370ms/step - loss: 0.6936 - accuracy: 0.5791 - val_loss: 0.6932 - val_accuracy: 0.5345
Epoch 4/6
191/191 [==============================] - 70s 369ms/step - loss: 0.6934 - accuracy: 0.5793 - val_loss: 0.6932 - val_accuracy: 0.5345
Epoch 5/6
191/191 [==============================] - 71s 371ms/step - loss: 0.6934 - accuracy: 0.5793 - val_loss: 0.6932 - val_accuracy: 0.5345
Epoch 6/6
191/191 [==============================] - 72s 375ms/step - loss: 0.6933 - accuracy: 0.5793 - val_loss: 0.6931 - val_accuracy: 0.5345
Related