Extremely large loss values with tensorflow-probability and ELBO loss function

Viewed 237

I'm trying to train a CNN in tensorflow-probability using the ELBO loss function. When I do so, the model produces very large training and validation loss values, in the hundreds of billions. These values decrease minimally over time no matter how long I train the model. The accuracy values do not improve either.

Does anyone know why this is happening? Running the same model as below using standard tensorflow layers in place of the tfp layers works quite well, producing normal sized loss values and reaching accuracy above 90% very quickly. I have a corpus of 3064 512x512 images. I'm using Tensorflow version 2.4.1 and tensorflow_probability version 0.12.1. I'm running everything on Google Colab.

import numpy as np
import os
import pathlib
import PIL
import tensorflow as tf
from tensorflow.keras import layers
import tensorflow_probability as tfp

from google.colab import drive 
drive.mount('/content/gdrive')
%cd /content/gdrive/My\ Drive/
%cd './data/'

train_data_dir = '/content/gdrive/My Drive/data/classes_train'
os.chdir(train_data_dir)
train_data_dir = pathlib.Path(train_data_dir)

test_data_dir = '/content/gdrive/My Drive/data/classes_test'
os.chdir(test_data_dir)
test_data_dir = pathlib.Path(test_data_dir)

batch_size = 32
img_height = 512
img_width = 512

train_ds = tf.keras.preprocessing.image_dataset_from_directory(
  train_data_dir,
  validation_split=0.25,
  subset="training",
  seed=123,
  image_size=(img_height, img_width),
  batch_size=batch_size)

val_ds = tf.keras.preprocessing.image_dataset_from_directory(
  train_data_dir,
  validation_split=0.2,
  subset="validation",
  seed=123,
  image_size=(img_height, img_width),
  batch_size=batch_size)

AUTOTUNE = tf.data.AUTOTUNE
train_ds = train_ds.cache().prefetch(buffer_size=AUTOTUNE)
val_ds = val_ds.cache().prefetch(buffer_size=AUTOTUNE)

n_classes = 3
n_epochs = 100
model = tf.keras.Sequential([
  layers.experimental.preprocessing.Rescaling(1./255),
  tfp.layers.Convolution2DFlipout(32, 5, activation='relu'),
  layers.MaxPooling2D(),
  tfp.layers.Convolution2DFlipout(64, 5, activation='relu'),
  layers.MaxPooling2D(),
  tfp.layers.Convolution2DFlipout(128, 5, activation='relu'),
  layers.MaxPooling2D(),
  layers.Flatten(),
  tfp.layers.DenseFlipout(128, activation='relu'),
  tfp.layers.DenseFlipout(n_classes)
])

@tf.function
def elbo_loss(labels, logits):
    loss_en = tf.nn.softmax_cross_entropy_with_logits(labels, logits)
    loss_kl = tf.keras.losses.KLD(labels, logits)
    loss = tf.reduce_mean(tf.add(loss_en, loss_kl))
    return loss

model = build_cnn(len(train_ds))
optimizer = tf.keras.optimizers.Adam()
model.compile(
  optimizer=optimizer,
  loss=elbo_loss,
  metrics=['accuracy'])

model.fit(
  train_ds,
  validation_data=val_ds,
  epochs=n_epochs
)
print('Stopped training after', len(model.history.history['loss']), 'epochs')
0 Answers
Related