I am training a large dataset with Keras. I have about 500 000 npy-files (2x 240000) that are loaded batch-wise with a data generator. During training the GPU utilization is high in the very beginning, then it stabilizes at 0%, sometimes spiking up to 20-30%.
When testing the generator in a for-loop, it shows the same tendency. It works perfectly for a short time in the beginning, then it suddenly slows down considerably (approx. 100 times as slow).
I have narrowed it down to the np.load(). As suggested in other relevant posts, I have tried gc.collect(), and opening and closing the file for each load. None of these solved the problem.
I would like to avoid rebuilding the dataset.
Quite new to this. Does anyone have suggestions for how to improve the performance of the data generator?
The data generator:
def data_generator(train = True, batch_size = BATCH_SIZE, dim_x=(300,69), dim_y=(300,66)):
if train:
filenames = x_train_fn
else:
filenames = x_val_fn
batch_i = 0
while True:
if batch_i >= (len(filenames)// batch_size):
batch_i = 0
np.random.shuffle(filenames)
file_chunk = filenames[batch_i*batch_size:(batch_i+1)*batch_size]
X = np.zeros((batch_size,*dim_x))
y = np.zeros((batch_size,*dim_y))
for i, ID in enumerate(file_chunk):
X[i,] = np.load(data_dir_x + ID)
y[i,] = np.load(data_dir_y + ID.replace("x", "y"))
yield X, y
batch_i += 1
train_dataset = tf.data.Dataset.from_generator(lambda: data_generator(train=True),(tf.float32, tf.float32))
validation_dataset = tf.data.Dataset.from_generator(lambda: data_generator(train=False), (tf.float32, tf.float32))
train_dataset = train_dataset.prefetch(buffer_size = tf.data.experimental.AUTOTUNE)
validation_dataset = validation_dataset.prefetch(buffer_size = tf.data.experimental.AUTOTUNE)
The model used for training:
model = keras.models.Sequential()
model.add(keras.layers.Input(shape=(SEQ_LEN,INPUT_DIMS)))
model.add(keras.layers.LSTM(HIDDEN_UNITS1, return_sequences=True, dropout=d, name= 'lstm1'))
model.add(keras.layers.LSTM(HIDDEN_UNITS2, return_sequences=True, dropout=d,name= 'lstm2'))
model.add(keras.layers.LSTM(HIDDEN_UNITS3, return_sequences=True, dropout=d, name= 'lstm3'))
model.add(keras.layers.TimeDistributed(mdn.MDN(OUTPUT_DIMS, N_MIXES, name='mdn_outputs'),name='td_mdn'))
model.compile(loss=mdn.get_mixture_loss_func(OUTPUT_DIMS,N_MIXES), optimizer=opt)
model.summary()
model.fit(train_dataset,
steps_per_epoch = int(num_train_ex // BATCH_SIZE),
epochs = EPOCHS,
verbose = 1,
validation_data = validation_dataset,
validation_steps = int(num_val_ex// BATCH_SIZE),
callbacks=callbacks, use_multiprocessing=True, workers=8)
Have been using this for-loop for testing the generator:
dataset_check = tf.data.Dataset.from_generator(lambda: data_generator(train=True), (tf.float32, tf.float32))
for x, y in dataset_check.take(1000):
print('ok')