I am trying to tune the batch size for LSTM using hyperband optimisation using the following codes, but it didn't work because the numbers of the training items did not change in different batch size conditions. Do you know how to improve the codes? Thank you in advance.
import keras_tuner as kt
import tensorflow as tf
from tensorflow import keras
import numpy as np
x_train = np.random.rand(63, 92)
y_train = np.random.randint(0,6, (63))
x_val = np.random.rand(63, 92)
y_val = np.random.randint(0,7, (63))
train_set = tf.keras.preprocessing.timeseries_dataset_from_array(
x_train, y_train, sequence_length=10)
val_set = tf.keras.preprocessing.timeseries_dataset_from_array(
x_val, y_val, sequence_length=10)
train_set = tf.keras.preprocessing.timeseries_dataset_from_array(
x_train, y_train, sequence_length=10, batch_size =1)
val_set = tf.keras.preprocessing.timeseries_dataset_from_array(
x_val, y_val, sequence_length=10, batch_size =1)
def model_builder(hp):
lr = hp.Choice('learning_rate', values=[1e-2, 1e-3, 1e-4])
hp_units = hp.Int('units', min_value=32, max_value=256, step=32)
hp_units1 = hp.Int('units1', min_value=32, max_value=256, step=32)
lstm_model = tf.keras.models.Sequential([
tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(hp_units, return_sequences=True)),
tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(hp_units1)),
tf.keras.layers.Dense(units=1)
])
lstm_model.compile(loss='mse',
optimizer=tf.optimizers.Adam(learning_rate=lr),
metrics=['mse'])
return lstm_model
class MyTuner(kt.tuners.Hyperband):
def run_trial(self, trial, *args, **kwargs):
kwargs['batch_size'] = trial.hyperparameters.Int('batch_size', 2, 6, step=2)
return super(MyTuner, self).run_trial(trial, *args, **kwargs)
tuner = MyTuner(model_builder,
objective='val_loss',
max_epochs=4,
factor=3,
directory='KT',
project_name='intro_to_kt0207',
overwrite=True)
early_stop = tf.keras.callbacks.EarlyStopping(
monitor='val_loss', min_delta=0, patience=20, verbose=0,
mode='min', baseline=None, restore_best_weights=True
)
tuner.search(train_set, epochs=1000,
validation_data = val_set)
This is the output:
Search: Running Trial #2
Hyperparameter |Value |Best Value So Far
learning_rate |0.0001 |0.01
units |192 |224
units1 |192 |64
batch_size |4 |2
tuner/epochs |2 |2
tuner/initial_e...|0 |0
tuner/bracket |1 |1
tuner/round |0 |0
Epoch 1/2 54/54 [==============================] - 7s 37ms/step - loss: 4.4262 - mse: 4.4262 - val_loss: 3.4221 - val_mse: 3.4221 Epoch 2/2 54/54 [==============================] - 1s 12ms/step - loss: 3.1213 - mse: 3.1213 - val_loss: 3.4463 - val_mse: 3.4463 Trial 2 Complete [00h 00m 08s] val_loss: 3.42207670211792
Best val_loss So Far: 3.1889588832855225 Total elapsed time: 00h 00m 18s
Search: Running Trial #3
Hyperparameter |Value |Best Value So Far
learning_rate |0.01 |0.01
units |192 |224
units1 |96 |64
batch_size |2 |2
tuner/epochs |2 |2
tuner/initial_e...|0 |0
tuner/bracket |1 |1
tuner/round |0 |0
Epoch 1/2 54/54 [==============================] - 7s 34ms/step - loss: 3.6241 - mse: 3.6241 - val_loss: 3.1699 - val_mse: 3.1699 Epoch 2/2 54/54 [==============================] - 1s 12ms/step - loss: 3.1807 - mse: 3.1807 - val_loss: 3.2480 - val_mse: 3.2480 Trial 3 Complete [00h 00m 08s] val_loss: 3.1699421405792236
Best val_loss So Far: 3.1699421405792236 Total elapsed time: 00h 00m 26s