How can I modify this training script to make it run faster?

Viewed 64

Suppose, I have two GPUs on my computer. My target is to use both GPUs simultaneously for the training of my ANN so that the training can be achieved as fast as possible:

The following is the outline of my MLP training script written in Python:

import os

os.environ["TF_CPP_MIN_LOG_LEVEL"] = "2"
os.environ["CUDA_DEVICE_ORDER"] = "PCI_BUS_ID"
os.environ["CUDA_VISIBLE_DEVICES"] = "0,1" # Use both gpus for training.


import sys, random
import time
import tensorflow as tf
from tensorflow import keras
from tensorflow.keras.models import Sequential
from tensorflow.keras.callbacks import ModelCheckpoint
import numpy as np
from lxml import etree, objectify


# <editor-fold desc="GPU">
# resolve GPU related issues.
try:
    physical_devices = tf.config.list_physical_devices('GPU') 
    for gpu_instance in physical_devices: 
        tf.config.experimental.set_memory_growth(gpu_instance, True)
except Exception as e:
    pass
# END of try
# </editor-fold>

... ... ...
... ... ...
... ... ...

# <editor-fold desc="def create_model()">
def create_model(n_hidden_1, n_hidden_2, n_hidden_3, num_classes, num_features):
    # create the model
    ... ... ...
    ... ... ...
    ... ... ...
    # return model
# END of the function
# </editor-fold>


if __name__ == "__main__":
    # assign hyperparameters.
    ... ... ...
    ... ... ...
    ... ... ... 
    
    # <editor-fold desc="(load data from file)">
    # load training data from the disk
    train_x, train_y, _, validate_x, validate_y, _ = \
        load_data_k(
            fname=input_file_name_str,
            class_index=CLASS_INDEX,
            feature_start_index=FEATURE_INDEX,
            random_n_lines=INPUT_LINES_COUNT,
            validation_part=VALIDATION_PART
        )

    print("training data size : ", len(train_x))
    print("validation data size : ", len(validate_x))
    # </editor-fold>

    # <editor-fold desc="(model creation)">
    # load previously saved NN model
    model = None
    try:
        model = keras.models.load_model(model_file_path)
        print("Loading NN model from file.")
        model.summary()
    except Exception as ex:
        print("No NN model found for loading.")
    # END of try-except
    # </editor-fold>

    # <editor-fold desc="(model run)">
    # # if there is no model loaded, create a new model
    if model is None:
        csv_logger = keras.callbacks.CSVLogger(training_progress_file_path_str)

        checkpoint = ModelCheckpoint(
            model_file_path,
            monitor='loss',
            verbose=1,
            save_best_only=True,
            mode='auto',
            save_freq='epoch'
        )

        callbacks_vector = [
            csv_logger,
            checkpoint
        ]

        # Set mirror strategy
        strategy = tf.distribute.MirroredStrategy(devices=["/device:GPU:0","/device:GPU:1"])

        with strategy.scope():
            print("New NN model created.")
            # create sequential NN model
            model = create_model(
                n_hidden_1=HIDDEN_LAYER_1_NEURON_COUNT,
                n_hidden_2=HIDDEN_LAYER_2_NEURON_COUNT,
                n_hidden_3=HIDDEN_LAYER_3_NEURON_COUNT,
                num_classes=CLASS_COUNT,
                num_features=FEATURES_COUNT
            )

            # Train the model with the new callback
            history = model.fit(
                train_x, train_y,
                validation_data=(validate_x, validate_y),
                batch_size=BATCH_SIZE,
                epochs=EPOCH_SIZE,
                callbacks=[callbacks_vector],
                shuffle=True,
                verbose=2
            )

            print(history.history.keys())
        # END of ... with
    # END of ... if
    # </editor-fold>

I have a feeling that the training process is not fast enough as TensorFlow's GPU-related settings are not properly configured.

Can you check this script and tell me if my GPU-related settings are properly configured or not?

0 Answers
Related