Great performance gap between PyTorch and Tensorflow

Viewed 1771

PyTorch was my go to framework for deep learning for quite some time, but I decided to give Tensorflow a shot and I experimented a bit how the frameworks compare performance wise. I used the Mnist example from Tensorflow's tutorial site and created same network in Pytorch. I ran both models for 30 epoch on full 60k training examples from Mnist with batch size of 1000.

What I found out is that GPU and CPU utilisation in Tensorflow is much better than that of Pytorch and consequently the train times are much shorter in the former. GPU and CPU utilisation stats as well as corresponding code for both frameworks is found below. From nvidia-smi utility it is visible that Pytorch uses only about 800MB of GPU memory, while Tensorflow essentially uses whole memory.

Train times under above mentioned conditions: TensorFlow: 7.44318 s PyTorch: 27.94735 s

I am wondering wha they did in TensorFlow to be so much more efficient, and if there is any way to achieve comparable performance in Pytorch? Or is there just some mistake in Pytorch version of the code?

Environment settings: PyTorch:

  • Pytorch 1.5.1
  • cuda 10.2

TensorFlow:

  • tensorflow 2.20
  • cuda 10.2

Tensorflow code

import tensorflow as tf
import time

start_time = time.time()

print("Num GPUs Available: ", len(tf.config.experimental.list_physical_devices('GPU')))
tf.debugging.set_log_device_placement(True)
print(tf.test.is_gpu_available(
    cuda_only=False, min_cuda_compute_capability=None
))

with tf.device("/device:CPU:0"):

  mnist = tf.keras.datasets.mnist

  (x_train, y_train), (x_test, y_test) = mnist.load_data()
  x_train, x_test = x_train / 255.0, x_test / 255.0


  model = tf.keras.models.Sequential([
    tf.keras.layers.Flatten(input_shape=(28, 28)),
    tf.keras.layers.Dense(128, activation='relu'),
    tf.keras.layers.Dropout(0.2),
    tf.keras.layers.Dense(10)
  ])


  loss_fn = tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True)
  model.compile(optimizer='adam',
                loss=loss_fn,
                metrics=['accuracy'])
  model.fit(x_train, y_train, epochs=30,batch_size=1000,max_queue_size=10)
  model.evaluate(x_test,  y_test, verbose=2)

  probability_model = tf.keras.Sequential([
    model,
    tf.keras.layers.Softmax()
  ])

  probability_model(x_test[:5])


end_time = time.time()

print(f"TF model took:{end_time-start_time}")

PyTorch code

#Imports from external libraries
import torch
import torchvision
import torchvision.transforms as transforms
import torch.nn as nn
import torch.nn.functional as F
from torch.utils.data import DataLoader
import matplotlib.pyplot as plt
import numpy as np
import torch.optim as optim
import time

#Imports from internal libraries

class MyModel(nn.Module):
    def __init__(self):
        super(MyModel, self).__init__()
        self.hidden1 = nn.Linear(784, 128)
        self.hidden2 = nn.Linear(128,10)
        self.dropout = nn.Dropout(0.2)


    def forward(self, x):
        x = F.relu(self.hidden1(x))
        x = self.dropout(x)
        x = self.hidden2(x)
        x = F.softmax(x, dim=1)
        return x
device = torch.device("cuda:0" if torch.cuda.is_available() else "cpu")
# device = torch.device("cpu")
model = MyModel()
model.to(device)
criterion = nn.CrossEntropyLoss()
optimizer = optim.Adam(model.parameters(), lr=0.001)


train_dataset = torchvision.datasets.MNIST(
    root='./data_mnist',
    train=True,
    download=True,
    transform=transforms.ToTensor()
)

val_dataset = torchvision.datasets.MNIST(
    root='./data_mnist',
    train=False,
    download=True,
    transform=transforms.ToTensor()
)

train_loader = DataLoader(
    train_dataset,
    batch_size=1000,
    shuffle=True,
    num_workers=24
)

val_loader = DataLoader(
    val_dataset,
    batch_size=100,
    shuffle=False,
    num_workers=20
)

start_time = time.time()
torch.backends.cudnn.benchmark = True

for epoch in range(30):
    train_loss = 0.
    val_loss = 0.
    train_acc = 0.
    val_acc = 0.
    
    num_samples = 0
    for data, target in train_loader:
        data = data.to(device)
        target = target.to(device)
        optimizer.zero_grad()
        num_samples += data.size(0)
        output = model(data.view(data.size(0), -1))
        loss = criterion(output, target)
        loss.backward()
        optimizer.step()
        
        train_loss += loss.item()
        train_acc += (torch.argmax(output, 1) == target).float().sum()
    
    print(f"Num datapoints in trainset: {num_samples}")
        
    with torch.no_grad():
        for data, target in val_loader:
            data = data.to(device)
            target = target.to(device)
            output = model(data.view(data.size(0), -1))
            loss = criterion(output, target)            
            val_loss += loss.item()
            val_acc += (torch.argmax(output, 1) == target).float().sum()
    
    train_loss /= len(train_loader)
    train_acc /= len(train_dataset)
    val_loss /= len(val_loader)
    val_acc /= len(val_dataset)
   
    print('Epoch {}, train_loss {}, val_loss {}, train_acc {}, val_acc {}'.format(
        epoch, train_loss, val_loss, train_acc, val_acc))
end_time = time.time()

print(f"pytorch training took: {end_time-start_time} s") 

Tensorflow GPU utilisation enter image description here

Pytorch GPU utilisation enter image description here

EDIT: As pointed out in the comments I changed the number of workers in PyTorch implementation to 8 since I found out that there is no performance improvement with more than 8 workers for this example. I commented out the validation code which was giving about 10 sec overhead, and I removed the softmax function in the forward method of the network.

This gave PyTorch new running time of about 16 seconds which is still about twice the time of the TF implementation of 7.5 seconds.

Edited code:

#Imports from external libraries
import torch
import torchvision
import torchvision.transforms as transforms
import torch.nn as nn
import torch.nn.functional as F
from torch.utils.data import DataLoader
import matplotlib.pyplot as plt
import numpy as np
import torch.optim as optim
import time

#Imports from internal libraries

class MyModel(nn.Module):
    def __init__(self):
        super(MyModel, self).__init__()
        self.hidden1 = nn.Linear(784, 128)
        self.hidden2 = nn.Linear(128,10)
        self.dropout = nn.Dropout(0.2)


    def forward(self, x):
        x = F.relu(self.hidden1(x))
        x = self.dropout(x)
        x = self.hidden2(x)
        return x
device = torch.device("cuda:0" if torch.cuda.is_available() else "cpu")
# device = torch.device("cpu")
model = MyModel()
model.to(device)
criterion = nn.CrossEntropyLoss()
optimizer = optim.Adam(model.parameters(), lr=0.001)


train_dataset = torchvision.datasets.MNIST(
    root='./data_mnist',
    train=True,
    download=True,
    transform=transforms.ToTensor()
)

val_dataset = torchvision.datasets.MNIST(
    root='./data_mnist',
    train=False,
    download=True,
    transform=transforms.ToTensor()
)

train_loader = DataLoader(
    train_dataset,
    batch_size=1000,
    shuffle=True,
    num_workers=8
)

val_loader = DataLoader(
    val_dataset,
    batch_size=100,
    shuffle=False,
    num_workers=20
)

start_time = time.time()
torch.backends.cudnn.benchmark = True

for epoch in range(30):
    train_loss = 0.
    val_loss = 0.
    train_acc = 0.
    val_acc = 0.
    
    num_samples = 0
    for data, target in train_loader:
        data = data.to(device)
        target = target.to(device)
        optimizer.zero_grad()
        num_samples += data.size(0)
        output = model(data.view(data.size(0), -1))
        loss = criterion(output, target)
        loss.backward()
        optimizer.step()
        
        train_loss += loss.item()
        train_acc += (torch.argmax(output, 1) == target).float().sum()
    
    print(f"Num datapoints in trainset: {num_samples}")
        
    # with torch.no_grad():     
    #     for data, target in val_loader:
    #         data = data.to(device)
    #         target = target.to(device)
    #         output = model(data.view(data.size(0), -1))
    #         loss = criterion(output, target)            
    #         val_loss += loss.item()
    #         val_acc += (torch.argmax(output, 1) == target).float().sum()
    
    # train_loss /= len(train_loader)
    # train_acc /= len(train_dataset)
    # val_loss /= len(val_loader)
    # val_acc /= len(val_dataset)

    print('Epoch {}, train_loss {}, val_loss {}, train_acc {}, val_acc {}'.format(
        epoch, train_loss, val_loss, train_acc, val_acc))
end_time = time.time()

print(f"pytorch training took: {end_time-start_time} s") 
0 Answers
Related