When I add workers to neutral network I get an error, pytorch

Viewed 170

I have checked a lot of post and none of them seem to work for me. But when I try to add workers to the dataloader in pytorch it just feeds me an error back. I have tired reading it and figuring it out but I can't seem to find a solution. I assume there is something I'm supposed to add to make the workers able to do their job.

I have 64GB of ram, i9-9900k, and a 3080ti. So I don't think its a memory error is it? I included the error code with 1 worker and 4 workers because they seem to be different. Also it works with zero workers.

here is the error with 4 workers:

    Traceback (most recent call last):
      File "<string>", line 1, in <module>
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 105, in spawn_main
        exitcode = _main(fd)
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 114, in _main     
        prepare(preparation_data)
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 225, in prepare   
        _fixup_main_from_path(data['init_main_from_path'])
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 277, in _fixup_main_from_path
        run_name="__mp_main__")
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\runpy.py", line 263, in run_path
        pkg_name=pkg_name, script_name=fname)
    
                if __name__ == '__main__':
                    freeze_support()
                    ...
    
            The "freeze_support()" line can be omitted if the program
            is not going to be frozen to produce an executable.
    Traceback (most recent call last):
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 990, in _try_get_data
        data = self._data_queue.get(timeout=timeout)
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\queue.py", line 172, in get    raise Empty
    queue.Empty
    The above exception was the direct cause of the following exception:
    
    Traceback (most recent call last):
      File "c:/Users/14055/Desktop/Class 1 Project/Chegg.py", line 202, in <module>
        training()
      File "c:/Users/14055/Desktop/Class 1 Project/Chegg.py", line 122, in training
        for data, target in load_data.train_loader:
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 521, in __next__    data = self._next_data()
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 1186, in _next_data    idx, data = self._get_data()
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 1142, in _get_data    success, data = self._try_get_data()
      File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 1003, in _try_get_data
        raise RuntimeError('DataLoader worker (pid(s) {}) exited unexpectedly'.format(pids_str)) from e
    RuntimeError: DataLoader worker (pid(s) 23204, 7668, 13636, 6132) exited unexpectedly

Error with 1 worker:

Traceback (most recent call last):
  File "<string>", line 1, in <module>
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 105, in spawn_main
    exitcode = _main(fd)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 114, in _main     
    prepare(preparation_data)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 225, in prepare   
    _fixup_main_from_path(data['init_main_from_path'])
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 277, in _fixup_main_from_path
    run_name="__mp_main__")
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\runpy.py", line 263, in run_path
    pkg_name=pkg_name, script_name=fname)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\runpy.py", line 96, in _run_module_code
    mod_name, mod_spec, pkg_name, script_name)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\runpy.py", line 85, in _run_code
    exec(code, run_globals)
  File "c:\Users\14055\Desktop\Class 1 Project\Chegg.py", line 202, in <module>
    training()
  File "c:\Users\14055\Desktop\Class 1 Project\Chegg.py", line 122, in training
    for data, target in load_data.train_loader:
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 359, in __iter__
    return self._get_iterator()
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 305, in _get_iterator
    return _MultiProcessingDataLoaderIter(self)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 918, in __init__
    w.start()
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\process.py", line 105, in start   
    self._popen = self._Popen(self)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\context.py", line 223, in _Popen  
    return _default_context.get_context().Process._Popen(process_obj)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\context.py", line 322, in _Popen  
    return Popen(process_obj)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\popen_spawn_win32.py", line 33, in __init__
    prep_data = spawn.get_preparation_data(process_obj._name)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 143, in get_preparation_data
    _check_not_importing_main()
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\multiprocessing\spawn.py", line 136, in _check_not_importing_main
        This probably means that you are not using fork to start your
        child processes and you have forgotten to use the proper idiom
        in the main module:

            if __name__ == '__main__':
                freeze_support()
                ...

        The "freeze_support()" line can be omitted if the program
        is not going to be frozen to produce an executable.
Traceback (most recent call last):
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 990, in _try_get_data
    data = self._data_queue.get(timeout=timeout)
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\queue.py", line 172, in get
    raise Empty
queue.Empty

The above exception was the direct cause of the following exception:

Traceback (most recent call last):
  File "c:/Users/14055/Desktop/Class 1 Project/Chegg.py", line 202, in <module>
    training()
  File "c:/Users/14055/Desktop/Class 1 Project/Chegg.py", line 122, in training
    for data, target in load_data.train_loader:
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 521, in __next__
    data = self._next_data()
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 1186, in _next_data
    idx, data = self._get_data()
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 1142, in _get_data
    success, data = self._try_get_data()
  File "C:\Users\14055\AppData\Local\Programs\Python\Python36\lib\site-packages\torch\utils\data\dataloader.py", line 1003, in _try_get_data
    raise RuntimeError('DataLoader worker (pid(s) {}) exited unexpectedly'.format(pids_str)) from e
RuntimeError: DataLoader worker (pid(s) 3372) exited unexpectedly

Code:

from numpy import testing
import torch.cuda
import numpy as np
import time
import array as arr
import os
from datetime import date, datetime
from torchvision import datasets
import torchvision.transforms as transforms
import matplotlib.pyplot as plt 
import torch.nn as nn
import torch.nn.functional as F
from torchsummary import summary
torch.cuda.set_device(0)
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
def load_data():
    num_workers = 1
    load_data.batch_size = 20
    transform = transforms.ToTensor()
    train_data = datasets.MNIST(root='data', train=True, download=True, transform=transform)
    load_data.train_loader = torch.utils.data.DataLoader(train_data, 
                                batch_size=load_data.batch_size, num_workers=num_workers, pin_memory=True,
                                shuffle=True)

    test_data = datasets.MNIST(root='data', train=False, download=True, transform=transform)
    load_data.test_loader = torch.utils.data.DataLoader(test_data, 
                                batch_size=load_data.batch_size, num_workers=num_workers, pin_memory=True,
                                shuffle=True)
def visualize():
    dataiter = iter(load_data.train_loader)
    visualize.images, labels = dataiter.next()
    visualize.images = visualize.images.numpy()
    fig = plt.figure(figsize=(25, 4))
    for idx in np.arange(load_data.batch_size):
        ax = fig.add_subplot(2, load_data.batch_size/2, idx+1, xticks=[], yticks=[])
        ax.imshow(np.squeeze(visualize.images[idx]), cmap='gray')
        ax.set_title(str(labels[idx].item()))
    #plt.show()

def fig_values():
    img = np.squeeze(visualize.images[1])
    fig = plt.figure(figsize = (12,12))
    ax = fig.add_subplot(111)
    ax.imshow(img, cmap='gray')
    width, height = img.shape
    thresh = img.max()/2.5
    for x in range(width):
        for y in range(height):
            val = round(img[x][y],2) if img[x][y] !=0 else 0
            ax.annotate(str(val), xy=(y,x),
                        horizontalalignment='center',
                        verticalalignment='center',
                        color='white' if img[x][y]<thresh else 'black')
    #plt.show()

load_data()
#visualize()
#fig_values()

class NeuralNet(nn.Module):
    def __init__(self, gpu = True):
        super(NeuralNet, self ).__init__()
        self.conv1 = nn.Conv2d(in_channels=1, out_channels=128, kernel_size=3, padding=1)
        self.bn1  = nn.BatchNorm2d(num_features=128)

        self.tns1 = nn.Conv2d(in_channels=128, out_channels=4, kernel_size=1, padding=1)

        self.conv2 = nn.Conv2d(in_channels=4, out_channels=16, kernel_size=3, padding=1)
        self.bn2 = nn.BatchNorm2d(num_features=16)

        self.pool1 = nn.MaxPool2d(2,2)

        self.conv3 = nn.Conv2d(in_channels=16, out_channels=16, kernel_size=3, padding=1)
        self.bn3 = nn.BatchNorm2d(num_features=16)

        self.conv4 = nn.Conv2d(in_channels=16, out_channels=32, kernel_size=3, padding=1)
        self.bn4 = nn.BatchNorm2d(num_features=32)

        self.pool2 = nn.MaxPool2d(2,2)

        self.tns2 = nn.Conv2d(in_channels=32, out_channels=16, kernel_size=1, padding=1)

        self.conv5 = nn.Conv2d(in_channels=16, out_channels=16, kernel_size=3, padding=1)
        self.bn5 = nn.BatchNorm2d(num_features=16)

        self.conv6 = nn.Conv2d(in_channels=16, out_channels=32, kernel_size=3, padding=1)
        self.bn6 = nn.BatchNorm2d(num_features=32)

        self.conv7 = nn.Conv2d(in_channels=32, out_channels=10, kernel_size=1, padding=1)
        self.gpool = nn.AvgPool2d(kernel_size=7)

        self.drop = nn.Dropout2d(0.1)

    def forward(self, x):
        x = self.tns1(self.drop(self.bn1(F.relu(self.conv1(x)))))
        x = self.drop(self.bn2(F.relu(self.conv2(x))))
        x = self.pool1(x)
        x = self.drop(self.bn3(F.relu(self.conv3(x))))
        x = self.drop(self.bn4(F.relu(self.conv4(x))))
        x = self.tns2(self.pool2(x))
        x = self.drop(self.bn5(F.relu(self.conv5(x))))
        x = self.drop(self.bn6(F.relu(self.conv6(x))))
        x = self.conv7(x)
        x = self.gpool(x)
        x = x.view(-1, 10)

        return F.log_softmax(x).to(device)

#has antioverfit 
def training():
    model.to(device) 
    optimizer= torch.optim.SGD(model.parameters(), lr=0.003, weight_decay= 0.00005, momentum = .9, nesterov = True)
    n_epochs = 20000
    a = np.float64([9,9,9,9,9]) #antioverfit 
    testing_loss = 0.0
    for epoch in range(n_epochs) :
        if(testing_loss <= a[4]): # part of anti overfit
            train_loss = 0.0
            testing_loss = 0.0
            model.train().to(device)
            for data, target in load_data.train_loader:
                optimizer.zero_grad()
                data = data.to(device) #gpu
                target = target.to(device) #gpu
                output = model(data).to(device)
                loss = F.nll_loss(output, target)
                loss.backward()
                optimizer.step()
                train_loss += loss.item()*data.size(0)
                
            train_loss = train_loss/len(load_data.train_loader.dataset)
            print('Epoch: {} \tTraining Loss: {:.6f}'.format(epoch+1, train_loss))          
            model.eval().to(device)  # Gets Validation loss 
            train_loss = 0.0       
            with torch.no_grad():
                for data, target in load_data.test_loader:   

                    data = data.to(device)
                    target = target.to(device)

                    output = model(data).to(device)
                    loss =F.nll_loss(output, target)
                    testing_loss += loss.item()*data.size(0)
            testing_loss = testing_loss / len(load_data.test_loader.dataset)           
            print('Validation loss = ' , testing_loss)           
            a = np.insert(a,0,testing_loss) # part of anti overfit          
            a = np.delete(a,5)
    print('Validation loss = ' , testing_loss) 

def evalution():
    test_loss = 0.0
    class_correct = list(0. for i in range(10))
    class_total = list(0. for i in range(10))
    model.eval().to(device)

    for data, target in load_data.test_loader:
        data = data.to(device)
        target = target.to(device)
        output = model(data).to(device)
        loss =F.nll_loss(output, target)
        test_loss += loss.item()*data.size(0)
        _, pred = torch.max(output, 1)
        correct = np.squeeze(pred.eq(target.data.view_as(pred))).to(device)
        for i in range(load_data.batch_size):
            try:
                label = target.data[i]
                class_correct[label] += correct[i].item()
                class_total[label] += 1
            except IndexError:
                break

    # calculate and print avg test loss
    test_loss = test_loss/len(load_data.test_loader.dataset)
    print('Test Loss: {:.6f}\n'.format(test_loss))

    for i in range(10):
        if class_total[i] > 0:
            print('Test Accuracy of %5s: %2d%% (%2d/%2d)' % (
                str(i), 100 * class_correct[i] / class_total[i],
                np.sum(class_correct[i]), np.sum(class_total[i])))
        else:
            print('Test Accuracy of %5s: N/A (no training examples)' )

    print('\nTest Accuracy (Overall): %2d%% (%2d/%2d)' % (
        100. * np.sum(class_correct) / np.sum(class_total),
        np.sum(class_correct), np.sum(class_total)))
    acc = (
        100. * np.sum(class_correct) / np.sum(class_total),
        np.sum(class_correct), np.sum(class_total))

    name = f"model-{acc}.pt"
    name2 = f"model-{acc}.pth"
    save_path = os.path.join("models", name)
    save_path2 = os.path.join("models", name2)
    torch.save(model, save_path)
    torch.save(model, save_path2)


model = NeuralNet().to(device)
summary(model, input_size=(1, 28, 28))
training()
evalution()
0 Answers
Related