I am trying to run a project that use nccl architecture from pytorch(multiprocessing architecture) could I run this project with only one GPU? If the answer is No could I run with one system with 2 graphic card on it with SLI or I should run this project in some different PC? some of script python in this project is as follows :
import torch
from torch import distributed as dist
def synchronize():
if not dist.is_nccl_available():
return
if not dist.is_initialized():
return
world_size = dist.get_world_size()
if world_size == 1:
return
dist.barrier()
def get_rank():
if not dist.is_nccl_available():
return 0
if not dist.is_initialized():
return 0
return dist.get_rank()
def is_main_process():
return get_rank() == 0
def get_world_size():
if not dist.is_nccl_available():
return 1
if not dist.is_initialized():
return 1
return dist.get_world_size()
def broadcast_tensor(tensor, src=0):
world_size = get_world_size()
if world_size < 2:
return tensor
with torch.no_grad():
dist.broadcast(tensor, src=0)
return tensor
def broadcast_scalar(scalar, src=0, device="cpu"):
if get_world_size() < 2:
return scalar
scalar_tensor = torch.tensor(scalar).long().to(device)
scalar_tensor = broadcast_tensor(scalar_tensor, src)
return scalar_tensor.item()
def reduce_tensor(tensor):
world_size = get_world_size()
if world_size < 2:
return tensor
with torch.no_grad():
dist.reduce(tensor, dst=0)
if dist.get_rank() == 0:
tensor = tensor.div(world_size)
return tensor
def gather_tensor(tensor):
world_size = get_world_size()
if world_size < 2:
return tensor
with torch.no_grad():
tensor_list = []
for _ in range(world_size):
tensor_list.append(torch.zeros_like(tensor))
dist.all_gather(tensor_list, tensor)
tensor_list = torch.stack(tensor_list, dim=0)
return tensor_list
def reduce_dict(dictionary):
world_size = get_world_size()
if world_size < 2:
return dictionary
with torch.no_grad():
if len(dictionary) == 0:
return dictionary
keys, values = zip(*sorted(dictionary.items()))
values = torch.stack(values, dim=0)
dist.reduce(values, dst=0)
if dist.get_rank() == 0:
# only main process gets accumulated, so only divide by
# world_size in this case
values /= world_size
reduced_dict = {k: v for k, v in zip(keys, values)}
return reduced_dict
def print_only_main(string):
if is_main_process():
print(string)