DLL hangs on first call cudaMalloc or cudaMemGetInfo

Viewed 49

I am trying to run a DLL compiled in VS2019, CUDA Toolkit 11.7, Win10, x64. The code is compiled to a x64 target and called by a 64-bit application.

This code snippet (forgive some poor error handling and excessing fflushing please...):

cudaError err;
int N_devices, device;
cudaDeviceProp props;

err = cudaGetDeviceCount(&N_devices);
if (N_devices == 0 || err)
{
    fprintf(logfile, "Error finding any CUDA devices!\n");
    LogCudaError(err, __FILE__, __LINE__);
    return err;
}
fprintf(logfile, "Found %d CUDA device(s)\n", N_devices);

err = cudaGetDevice(&device);
LogCudaError(err, __FILE__, __LINE__);
fprintf(logfile, "Using CUDA device %d\n", device);
fflush(logfile);

err = cudaGetDeviceProperties(&props, device);
if(LogCudaError(err, __FILE__, __LINE__))
{
    fprintf(logfile, "Error getting CUDA device properties!\n");
    return err;
}
fprintf(logfile, "CUDA device properties:\n");
fprintf(logfile, "\tName: %s\n", props.name);
fprintf(logfile, "\tCompute capability: %d.%d\n", props.major, props.minor);
fprintf(logfile, "\tTotal global memory: %0.1f MB\n", (float)(props.totalGlobalMem) / 1048576.f);
fprintf(logfile, "\tShared memory per block: %0.1f kB\n", (float)(props.sharedMemPerBlock >> 10));
fprintf(logfile, "\tMax threads per block: %d\n", props.maxThreadsPerBlock);
fprintf(logfile, "\tClock Rate: %.3f GHz\n", ((float)props.clockRate) / 1048576.f);
fflush(logfile);

int driverVersion, runtimeVersion;
err = cudaDriverGetVersion(&driverVersion);
err = cudaRuntimeGetVersion(&runtimeVersion);
fprintf(logfile, "Driver version: %d\n", driverVersion);
fprintf(logfile, "Runtime version: %d\n", runtimeVersion);
fflush(logfile);

fprintf(logfile, "Memory test:\n\tAttempting to allocate 1 MB...");
fflush(logfile);
float* testptr;
err = cudaMalloc((void**)&testptr, 1024 * 1024 * sizeof(float));
if (!err) fprintf(logfile, "\tSuccessfully allocated 1 MB with address %zx\n", (size_t) testptr); 
else fprintf(logfile, "\tAllocation failed with error code %d\n", err);
fflush(logfile);
fprintf(logfile, "\tFreeing test memory\n"); 
fflush(logfile);
err = cudaFree(testptr);
if (!err) fprintf(logfile, "\tSuccessfully deallocated test array");
else fprintf(logfile, "\tDeallocation failed with error code %d\n", err);
fflush(logfile);

fprintf(logfile, "\tFree memory: %0.1f MB\n", (float)GetFreeMem() / 1048576.f); 
fflush(logfile);
return err;

Produces this output:

Found 1 CUDA device(s)
Using CUDA device 0
CUDA device properties:
    Name: NVIDIA GeForce GTX 1060 6GB
    Compute capability: 6.1
    Total global memory: 6143.8 MB
    Shared memory per block: 48.0 kB
    Max threads per block: 1024
    Clock Rate: 1.629 GHz
Driver version: 11070
Runtime version: 11070
Memory test:
    Attempting to allocate 1 MB...

And then hangs until I force-quit the calling application. I have previously compiled and run CUDA code on this computer/GPU, but that was back probably on CUDA 7.0 or so. Any thoughts or advice? Also calling cudaMemGetInfo() similarly hangs the program.

0 Answers
Related