Why are these simple Metal GPU compute kernels slower or equal than a CPU implementation?

Viewed 140

I wrote a lattice dynamics simulation using Metal/Swift on macOS. It contains only highly parallel multiply-and-adds, but I still can't get the Metal/GPU to beat the CPU. (6-core i5 vs Radeon Pro 5300).

The code should execute the kernel 78k times over a dataset consisting of 46080 floats.

EDIT: The 78k iterations should be executed sequentially, as they correspond to time steps of a simulation, and each of them involves ~500k (highly-parallel) floating point operations.

Is there anything basic that I'm missing?

GPU code:

kernel void eom(const device float *mtK     [[ buffer(0) ]],
            const device float *mtKnl   [[ buffer(1) ]],
            const device float *mtB     [[ buffer(2) ]],
            const device float *mtKh    [[ buffer(3) ]],
            const device float *mtKv    [[ buffer(4) ]],
            const device float *exm     [[ buffer(5) ]],
            const device float *exwfm   [[ buffer(6) ]],
            const device float *sourceX [[ buffer(7) ]],
            const device float *sourceV [[ buffer(8) ]],
                  device float *dest    [[ buffer(9) ]],
            uint3 id                    [[ thread_position_in_grid ]]) {

uint materialpoint = basisOffset + id.x + linestride*id.y;
uint samplepoint   = basisOffset + id.x + linestride*id.y + blockstride*id.z;

float k    = mtK   [materialpoint];
float b    = mtB   [materialpoint];
float knl  = mtKnl [materialpoint];
float kh   = mtKh  [materialpoint];
float kv   = mtKv  [materialpoint];
float khp  = mtKh  [materialpoint+linestride];
float kvp  = mtKv  [materialpoint+1];
float ex   = exwfm [id.z]*exm[materialpoint];

dest[samplepoint] = -sourceV[samplepoint]*b -sourceX[samplepoint]*(k + knl*sourceX[samplepoint]*sourceX[samplepoint]) + kv*sourceX[samplepoint-1] + kh*sourceX[samplepoint-linestride] + kvp*sourceX[samplepoint+1] + khp*sourceX[samplepoint+linestride] + ex; 
}

CPU code:

let threadgroup_dx = 24
let threadgroup_dy = 32
let threadgroup_dz = 1

let red_widthG = 1
let red_heightG = 2
let waveformStrideG = 30

let device = MTLCreateSystemDefaultDevice()!
let commandQueue = device.makeCommandQueue()!
let library = try device.makeLibrary(filepath: "compute.metallib")

// Initialize buffers
let dydxV = device.makeBuffer(length: totalExtendedSize*MemoryLayout<Float>.stride, options: MTLResourceOptions.storageModeManaged)!
[...] // More buffers and loading code go here

// Create pipeline state
let computeDescriptor = MTLComputePipelineDescriptor()
computeDescriptor.threadGroupSizeIsMultipleOfThreadExecutionWidth = true
computeDescriptor.computeFunction = library.makeFunction(name: "eom")
let pipEOM = try device.makeComputePipelineState(descriptor: computeDescriptor, options: [], reflection: nil)

let commandBuffer = commandQueue.makeCommandBuffer()!
let encoder = commandBuffer.makeComputeCommandEncoder(dispatchType: MTLDispatchType.serial)!

// Computation
for i in 1...78000 {
    let offset = i*numSamples
    encoder.setComputePipelineState(pipEOM)
    encoder.setBuffer(mtrK,     offset: 0,      index: 0)
    encoder.setBuffer(mtrKnl,   offset: 0,      index: 1)
    encoder.setBuffer(mtrB,     offset: 0,      index: 2)
    encoder.setBuffer(mtrKh,    offset: 0,      index: 3)
    encoder.setBuffer(mtrKv,    offset: 0,      index: 4)
    encoder.setBuffer(exm,      offset: 0,      index: 5)
    encoder.setBuffer(wvf,      offset: offset, index: 6)
    encoder.setBuffer(yX,       offset: 0,      index: 7)
    encoder.setBuffer(yV,       offset: 0,      index: 8)
    encoder.setBuffer(dydxV,    offset: 0,      index: 9)
    let numThreadGroups = MTLSize(width: red_widthG, height: red_heightG, depth: waveformStrideG)
    let threadsPerThreadgroup = MTLSize(width: threadgroup_dx, height: threadgroup_dy, depth: threadgroup_dz)
    encoder.dispatchThreadgroups(numThreadGroups, threadsPerThreadgroup: threadsPerThreadgroup)
}

encoder.endEncoding()
commandBuffer.commit()
commandBuffer.waitUntilCompleted()
0 Answers
Related