I wrote a lattice dynamics simulation using Metal/Swift on macOS. It contains only highly parallel multiply-and-adds, but I still can't get the Metal/GPU to beat the CPU. (6-core i5 vs Radeon Pro 5300).
The code should execute the kernel 78k times over a dataset consisting of 46080 floats.
EDIT: The 78k iterations should be executed sequentially, as they correspond to time steps of a simulation, and each of them involves ~500k (highly-parallel) floating point operations.
Is there anything basic that I'm missing?
GPU code:
kernel void eom(const device float *mtK [[ buffer(0) ]],
const device float *mtKnl [[ buffer(1) ]],
const device float *mtB [[ buffer(2) ]],
const device float *mtKh [[ buffer(3) ]],
const device float *mtKv [[ buffer(4) ]],
const device float *exm [[ buffer(5) ]],
const device float *exwfm [[ buffer(6) ]],
const device float *sourceX [[ buffer(7) ]],
const device float *sourceV [[ buffer(8) ]],
device float *dest [[ buffer(9) ]],
uint3 id [[ thread_position_in_grid ]]) {
uint materialpoint = basisOffset + id.x + linestride*id.y;
uint samplepoint = basisOffset + id.x + linestride*id.y + blockstride*id.z;
float k = mtK [materialpoint];
float b = mtB [materialpoint];
float knl = mtKnl [materialpoint];
float kh = mtKh [materialpoint];
float kv = mtKv [materialpoint];
float khp = mtKh [materialpoint+linestride];
float kvp = mtKv [materialpoint+1];
float ex = exwfm [id.z]*exm[materialpoint];
dest[samplepoint] = -sourceV[samplepoint]*b -sourceX[samplepoint]*(k + knl*sourceX[samplepoint]*sourceX[samplepoint]) + kv*sourceX[samplepoint-1] + kh*sourceX[samplepoint-linestride] + kvp*sourceX[samplepoint+1] + khp*sourceX[samplepoint+linestride] + ex;
}
CPU code:
let threadgroup_dx = 24
let threadgroup_dy = 32
let threadgroup_dz = 1
let red_widthG = 1
let red_heightG = 2
let waveformStrideG = 30
let device = MTLCreateSystemDefaultDevice()!
let commandQueue = device.makeCommandQueue()!
let library = try device.makeLibrary(filepath: "compute.metallib")
// Initialize buffers
let dydxV = device.makeBuffer(length: totalExtendedSize*MemoryLayout<Float>.stride, options: MTLResourceOptions.storageModeManaged)!
[...] // More buffers and loading code go here
// Create pipeline state
let computeDescriptor = MTLComputePipelineDescriptor()
computeDescriptor.threadGroupSizeIsMultipleOfThreadExecutionWidth = true
computeDescriptor.computeFunction = library.makeFunction(name: "eom")
let pipEOM = try device.makeComputePipelineState(descriptor: computeDescriptor, options: [], reflection: nil)
let commandBuffer = commandQueue.makeCommandBuffer()!
let encoder = commandBuffer.makeComputeCommandEncoder(dispatchType: MTLDispatchType.serial)!
// Computation
for i in 1...78000 {
let offset = i*numSamples
encoder.setComputePipelineState(pipEOM)
encoder.setBuffer(mtrK, offset: 0, index: 0)
encoder.setBuffer(mtrKnl, offset: 0, index: 1)
encoder.setBuffer(mtrB, offset: 0, index: 2)
encoder.setBuffer(mtrKh, offset: 0, index: 3)
encoder.setBuffer(mtrKv, offset: 0, index: 4)
encoder.setBuffer(exm, offset: 0, index: 5)
encoder.setBuffer(wvf, offset: offset, index: 6)
encoder.setBuffer(yX, offset: 0, index: 7)
encoder.setBuffer(yV, offset: 0, index: 8)
encoder.setBuffer(dydxV, offset: 0, index: 9)
let numThreadGroups = MTLSize(width: red_widthG, height: red_heightG, depth: waveformStrideG)
let threadsPerThreadgroup = MTLSize(width: threadgroup_dx, height: threadgroup_dy, depth: threadgroup_dz)
encoder.dispatchThreadgroups(numThreadGroups, threadsPerThreadgroup: threadsPerThreadgroup)
}
encoder.endEncoding()
commandBuffer.commit()
commandBuffer.waitUntilCompleted()