26 lines
959 B
Plaintext
26 lines
959 B
Plaintext
|
|
#pragma once
|
||
|
|
|
||
|
|
#include <cassert>
|
||
|
|
#include <cstdio>
|
||
|
|
#include <cstdlib>
|
||
|
|
#include <cublas_v2.h>
|
||
|
|
#include <cuda_runtime.h>
|
||
|
|
|
||
|
|
template <const uint BLOCKSIZE>
|
||
|
|
// __global__ is used to specify that the function is run on GPU, called by host (CPU)
|
||
|
|
__global__ void sgemm_global_mem_coalesce(int M, int N, int K, float alpha,
|
||
|
|
const float *A, const float *B,
|
||
|
|
float beta, float *C) {
|
||
|
|
const int cRow = blockIdx.x * BLOCKSIZE + (threadIdx.x / BLOCKSIZE); // note that blockDim is now 1-dimensional
|
||
|
|
const int cCol = blockIdx.y * BLOCKSIZE + (threadIdx.x % BLOCKSIZE);
|
||
|
|
|
||
|
|
// if statement is necessary to make things work under tile quantization
|
||
|
|
if (cRow < M && cCol < N) {
|
||
|
|
float tmp = 0.0;
|
||
|
|
for (int i = 0; i < K; ++i) {
|
||
|
|
tmp += A[cRow * K + i] * B[i * N + cCol];
|
||
|
|
}
|
||
|
|
C[cRow * N + cCol] = alpha * tmp + beta * C[cRow * N + cCol];
|
||
|
|
}
|
||
|
|
}
|