2
0

SWDEV-470698 - fix formatting, add format check workflow (#657)

Este cometimento está contido em:
Danylo Lytovchenko
2025-08-20 16:28:06 +02:00
cometido por GitHub
ascendente 5840940caa
cometimento f7338717ae
1574 ficheiros modificados com 162972 adições e 199346 eliminações
@@ -38,86 +38,87 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
out[y * width + x] = in[x * width + y];
out[y * width + x] = in[x * width + y];
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(
matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix,
gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
+105 -104
Ver ficheiro
@@ -36,128 +36,129 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
asm volatile("v_mov_b32_e32 %0, %1" : "=v"(out[x * width + y]) : "v"(in[y * width + x]));
asm volatile("v_mov_b32_e32 %0, %1" : "=v"(out[x * width + y]) : "v"(in[y * width + x]));
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
hipEvent_t start, stop;
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
hipEvent_t start, stop;
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyHostToDevice time taken = %6.3fms\n", eventMs);
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Lauching kernel from host
hipLaunchKernelGGL(
matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
checkHipErrors(hipEventDestroy(start));
checkHipErrors(hipEventDestroy(stop));
printf("hipMemcpyDeviceToHost time taken = %6.3fms\n", eventMs);
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("gpu%f cpu %f \n", TransposeMatrix[i], cpuTransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyHostToDevice time taken = %6.3fms\n", eventMs);
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix,
gpuMatrix, WIDTH);
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
checkHipErrors(hipEventDestroy(start));
checkHipErrors(hipEventDestroy(stop));
printf("hipMemcpyDeviceToHost time taken = %6.3fms\n", eventMs);
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("gpu%f cpu %f \n", TransposeMatrix[i], cpuTransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
@@ -21,66 +21,74 @@ THE SOFTWARE.
*/
#include "hip/hip_runtime.h"
extern "C" __global__ void tex2dKernelChar(char* outputData,hipTextureObject_t texObj, int width, int height) {
extern "C" __global__ void tex2dKernelChar(char* outputData, hipTextureObject_t texObj, int width,
int height) {
#if !defined(__HIP_NO_IMAGE_SUPPORT) || !__HIP_NO_IMAGE_SUPPORT
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<char>(texObj, x, y);
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<char>(texObj, x, y);
#endif
}
extern "C" __global__ void tex2dKernelShort(short* outputData,hipTextureObject_t texObj, int width, int height) {
extern "C" __global__ void tex2dKernelShort(short* outputData, hipTextureObject_t texObj, int width,
int height) {
#if !defined(__HIP_NO_IMAGE_SUPPORT) || !__HIP_NO_IMAGE_SUPPORT
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<short>(texObj, x, y);
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<short>(texObj, x, y);
#endif
}
extern "C" __global__ void tex2dKernelInt(int* outputData,hipTextureObject_t texObj ,int width, int height) {
extern "C" __global__ void tex2dKernelInt(int* outputData, hipTextureObject_t texObj, int width,
int height) {
#if !defined(__HIP_NO_IMAGE_SUPPORT) || !__HIP_NO_IMAGE_SUPPORT
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<int>(texObj, x, y);
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<int>(texObj, x, y);
#endif
}
extern "C" __global__ void tex2dKernelFloat(float* outputData,hipTextureObject_t texObj, int width, int height) {
extern "C" __global__ void tex2dKernelFloat(float* outputData, hipTextureObject_t texObj, int width,
int height) {
#if !defined(__HIP_NO_IMAGE_SUPPORT) || !__HIP_NO_IMAGE_SUPPORT
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<float>(texObj, x, y);
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<float>(texObj, x, y);
#endif
}
extern "C" __global__ void tex2dKernelChar4(char4* outputData,hipTextureObject_t texObj, int width, int height) {
extern "C" __global__ void tex2dKernelChar4(char4* outputData, hipTextureObject_t texObj, int width,
int height) {
#if !defined(__HIP_NO_IMAGE_SUPPORT) || !__HIP_NO_IMAGE_SUPPORT
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<char4>(texObj, x, y);
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<char4>(texObj, x, y);
#endif
}
extern "C" __global__ void tex2dKernelShort4(short4* outputData,hipTextureObject_t texObj, int width, int height) {
extern "C" __global__ void tex2dKernelShort4(short4* outputData, hipTextureObject_t texObj,
int width, int height) {
#if !defined(__HIP_NO_IMAGE_SUPPORT) || !__HIP_NO_IMAGE_SUPPORT
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<short4>(texObj, x, y);
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<short4>(texObj, x, y);
#endif
}
extern "C" __global__ void tex2dKernelInt4(int4* outputData,hipTextureObject_t texObj, int width, int height) {
extern "C" __global__ void tex2dKernelInt4(int4* outputData, hipTextureObject_t texObj, int width,
int height) {
#if !defined(__HIP_NO_IMAGE_SUPPORT) || !__HIP_NO_IMAGE_SUPPORT
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<int4>(texObj, x, y);
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<int4>(texObj, x, y);
#endif
}
extern "C" __global__ void tex2dKernelFloat4(float4* outputData,hipTextureObject_t texObj, int width, int height) {
extern "C" __global__ void tex2dKernelFloat4(float4* outputData, hipTextureObject_t texObj,
int width, int height) {
#if !defined(__HIP_NO_IMAGE_SUPPORT) || !__HIP_NO_IMAGE_SUPPORT
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<float4>(texObj, x, y);
int x = blockIdx.x * blockDim.x + threadIdx.x;
int y = blockIdx.y * blockDim.y + threadIdx.y;
outputData[y * width + x] = tex2D<float4>(texObj, x, y);
#endif
}
@@ -30,8 +30,7 @@ THE SOFTWARE.
bool testResult = true;
template<typename T,
typename std::enable_if<std::is_arithmetic<T>::value>::type *t = nullptr>
template <typename T, typename std::enable_if<std::is_arithmetic<T>::value>::type* t = nullptr>
static inline hipArray_Format getArrayFormat() {
if (std::is_same<char, T>::value) {
return HIP_AD_FORMAT_SIGNED_INT8;
@@ -45,97 +44,81 @@ static inline hipArray_Format getArrayFormat() {
return HIP_AD_FORMAT_HALF;
}
template<typename T,
typename std::enable_if<!std::is_arithmetic<T>::value>::type *t = nullptr>
template <typename T, typename std::enable_if<!std::is_arithmetic<T>::value>::type* t = nullptr>
static inline hipArray_Format getArrayFormat() {
return getArrayFormat<decltype(T::x)>();
}
template<typename T>
static inline constexpr int rank() {
template <typename T> static inline constexpr int rank() {
return sizeof(T) / sizeof(decltype(T::x));
}
#ifdef __HIP_PLATFORM_NVIDIA__
template <typename T,
typename std::enable_if<std::is_same<T, int4>::value ||
std::is_same<T, short4>::value ||
std::is_same<T, char4>::value ||
std::is_same<T, float4>::value>::type *t = nullptr>
static inline bool operator!=(const T& a, const T& b)
{
return (a.x != b.x) || (a.y != b.y) || (a.z != b.z) || (a.w != b.w);
typename std::enable_if<std::is_same<T, int4>::value || std::is_same<T, short4>::value ||
std::is_same<T, char4>::value ||
std::is_same<T, float4>::value>::type* t = nullptr>
static inline bool operator!=(const T& a, const T& b) {
return (a.x != b.x) || (a.y != b.y) || (a.z != b.z) || (a.w != b.w);
}
#endif
template<typename T>
static inline T getRandom() {
template <typename T> static inline T getRandom() {
double r = 0;
if (std::is_signed < T > ::value) {
if (std::is_signed<T>::value) {
r = (std::rand() - RAND_MAX / 2.0) / (RAND_MAX / 2.0 + 1.);
} else {
r = std::rand() / (RAND_MAX + 1.);
}
return static_cast<T>(std::numeric_limits < T > ::max() * r);
return static_cast<T>(std::numeric_limits<T>::max() * r);
}
template<typename T,
typename std::enable_if<std::is_arithmetic<T>::value>::type* = nullptr>
template <typename T, typename std::enable_if<std::is_arithmetic<T>::value>::type* = nullptr>
static inline constexpr int getChannels() {
return 1;
}
template<typename T,
typename std::enable_if<!std::is_arithmetic<T>::value>::type *t = nullptr,
typename std::enable_if<rank<T>() != 0>::type *r = nullptr>
template <typename T, typename std::enable_if<!std::is_arithmetic<T>::value>::type* t = nullptr,
typename std::enable_if<rank<T>() != 0>::type* r = nullptr>
static inline constexpr int getChannels() {
return rank<T>();
}
template<typename T,
typename std::enable_if<std::is_arithmetic<T>::value>::type* = nullptr>
static inline void printDiff(const int &i, const int &j, const T &expected,
const T &output) {
std::cout << "Difference [" << i << " " << j << "]: " << expected << " - "
<< output << "\n";
template <typename T, typename std::enable_if<std::is_arithmetic<T>::value>::type* = nullptr>
static inline void printDiff(const int& i, const int& j, const T& expected, const T& output) {
std::cout << "Difference [" << i << " " << j << "]: " << expected << " - " << output << "\n";
}
template<typename T,
typename std::enable_if<!std::is_arithmetic<T>::value>::type* = nullptr,
typename std::enable_if<rank<T>() == 4>::type* = nullptr>
static inline void printDiff(const int &i, const int &j, const T &expected,
const T &output) {
std::cout << "Difference [" << i << " " << j << "]: " << expected.x << ","
<< expected.y << "," << expected.z << "," << expected.w << " - "
<< output.x << "," << output.y << "," << output.z << "," << output.w
<< "\n";
template <typename T, typename std::enable_if<!std::is_arithmetic<T>::value>::type* = nullptr,
typename std::enable_if<rank<T>() == 4>::type* = nullptr>
static inline void printDiff(const int& i, const int& j, const T& expected, const T& output) {
std::cout << "Difference [" << i << " " << j << "]: " << expected.x << "," << expected.y << ","
<< expected.z << "," << expected.w << " - " << output.x << "," << output.y << ","
<< output.z << "," << output.w << "\n";
}
template<typename T,
typename std::enable_if<std::is_arithmetic<T>::value>::type* = nullptr>
static inline void initVal(T &val) {
template <typename T, typename std::enable_if<std::is_arithmetic<T>::value>::type* = nullptr>
static inline void initVal(T& val) {
val = getRandom<T>();
}
template<typename T,
typename std::enable_if<!std::is_arithmetic<T>::value>::type* = nullptr,
typename std::enable_if<rank<T>() == 4>::type* = nullptr>
static inline void initVal(T &val) {
template <typename T, typename std::enable_if<!std::is_arithmetic<T>::value>::type* = nullptr,
typename std::enable_if<rank<T>() == 4>::type* = nullptr>
static inline void initVal(T& val) {
val.x = getRandom<decltype(T::x)>();
val.y = getRandom<decltype(T::x)>();
val.z = getRandom<decltype(T::x)>();
val.w = getRandom<decltype(T::x)>();
}
template<typename T>
bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
template <typename T> bool runTest(hipModule_t& module, const char* refName, const char* funcName) {
hipArray_Format format = getArrayFormat<T>();
int channels = getChannels<T>();
unsigned int width = 256;
unsigned int height = 256;
unsigned int size = width * height * sizeof(T);
T *hData = (T*) malloc(size);
T* hData = (T*)malloc(size);
memset(hData, 0, size);
for (int i = 0; i < height; i++) {
for (int j = 0; j < width; j++) {
@@ -149,8 +132,8 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
const size_t spitch = width * sizeof(T);
checkHipErrors(hipMemcpy2DToArray(array, 0, 0, hData, spitch, width * sizeof(T),
height, hipMemcpyHostToDevice));
checkHipErrors(hipMemcpy2DToArray(array, 0, 0, hData, spitch, width * sizeof(T), height,
hipMemcpyHostToDevice));
hipResourceDesc resDesc;
memset(&resDesc, 0, sizeof(resDesc));
@@ -168,24 +151,24 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
hipTextureObject_t texObj;
checkHipErrors(hipCreateTextureObject(&texObj, &resDesc, &texDesc, nullptr));
T *dData = NULL;
checkHipErrors(hipMalloc((void** )&dData, size));
T* dData = NULL;
checkHipErrors(hipMalloc((void**)&dData, size));
struct {
void *_Ad;
void* _Ad;
hipTextureObject_t _texObj;
unsigned int _Bd;
unsigned int _Cd;
} args;
args._Ad = (void*) dData;
args._Ad = (void*)dData;
args._texObj = texObj;
args._Bd = width;
args._Cd = height;
size_t sizeTemp = sizeof(args);
void *config[] = { HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &sizeTemp, HIP_LAUNCH_PARAM_END };
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &sizeTemp,
HIP_LAUNCH_PARAM_END};
hipFunction_t Function;
checkHipErrors(hipModuleGetFunction(&Function, module, funcName));
@@ -193,11 +176,10 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
int temp1 = width / 16;
int temp2 = height / 16;
checkHipErrors(
hipModuleLaunchKernel(Function, 16, 16, 1, temp1, temp2, 1, 0, 0, NULL,
(void** )&config));
hipModuleLaunchKernel(Function, 16, 16, 1, temp1, temp2, 1, 0, 0, NULL, (void**)&config));
checkHipErrors(hipDeviceSynchronize());
T *hOutputData = (T*) malloc(size);
T* hOutputData = (T*)malloc(size);
memset(hOutputData, 0, size);
checkHipErrors(hipMemcpy(hOutputData, dData, size, hipMemcpyDeviceToHost));
@@ -220,10 +202,9 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
}
inline bool isImageSupported() {
int imageSupport = 1;
int imageSupport = 1;
#ifdef __HIP_PLATFORM_AMD__
checkHipErrors(hipDeviceGetAttribute(&imageSupport, hipDeviceAttributeImageSupport,
0));
checkHipErrors(hipDeviceGetAttribute(&imageSupport, hipDeviceAttributeImageSupport, 0));
#endif
return imageSupport != 0;
}
@@ -38,86 +38,87 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
out[y * width + x] = in[x * width + y];
out[y * width + x] = in[x * width + y];
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(
matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix,
gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
+116 -116
Ver ficheiro
@@ -23,153 +23,153 @@ THE SOFTWARE.
#define NUM 1000000
// Device (Kernel) function
__global__ void multiply(float* C, float* A, float* B, int N){
__global__ void multiply(float* C, float* A, float* B, int N) {
int tx = blockDim.x * blockIdx.x + threadIdx.x;
int tx = blockDim.x*blockIdx.x+threadIdx.x;
if (tx < N){
C[tx] = A[tx] * B[tx];
}
if (tx < N) {
C[tx] = A[tx] * B[tx];
}
}
// CPU implementation
void multiplyCPU(float* C, float* A, float* B, int N){
for(unsigned int i=0; i<N; i++){
C[i] = A[i] * B[i];
}
void multiplyCPU(float* C, float* A, float* B, int N) {
for (unsigned int i = 0; i < N; i++) {
C[i] = A[i] * B[i];
}
}
void launchKernel(float* C, float* A, float* B, bool manual){
void launchKernel(float* C, float* A, float* B, bool manual) {
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipEvent_t start, stop;
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
const unsigned threadsperblock = 32;
const unsigned blocks = (NUM / threadsperblock) + 1;
hipEvent_t start, stop;
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
const unsigned threadsperblock = 32;
const unsigned blocks = (NUM/threadsperblock)+1;
int mingridSize = 0;
int gridSize = 0;
int blockSize = 0;
int mingridSize = 0;
int gridSize = 0;
int blockSize = 0;
if (manual) {
blockSize = threadsperblock;
gridSize = blocks;
std::cout << std::endl << "Manual Configuration with block size " << blockSize << std::endl;
} else {
checkHipErrors(hipOccupancyMaxPotentialBlockSize(&mingridSize, &blockSize, multiply, 0, 0));
std::cout << std::endl
<< "Automatic Configuation based on hipOccupancyMaxPotentialBlockSize " << std::endl;
std::cout << "Suggested blocksize is " << blockSize << ", Minimum gridsize is " << mingridSize
<< std::endl;
gridSize = (NUM / blockSize) + 1;
}
if (manual){
blockSize = threadsperblock;
gridSize = blocks;
std::cout << std::endl << "Manual Configuration with block size " << blockSize << std::endl;
}
else{
checkHipErrors(hipOccupancyMaxPotentialBlockSize(&mingridSize, &blockSize, multiply, 0, 0));
std::cout << std::endl << "Automatic Configuation based on hipOccupancyMaxPotentialBlockSize " << std::endl;
std::cout << "Suggested blocksize is " << blockSize << ", Minimum gridsize is " << mingridSize << std::endl;
gridSize = (NUM/blockSize)+1;
}
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Launching the Kernel from Host
hipLaunchKernelGGL(multiply, dim3(gridSize), dim3(blockSize), 0, 0, C, A, B, NUM);
// Launching the Kernel from Host
hipLaunchKernelGGL(multiply, dim3(gridSize), dim3(blockSize), 0, 0, C, A, B, NUM);
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
checkHipErrors(hipEventDestroy(start));
checkHipErrors(hipEventDestroy(stop));
checkHipErrors(hipEventDestroy(start));
checkHipErrors(hipEventDestroy(stop));
// Calculate Occupancy
int numBlock = 0;
checkHipErrors(hipOccupancyMaxActiveBlocksPerMultiprocessor(&numBlock, multiply, blockSize, 0));
//Calculate Occupancy
int numBlock = 0;
checkHipErrors(hipOccupancyMaxActiveBlocksPerMultiprocessor(&numBlock, multiply, blockSize, 0));
if(devProp.maxThreadsPerMultiProcessor){
std::cout << "Theoretical Occupancy is " << (double)numBlock* blockSize/devProp.maxThreadsPerMultiProcessor * 100 << "%" << std::endl;
}
if (devProp.maxThreadsPerMultiProcessor) {
std::cout << "Theoretical Occupancy is "
<< (double)numBlock * blockSize / devProp.maxThreadsPerMultiProcessor * 100 << "%"
<< std::endl;
}
}
int main() {
float *A, *B, *C0, *C1, *cpuC;
float *Ad, *Bd, *C0d, *C1d;
int errors=0;
int i;
float *A, *B, *C0, *C1, *cpuC;
float *Ad, *Bd, *C0d, *C1d;
int errors = 0;
int i;
// initialize the input data
A = (float *)malloc(NUM * sizeof(float));
B = (float *)malloc(NUM * sizeof(float));
C0 = (float *)malloc(NUM * sizeof(float));
C1 = (float *)malloc(NUM * sizeof(float));
cpuC = (float *)malloc(NUM * sizeof(float));
// initialize the input data
A = (float*)malloc(NUM * sizeof(float));
B = (float*)malloc(NUM * sizeof(float));
C0 = (float*)malloc(NUM * sizeof(float));
C1 = (float*)malloc(NUM * sizeof(float));
cpuC = (float*)malloc(NUM * sizeof(float));
for(i=0; i< NUM; i++){
A[i] = i;
B[i] = i;
}
for (i = 0; i < NUM; i++) {
A[i] = i;
B[i] = i;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&Ad, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&Bd, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&C0d, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&C1d, NUM * sizeof(float)));
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&Ad, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&Bd, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&C0d, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&C1d, NUM * sizeof(float)));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(Ad,A,NUM * sizeof(float), hipMemcpyHostToDevice));
checkHipErrors(hipMemcpy(Bd,B,NUM * sizeof(float), hipMemcpyHostToDevice));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(Ad, A, NUM * sizeof(float), hipMemcpyHostToDevice));
checkHipErrors(hipMemcpy(Bd, B, NUM * sizeof(float), hipMemcpyHostToDevice));
//Kernel launch with manual/default block size
launchKernel(C0d, Ad, Bd, 1);
// Kernel launch with manual/default block size
launchKernel(C0d, Ad, Bd, 1);
//Kernel launch with the block size suggested by hipOccupancyMaxPotentialBlockSize
launchKernel(C1d, Ad, Bd, 0);
// Kernel launch with the block size suggested by hipOccupancyMaxPotentialBlockSize
launchKernel(C1d, Ad, Bd, 0);
// Memory transfer from device to host
checkHipErrors(hipMemcpy(C0,C0d, NUM * sizeof(float), hipMemcpyDeviceToHost));
checkHipErrors(hipMemcpy(C1,C1d, NUM * sizeof(float), hipMemcpyDeviceToHost));
// Memory transfer from device to host
checkHipErrors(hipMemcpy(C0, C0d, NUM * sizeof(float), hipMemcpyDeviceToHost));
checkHipErrors(hipMemcpy(C1, C1d, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU computation
multiplyCPU(cpuC, A, B, NUM);
// CPU computation
multiplyCPU(cpuC, A, B, NUM);
//verify the results
double eps = 1.0E-6;
// verify the results
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(C0[i] - cpuC[i]) > eps) {
errors++;
}
}
for (i = 0; i < NUM; i++) {
if (std::abs(C0[i] - cpuC[i]) > eps) {
errors++;
}
}
if (errors != 0){
printf("\nManual Test FAILED: %d errors\n", errors);
errors=0;
} else {
printf("\nManual Test PASSED!\n");
}
if (errors != 0) {
printf("\nManual Test FAILED: %d errors\n", errors);
errors = 0;
} else {
printf("\nManual Test PASSED!\n");
}
for (i = 0; i < NUM; i++) {
if (std::abs(C1[i] - cpuC[i]) > eps) {
errors++;
}
}
for (i = 0; i < NUM; i++) {
if (std::abs(C1[i] - cpuC[i]) > eps) {
errors++;
}
}
if (errors != 0){
printf("\n Automatic Test FAILED: %d errors\n", errors);
} else {
printf("\nAutomatic Test PASSED!\n");
}
if (errors != 0) {
printf("\n Automatic Test FAILED: %d errors\n", errors);
} else {
printf("\nAutomatic Test PASSED!\n");
}
checkHipErrors(hipFree(Ad));
checkHipErrors(hipFree(Bd));
checkHipErrors(hipFree(C0d));
checkHipErrors(hipFree(C1d));
checkHipErrors(hipFree(Ad));
checkHipErrors(hipFree(Bd));
checkHipErrors(hipFree(C0d));
checkHipErrors(hipFree(C1d));
free(A);
free(B);
free(C0);
free(C1);
free(cpuC);
return 0;
free(A);
free(B);
free(C0);
free(C1);
free(cpuC);
return 0;
}
+4 -4
Ver ficheiro
@@ -21,10 +21,10 @@ THE SOFTWARE.
#include <iostream>
#include "hip_helper.h"
#define THREADS_PER_BLOCK 64
#define BLOCKS_PER_GRID 4
#define SIZE (BLOCKS_PER_GRID * THREADS_PER_BLOCK)
#define NOT_SUPPORTED -99 // dummy number indicates unsupported operation
#define THREADS_PER_BLOCK 64
#define BLOCKS_PER_GRID 4
#define SIZE (BLOCKS_PER_GRID * THREADS_PER_BLOCK)
#define NOT_SUPPORTED -99 // dummy number indicates unsupported operation
// Using __gfx*__ macro one can have GPU architecture specific code flow
// For example: If below kernel runs on gfx908 it will increment 'in' by 'value' and store into
@@ -22,7 +22,4 @@
#include <hip/hip_runtime.h>
__device__ int square_me(int A) {
return A*A;
}
__device__ int square_me(int A) { return A * A; }
@@ -38,40 +38,39 @@
extern __device__ int square_me(int);
__global__ void square_and_save(int* A, int* B) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
B[tid] = square_me(A[tid]);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
B[tid] = square_me(A[tid]);
}
void run_test2() {
int *A_h, *B_h, *A_d, *B_d;
A_h = new int[LEN];
B_h = new int[LEN];
for (unsigned i = 0; i < LEN; i++) {
A_h[i] = i;
B_h[i] = 0;
}
size_t valbytes = LEN*sizeof(int);
int *A_h, *B_h, *A_d, *B_d;
A_h = new int[LEN];
B_h = new int[LEN];
for (unsigned i = 0; i < LEN; i++) {
A_h[i] = i;
B_h[i] = 0;
}
size_t valbytes = LEN * sizeof(int);
HIP_ASSERT(hipMalloc((void**)&A_d, valbytes));
HIP_ASSERT(hipMalloc((void**)&B_d, valbytes));
HIP_ASSERT(hipMalloc((void**)&A_d, valbytes));
HIP_ASSERT(hipMalloc((void**)&B_d, valbytes));
HIP_ASSERT(hipMemcpy(A_d, A_h, valbytes, hipMemcpyHostToDevice));
hipLaunchKernelGGL(square_and_save, dim3(LEN/64), dim3(64),
0, 0, A_d, B_d);
HIP_ASSERT(hipMemcpy(B_h, B_d, valbytes, hipMemcpyDeviceToHost));
HIP_ASSERT(hipMemcpy(A_d, A_h, valbytes, hipMemcpyHostToDevice));
hipLaunchKernelGGL(square_and_save, dim3(LEN / 64), dim3(64), 0, 0, A_d, B_d);
HIP_ASSERT(hipMemcpy(B_h, B_d, valbytes, hipMemcpyDeviceToHost));
for (unsigned i = 0; i < LEN; i++) {
assert(A_h[i]*A_h[i] == B_h[i]);
}
for (unsigned i = 0; i < LEN; i++) {
assert(A_h[i] * A_h[i] == B_h[i]);
}
HIP_ASSERT(hipFree(A_d));
HIP_ASSERT(hipFree(B_d));
delete [] A_h;
delete [] B_h;
std::cout << "Test Passed!\n";
HIP_ASSERT(hipFree(A_d));
HIP_ASSERT(hipFree(B_d));
delete[] A_h;
delete[] B_h;
std::cout << "Test Passed!\n";
}
int main(){
int main() {
// Run test that generates static lib with ar
run_test2();
}
@@ -22,7 +22,7 @@
extern void run_test1();
int main(){
int main() {
// Run test that generates static lib with -emit-static-lib
run_test1();
}
@@ -26,69 +26,64 @@ THE SOFTWARE.
*
* Square each element in the array A and write to array C.
*/
template <typename T>
__global__ void
vector_square(T *C_d, T *A_d, size_t N)
{
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x ;
template <typename T> __global__ void vector_square(T* C_d, T* A_d, size_t N) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
for (size_t i=offset; i<N; i+=stride) {
C_d[i] = A_d[i] * A_d[i];
}
for (size_t i = offset; i < N; i += stride) {
C_d[i] = A_d[i] * A_d[i];
}
}
int main(int argc, char *argv[])
{
float *A_d, *C_d;
float *A_h, *C_h;
size_t N = 1000000;
size_t Nbytes = N * sizeof(float);
int main(int argc, char* argv[]) {
float *A_d, *C_d;
float *A_h, *C_h;
size_t N = 1000000;
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
hipDeviceProp_t props;
checkHipErrors(hipGetDeviceProperties(&props, 0 /*deviceID*/));
printf("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
checkHipErrors(A_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
C_h = (float*)malloc(Nbytes);
checkHipErrors(C_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
A_h[i] = 1.618f + i;
printf("info: allocate host mem (%6.2f MB)\n", 2 * Nbytes / 1024.0 / 1024.0);
A_h = (float*)malloc(Nbytes);
checkHipErrors(A_h == 0 ? hipErrorMemoryAllocation : hipSuccess);
C_h = (float*)malloc(Nbytes);
checkHipErrors(C_h == 0 ? hipErrorMemoryAllocation : hipSuccess);
// Fill with Phi + i
for (size_t i = 0; i < N; i++) {
A_h[i] = 1.618f + i;
}
printf("info: allocate device mem (%6.2f MB)\n", 2 * Nbytes / 1024.0 / 1024.0);
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf("info: copy Host2Device\n");
checkHipErrors(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
printf("info: launch 'vector_square' kernel\n");
vector_square<<<blocks, threadsPerBlock>>>(C_d, A_d, N);
printf("info: copy Device2Host\n");
checkHipErrors(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf("info: checkHipErrors result\n");
for (size_t i = 0; i < N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
checkHipErrors(hipErrorUnknown);
}
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
checkHipErrors(hipFree(A_d));
checkHipErrors(hipFree(C_d));
free(A_h);
free(C_h);
printf ("info: copy Host2Device\n");
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
printf ("info: launch 'vector_square' kernel\n");
vector_square <<<blocks, threadsPerBlock>>> (C_d, A_d, N);
printf ("info: copy Device2Host\n");
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: checkHipErrors result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
checkHipErrors(hipErrorUnknown);
}
}
checkHipErrors(hipFree(A_d));
checkHipErrors(hipFree(C_d));
free(A_h);
free(C_h);
printf ("PASSED!\n");
printf("PASSED!\n");
}
@@ -26,67 +26,62 @@ THE SOFTWARE.
*
* Square each element in the array A and write to array C.
*/
template <typename T>
__global__ void
vector_square(T *C_d, T *A_d, size_t N)
{
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x ;
template <typename T> __global__ void vector_square(T* C_d, T* A_d, size_t N) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
for (size_t i=offset; i<N; i+=stride) {
C_d[i] = A_d[i] * A_d[i];
}
for (size_t i = offset; i < N; i += stride) {
C_d[i] = A_d[i] * A_d[i];
}
}
int main(int argc, char *argv[])
{
float *A_d, *C_d;
float *A_h, *C_h;
size_t N = 1000000;
size_t Nbytes = N * sizeof(float);
int main(int argc, char* argv[]) {
float *A_d, *C_d;
float *A_h, *C_h;
size_t N = 1000000;
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
hipDeviceProp_t props;
checkHipErrors(hipGetDeviceProperties(&props, 0 /*deviceID*/));
printf("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
checkHipErrors(A_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
C_h = (float*)malloc(Nbytes);
checkHipErrors(C_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
A_h[i] = 1.618f + i;
printf("info: allocate host mem (%6.2f MB)\n", 2 * Nbytes / 1024.0 / 1024.0);
A_h = (float*)malloc(Nbytes);
checkHipErrors(A_h == 0 ? hipErrorMemoryAllocation : hipSuccess);
C_h = (float*)malloc(Nbytes);
checkHipErrors(C_h == 0 ? hipErrorMemoryAllocation : hipSuccess);
// Fill with Phi + i
for (size_t i = 0; i < N; i++) {
A_h[i] = 1.618f + i;
}
printf("info: allocate device mem (%6.2f MB)\n", 2 * Nbytes / 1024.0 / 1024.0);
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf("info: copy Host2Device\n");
checkHipErrors(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
printf("info: launch 'vector_square' kernel\n");
vector_square<<<blocks, threadsPerBlock>>>(C_d, A_d, N);
printf("info: copy Device2Host\n");
checkHipErrors(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf("info: checkHipErrors result\n");
for (size_t i = 0; i < N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
checkHipErrors(hipErrorUnknown);
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf ("info: copy Host2Device\n");
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
printf ("info: launch 'vector_square' kernel\n");
vector_square <<<blocks, threadsPerBlock>>> (C_d, A_d, N);
printf ("info: copy Device2Host\n");
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: checkHipErrors result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
checkHipErrors(hipErrorUnknown);
}
}
checkHipErrors(hipFree(A_d));
checkHipErrors(hipFree(C_d));
free(A_h);
free(C_h);
printf ("PASSED!\n");
}
checkHipErrors(hipFree(A_d));
checkHipErrors(hipFree(C_d));
free(A_h);
free(C_h);
printf("PASSED!\n");
}
@@ -27,67 +27,62 @@ THE SOFTWARE.
/*
* Square each element in the array A and write to array C.
*/
template <typename T>
__global__ void
vector_square(T *C_d, T *A_d, size_t N)
{
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x ;
template <typename T> __global__ void vector_square(T* C_d, T* A_d, size_t N) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
for (size_t i=offset; i<N; i+=stride) {
C_d[i] = A_d[i] * A_d[i];
}
for (size_t i = offset; i < N; i += stride) {
C_d[i] = A_d[i] * A_d[i];
}
}
int main(int argc, char *argv[])
{
float *A_d, *C_d;
float *A_h, *C_h;
size_t N = 1000000;
size_t Nbytes = N * sizeof(float);
int main(int argc, char* argv[]) {
float *A_d, *C_d;
float *A_h, *C_h;
size_t N = 1000000;
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
hipDeviceProp_t props;
checkHipErrors(hipGetDeviceProperties(&props, 0 /*deviceID*/));
printf("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
checkHipErrors(A_h == 0 ? hipErrorOutOfMemory : hipSuccess );
C_h = (float*)malloc(Nbytes);
checkHipErrors(C_h == 0 ? hipErrorOutOfMemory : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
A_h[i] = 1.618f + i;
printf("info: allocate host mem (%6.2f MB)\n", 2 * Nbytes / 1024.0 / 1024.0);
A_h = (float*)malloc(Nbytes);
checkHipErrors(A_h == 0 ? hipErrorOutOfMemory : hipSuccess);
C_h = (float*)malloc(Nbytes);
checkHipErrors(C_h == 0 ? hipErrorOutOfMemory : hipSuccess);
// Fill with Phi + i
for (size_t i = 0; i < N; i++) {
A_h[i] = 1.618f + i;
}
printf("info: allocate device mem (%6.2f MB)\n", 2 * Nbytes / 1024.0 / 1024.0);
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf("info: copy Host2Device\n");
checkHipErrors(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
printf("info: launch 'vector_square' kernel\n");
hipLaunchKernelGGL(vector_square, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d, N);
printf("info: copy Device2Host\n");
checkHipErrors(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf("info: checkHipErrors result\n");
for (size_t i = 0; i < N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
checkHipErrors(hipErrorUnknown);
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf ("info: copy Host2Device\n");
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
printf ("info: launch 'vector_square' kernel\n");
hipLaunchKernelGGL(vector_square, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d, N);
printf ("info: copy Device2Host\n");
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: checkHipErrors result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
checkHipErrors(hipErrorUnknown);
}
}
checkHipErrors(hipFree(A_d));
checkHipErrors(hipFree(C_d));
free(A_h);
free(C_h);
printf ("PASSED!\n");
}
checkHipErrors(hipFree(A_d));
checkHipErrors(hipFree(C_d));
free(A_h);
free(C_h);
printf("PASSED!\n");
}
@@ -37,86 +37,87 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
out[y * width + x] = in[x * width + y];
out[y * width + x] = in[x * width + y];
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(
matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix,
gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
+104 -103
Ver ficheiro
@@ -36,127 +36,128 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
out[y * width + x] = in[x * width + y];
out[y * width + x] = in[x * width + y];
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
hipEvent_t start, stop;
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
hipEvent_t start, stop;
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyHostToDevice time taken = %6.3fms\n", eventMs);
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Lauching kernel from host
hipLaunchKernelGGL(
matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyDeviceToHost time taken = %6.3fms\n", eventMs);
checkHipErrors(hipEventDestroy(start));
checkHipErrors(hipEventDestroy(stop));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyHostToDevice time taken = %6.3fms\n", eventMs);
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix,
gpuMatrix, WIDTH);
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
// Record the start event
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// Record the stop event
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyDeviceToHost time taken = %6.3fms\n", eventMs);
checkHipErrors(hipEventDestroy(start));
checkHipErrors(hipEventDestroy(stop));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
@@ -27,67 +27,62 @@ THE SOFTWARE.
/*
* Square each element in the array A and write to array C.
*/
template <typename T>
__global__ void
vector_square(T *C_d, T *A_d, size_t N)
{
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x ;
template <typename T> __global__ void vector_square(T* C_d, T* A_d, size_t N) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
for (size_t i=offset; i<N; i+=stride) {
C_d[i] = A_d[i] * A_d[i];
}
for (size_t i = offset; i < N; i += stride) {
C_d[i] = A_d[i] * A_d[i];
}
}
int main(int argc, char *argv[])
{
float *A_d, *C_d;
float *A_h, *C_h;
size_t N = 1000000;
size_t Nbytes = N * sizeof(float);
int main(int argc, char* argv[]) {
float *A_d, *C_d;
float *A_h, *C_h;
size_t N = 1000000;
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
hipDeviceProp_t props;
checkHipErrors(hipGetDeviceProperties(&props, 0 /*deviceID*/));
printf("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
checkHipErrors(A_h == 0 ? hipErrorOutOfMemory : hipSuccess );
C_h = (float*)malloc(Nbytes);
checkHipErrors(C_h == 0 ? hipErrorOutOfMemory : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
A_h[i] = 1.618f + i;
printf("info: allocate host mem (%6.2f MB)\n", 2 * Nbytes / 1024.0 / 1024.0);
A_h = (float*)malloc(Nbytes);
checkHipErrors(A_h == 0 ? hipErrorOutOfMemory : hipSuccess);
C_h = (float*)malloc(Nbytes);
checkHipErrors(C_h == 0 ? hipErrorOutOfMemory : hipSuccess);
// Fill with Phi + i
for (size_t i = 0; i < N; i++) {
A_h[i] = 1.618f + i;
}
printf("info: allocate device mem (%6.2f MB)\n", 2 * Nbytes / 1024.0 / 1024.0);
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf("info: copy Host2Device\n");
checkHipErrors(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
printf("info: launch 'vector_square' kernel\n");
hipLaunchKernelGGL(vector_square, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d, N);
printf("info: copy Device2Host\n");
checkHipErrors(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf("info: check result\n");
for (size_t i = 0; i < N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
checkHipErrors(hipErrorUnknown);
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf ("info: copy Host2Device\n");
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
printf ("info: launch 'vector_square' kernel\n");
hipLaunchKernelGGL(vector_square, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d, N);
printf ("info: copy Device2Host\n");
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: check result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
checkHipErrors(hipErrorUnknown);
}
}
checkHipErrors(hipFree(A_d));
checkHipErrors(hipFree(C_d));
free(A_h);
free(C_h);
printf ("PASSED!\n");
}
checkHipErrors(hipFree(A_d));
checkHipErrors(hipFree(C_d));
free(A_h);
free(C_h);
printf("PASSED!\n");
}
+97 -95
Ver ficheiro
@@ -35,7 +35,7 @@ static constexpr auto NUM_THREADS{128};
static constexpr auto NUM_BLOCKS{32};
static constexpr auto saxpy{
R"(
R"(
#include "test_header.h"
#include "test_header1.h"
extern "C"
@@ -49,109 +49,111 @@ void saxpy(real a, realptr x, realptr y, realptr out, size_t n)
}
)"};
int main()
{
using namespace std;
int main() {
using namespace std;
hiprtcProgram prog;
int num_headers = 2;
vector<const char*> header_names;
vector<const char*> header_sources;
header_names.push_back("test_header.h");
header_names.push_back("test_header1.h");
header_sources.push_back("#ifndef HIPRTC_TEST_HEADER_H\n#define HIPRTC_TEST_HEADER_H\ntypedef float real;\n#endif //HIPRTC_TEST_HEADER_H\n");
header_sources.push_back("#ifndef HIPRTC_TEST_HEADER1_H\n#define HIPRTC_TEST_HEADER1_H\ntypedef float* realptr;\n#endif //HIPRTC_TEST_HEADER1_H\n");
hiprtcCreateProgram(&prog, // prog
saxpy, // buffer
"saxpy.cu", // name
num_headers, // numHeaders
&header_sources[0], // headers
&header_names[0]); // includeNames
hiprtcProgram prog;
int num_headers = 2;
vector<const char*> header_names;
vector<const char*> header_sources;
header_names.push_back("test_header.h");
header_names.push_back("test_header1.h");
header_sources.push_back(
"#ifndef HIPRTC_TEST_HEADER_H\n#define HIPRTC_TEST_HEADER_H\ntypedef float real;\n#endif "
"//HIPRTC_TEST_HEADER_H\n");
header_sources.push_back(
"#ifndef HIPRTC_TEST_HEADER1_H\n#define HIPRTC_TEST_HEADER1_H\ntypedef float* "
"realptr;\n#endif //HIPRTC_TEST_HEADER1_H\n");
hiprtcCreateProgram(&prog, // prog
saxpy, // buffer
"saxpy.cu", // name
num_headers, // numHeaders
&header_sources[0], // headers
&header_names[0]); // includeNames
hipDeviceProp_t props;
int device = 0;
checkHipErrors(hipGetDeviceProperties(&props, device));
hipDeviceProp_t props;
int device = 0;
checkHipErrors(hipGetDeviceProperties(&props, device));
const char* options[] = {};
const char* options[] = {};
hiprtcResult compileResult{hiprtcCompileProgram(prog, 0, options)};
hiprtcResult compileResult{hiprtcCompileProgram(prog, 0, options)};
size_t logSize;
hiprtcGetProgramLogSize(prog, &logSize);
size_t logSize;
hiprtcGetProgramLogSize(prog, &logSize);
if (logSize) {
string log(logSize, '\0');
hiprtcGetProgramLog(prog, &log[0]);
if (logSize) {
string log(logSize, '\0');
hiprtcGetProgramLog(prog, &log[0]);
cout << log << '\n';
cout << log << '\n';
}
if (compileResult != HIPRTC_SUCCESS) {
cout << "Compilation failed." << endl;
}
size_t codeSize;
hiprtcGetCodeSize(prog, &codeSize);
vector<char> code(codeSize);
hiprtcGetCode(prog, code.data());
hiprtcDestroyProgram(&prog);
hipModule_t module;
hipFunction_t kernel;
checkHipErrors(hipModuleLoadData(&module, code.data()));
checkHipErrors(hipModuleGetFunction(&kernel, module, "saxpy"));
size_t n = NUM_THREADS * NUM_BLOCKS;
size_t bufferSize = n * sizeof(float);
float a = 5.1f;
unique_ptr<float[]> hX{new float[n]};
unique_ptr<float[]> hY{new float[n]};
unique_ptr<float[]> hOut{new float[n]};
for (size_t i = 0; i < n; ++i) {
hX[i] = static_cast<float>(i);
hY[i] = static_cast<float>(i * 2);
}
hipDeviceptr_t dX, dY, dOut;
checkHipErrors(hipMalloc((void**)&dX, bufferSize));
checkHipErrors(hipMalloc((void**)&dY, bufferSize));
checkHipErrors(hipMalloc((void**)&dOut, bufferSize));
checkHipErrors(hipMemcpyHtoD(dX, hX.get(), bufferSize));
checkHipErrors(hipMemcpyHtoD(dY, hY.get(), bufferSize));
struct {
float a_;
hipDeviceptr_t b_;
hipDeviceptr_t c_;
hipDeviceptr_t d_;
size_t e_;
} args{a, dX, dY, dOut, n};
auto size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
checkHipErrors(hipModuleLaunchKernel(kernel, NUM_BLOCKS, 1, 1, NUM_THREADS, 1, 1, 0, nullptr,
nullptr, config));
checkHipErrors(hipMemcpyDtoH(hOut.get(), dOut, bufferSize));
for (size_t i = 0; i < n; ++i) {
if (fabs(a * hX[i] + hY[i] - hOut[i]) > fabs(hOut[i]) * 1e-6) {
cout << "Validation failed." << endl;
}
}
if (compileResult != HIPRTC_SUCCESS) {
cout << "Compilation failed." << endl;
}
checkHipErrors(hipFree((void*)dX));
checkHipErrors(hipFree((void*)dY));
checkHipErrors(hipFree((void*)dOut));
size_t codeSize;
hiprtcGetCodeSize(prog, &codeSize);
checkHipErrors(hipModuleUnload(module));
vector<char> code(codeSize);
hiprtcGetCode(prog, code.data());
hiprtcDestroyProgram(&prog);
hipModule_t module;
hipFunction_t kernel;
checkHipErrors(hipModuleLoadData(&module, code.data()));
checkHipErrors(hipModuleGetFunction(&kernel, module, "saxpy"));
size_t n = NUM_THREADS * NUM_BLOCKS;
size_t bufferSize = n * sizeof(float);
float a = 5.1f;
unique_ptr<float[]> hX{new float[n]};
unique_ptr<float[]> hY{new float[n]};
unique_ptr<float[]> hOut{new float[n]};
for (size_t i = 0; i < n; ++i) {
hX[i] = static_cast<float>(i);
hY[i] = static_cast<float>(i * 2);
}
hipDeviceptr_t dX, dY, dOut;
checkHipErrors(hipMalloc((void **)&dX, bufferSize));
checkHipErrors(hipMalloc((void **)&dY, bufferSize));
checkHipErrors(hipMalloc((void **)&dOut, bufferSize));
checkHipErrors(hipMemcpyHtoD(dX, hX.get(), bufferSize));
checkHipErrors(hipMemcpyHtoD(dY, hY.get(), bufferSize));
struct {
float a_;
hipDeviceptr_t b_;
hipDeviceptr_t c_;
hipDeviceptr_t d_;
size_t e_;
} args{a, dX, dY, dOut, n};
auto size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
checkHipErrors(hipModuleLaunchKernel(kernel, NUM_BLOCKS, 1, 1, NUM_THREADS, 1, 1,
0, nullptr, nullptr, config));
checkHipErrors(hipMemcpyDtoH(hOut.get(), dOut, bufferSize));
for (size_t i = 0; i < n; ++i) {
if (fabs(a * hX[i] + hY[i] - hOut[i]) > fabs(hOut[i])* 1e-6) {
cout << "Validation failed." << endl;
}
}
checkHipErrors(hipFree((void *)dX));
checkHipErrors(hipFree((void *)dY));
checkHipErrors(hipFree((void *)dOut));
checkHipErrors(hipModuleUnload(module));
cout << "SAXPY test completed" << endl;
cout << "SAXPY test completed" << endl;
}
@@ -36,93 +36,94 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
__shared__ float sharedMem[WIDTH * WIDTH];
__shared__ float sharedMem[WIDTH * WIDTH];
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
sharedMem[y * width + x] = in[x * width + y];
sharedMem[y * width + x] = in[x * width + y];
__syncthreads();
__syncthreads();
out[y * width + x] = sharedMem[y * width + x];
out[y * width + x] = sharedMem[y * width + x];
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(
matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("%d cpu: %f gpu %f\n", i, cpuTransposeMatrix[i], TransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix,
gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("%d cpu: %f gpu %f\n", i, cpuTransposeMatrix[i], TransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
+66 -65
Ver ficheiro
@@ -36,89 +36,90 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
int x = blockDim.x * blockIdx.x + threadIdx.x;
int x = blockDim.x * blockIdx.x + threadIdx.x;
float val = in[x];
float val = in[x];
for (int i = 0; i < width; i++) {
for (int j = 0; j < width; j++) out[i * width + j] = __shfl(val, j * width + i);
}
for (int i = 0; i < width; i++) {
for (int j = 0; j < width; j++) out[i * width + j] = __shfl(val, j * width + i);
}
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(1), dim3(THREADS_PER_BLOCK_X * THREADS_PER_BLOCK_Y), 0,
0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("%d cpu: %f gpu %f\n", i, cpuTransposeMatrix[i], TransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(1), dim3(THREADS_PER_BLOCK_X * THREADS_PER_BLOCK_Y), 0, 0,
gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("%d cpu: %f gpu %f\n", i, cpuTransposeMatrix[i], TransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
+65 -64
Ver ficheiro
@@ -37,87 +37,88 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
float val = in[y * width + x];
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
float val = in[y * width + x];
out[x * width + y] = __shfl(val, y * width + x);
out[x * width + y] = __shfl(val, y * width + x);
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(1), dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0,
gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("%d cpu: %f gpu %f\n", i, cpuTransposeMatrix[i], TransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(1), dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0,
gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("%d cpu: %f gpu %f\n", i, cpuTransposeMatrix[i], TransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
@@ -36,93 +36,95 @@ THE SOFTWARE.
// Device (Kernel) function, it must be void
__global__ void matrixTranspose(float* out, float* in, const int width) {
extern __shared__ float sharedMem[];
extern __shared__ float sharedMem[];
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
sharedMem[y * width + x] = in[x * width + y];
sharedMem[y * width + x] = in[x * width + y];
__syncthreads();
__syncthreads();
out[y * width + x] = sharedMem[y * width + x];
out[y * width + x] = sharedMem[y * width + x];
}
// CPU implementation of matrix transpose
void matrixTransposeCPUReference(float* output, float* input, const unsigned int width) {
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
for (unsigned int j = 0; j < width; j++) {
for (unsigned int i = 0; i < width; i++) {
output[i * width + j] = input[j * width + i];
}
}
}
int main() {
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* Matrix;
float* TransposeMatrix;
float* cpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
float* gpuMatrix;
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
int i;
int errors;
int i;
int errors;
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
Matrix = (float*)malloc(NUM * sizeof(float));
TransposeMatrix = (float*)malloc(NUM * sizeof(float));
cpuTransposeMatrix = (float*)malloc(NUM * sizeof(float));
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
// initialize the input data
for (i = 0; i < NUM; i++) {
Matrix[i] = (float)i * 10.0f;
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose,
dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), sizeof(float) * WIDTH * WIDTH,
0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("%d cpu: %f gpu %f\n", i, cpuTransposeMatrix[i], TransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("dynamic_shared PASSED!\n");
}
// allocate the memory on the device side
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// Memory transfer from host to device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), sizeof(float) * WIDTH * WIDTH,
0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
// verify the results
errors = 0;
double eps = 1.0E-6;
for (i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[i] - cpuTransposeMatrix[i]) > eps) {
printf("%d cpu: %f gpu %f\n", i, cpuTransposeMatrix[i], TransposeMatrix[i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("dynamic_shared PASSED!\n");
}
// free the resources on device side
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
free(TransposeMatrix);
free(cpuTransposeMatrix);
return errors;
return errors;
}
+70 -71
Ver ficheiro
@@ -34,106 +34,105 @@ THE SOFTWARE.
using namespace std;
__global__ void matrixTranspose_static_shared(float* out, float* in,
const int width) {
__shared__ float sharedMem[WIDTH * WIDTH];
__global__ void matrixTranspose_static_shared(float* out, float* in, const int width) {
__shared__ float sharedMem[WIDTH * WIDTH];
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
sharedMem[y * width + x] = in[x * width + y];
sharedMem[y * width + x] = in[x * width + y];
__syncthreads();
__syncthreads();
out[y * width + x] = sharedMem[y * width + x];
out[y * width + x] = sharedMem[y * width + x];
}
__global__ void matrixTranspose_dynamic_shared(float* out, float* in,
const int width) {
extern __shared__ float sharedMem[];
__global__ void matrixTranspose_dynamic_shared(float* out, float* in, const int width) {
extern __shared__ float sharedMem[];
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
sharedMem[y * width + x] = in[x * width + y];
sharedMem[y * width + x] = in[x * width + y];
__syncthreads();
__syncthreads();
out[y * width + x] = sharedMem[y * width + x];
out[y * width + x] = sharedMem[y * width + x];
}
void MultipleStream(float** data, float* randArray, float** gpuTransposeMatrix,
float** TransposeMatrix, int width) {
const int num_streams = 2;
hipStream_t streams[num_streams];
const int num_streams = 2;
hipStream_t streams[num_streams];
for (int i = 0; i < num_streams; i++) checkHipErrors(hipStreamCreate(&streams[i]));
for (int i = 0; i < num_streams; i++) checkHipErrors(hipStreamCreate(&streams[i]));
for (int i = 0; i < num_streams; i++) {
checkHipErrors(hipMalloc((void**)&data[i], NUM * sizeof(float)));
checkHipErrors(hipMemcpyAsync(data[i], randArray, NUM * sizeof(float), hipMemcpyHostToDevice, streams[i]));
}
for (int i = 0; i < num_streams; i++) {
checkHipErrors(hipMalloc((void**)&data[i], NUM * sizeof(float)));
checkHipErrors(
hipMemcpyAsync(data[i], randArray, NUM * sizeof(float), hipMemcpyHostToDevice, streams[i]));
}
hipLaunchKernelGGL(matrixTranspose_static_shared,
dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, streams[0],
gpuTransposeMatrix[0], data[0], width);
hipLaunchKernelGGL(matrixTranspose_static_shared,
dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, streams[0],
gpuTransposeMatrix[0], data[0], width);
hipLaunchKernelGGL(matrixTranspose_dynamic_shared,
dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), sizeof(float) * WIDTH * WIDTH,
streams[1], gpuTransposeMatrix[1], data[1], width);
hipLaunchKernelGGL(matrixTranspose_dynamic_shared,
dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), sizeof(float) * WIDTH * WIDTH,
streams[1], gpuTransposeMatrix[1], data[1], width);
for (int i = 0; i < num_streams; i++)
checkHipErrors(hipMemcpyAsync(TransposeMatrix[i], gpuTransposeMatrix[i], NUM * sizeof(float),
hipMemcpyDeviceToHost, streams[i]));
for (int i = 0; i < num_streams; i++)
checkHipErrors(hipMemcpyAsync(TransposeMatrix[i], gpuTransposeMatrix[i], NUM * sizeof(float),
hipMemcpyDeviceToHost, streams[i]));
}
int main() {
checkHipErrors(hipSetDevice(0));
checkHipErrors(hipSetDevice(0));
float *data[2], *TransposeMatrix[2], *gpuTransposeMatrix[2], *randArray;
float *data[2], *TransposeMatrix[2], *gpuTransposeMatrix[2], *randArray;
int width = WIDTH;
int width = WIDTH;
randArray = (float*)malloc(NUM * sizeof(float));
randArray = (float*)malloc(NUM * sizeof(float));
TransposeMatrix[0] = (float*)malloc(NUM * sizeof(float));
TransposeMatrix[1] = (float*)malloc(NUM * sizeof(float));
TransposeMatrix[0] = (float*)malloc(NUM * sizeof(float));
TransposeMatrix[1] = (float*)malloc(NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix[0], NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix[1], NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix[0], NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix[1], NUM * sizeof(float)));
for (int i = 0; i < NUM; i++) {
randArray[i] = (float)i * 1.0f;
for (int i = 0; i < NUM; i++) {
randArray[i] = (float)i * 1.0f;
}
MultipleStream(data, randArray, gpuTransposeMatrix, TransposeMatrix, width);
checkHipErrors(hipDeviceSynchronize());
// verify the results
int errors = 0;
double eps = 1.0E-6;
for (int i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[0][i] - TransposeMatrix[1][i]) > eps) {
printf("%d stream0: %f stream1 %f\n", i, TransposeMatrix[0][i], TransposeMatrix[1][i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("stream PASSED!\n");
}
MultipleStream(data, randArray, gpuTransposeMatrix, TransposeMatrix, width);
free(randArray);
for (int i = 0; i < 2; i++) {
checkHipErrors(hipFree(data[i]));
checkHipErrors(hipFree(gpuTransposeMatrix[i]));
free(TransposeMatrix[i]);
}
checkHipErrors(hipDeviceSynchronize());
// verify the results
int errors = 0;
double eps = 1.0E-6;
for (int i = 0; i < NUM; i++) {
if (std::abs(TransposeMatrix[0][i] - TransposeMatrix[1][i]) > eps) {
printf("%d stream0: %f stream1 %f\n", i, TransposeMatrix[0][i], TransposeMatrix[1][i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("stream PASSED!\n");
}
free(randArray);
for (int i = 0; i < 2; i++) {
checkHipErrors(hipFree(data[i]));
checkHipErrors(hipFree(gpuTransposeMatrix[i]));
free(TransposeMatrix[i]);
}
checkHipErrors(hipDeviceReset());
return 0;
checkHipErrors(hipDeviceReset());
return 0;
}
+143 -148
Ver ficheiro
@@ -34,193 +34,188 @@ using namespace std;
#define KRED "\x1B[31m"
#define failed(...) \
printf("%serror: ", KRED); \
printf(__VA_ARGS__); \
printf("\n"); \
printf("error: TEST FAILED\n%s", KNRM); \
abort();
printf("%serror: ", KRED); \
printf(__VA_ARGS__); \
printf("\n"); \
printf("error: TEST FAILED\n%s", KNRM); \
abort();
#define HIPCHECK(error) \
{ \
hipError_t localError = error; \
if ((localError != hipSuccess)&& (localError != hipErrorPeerAccessAlreadyEnabled)&& \
(localError != hipErrorPeerAccessNotEnabled )) { \
printf("%serror: '%s'(%d) from %s at %s:%d%s\n", KRED, hipGetErrorString(localError), \
localError, #error, __FILE__, __LINE__, KNRM); \
failed("API returned error code."); \
} \
}
{ \
hipError_t localError = error; \
if ((localError != hipSuccess) && (localError != hipErrorPeerAccessAlreadyEnabled) && \
(localError != hipErrorPeerAccessNotEnabled)) { \
printf("%serror: '%s'(%d) from %s at %s:%d%s\n", KRED, hipGetErrorString(localError), \
localError, #error, __FILE__, __LINE__, KNRM); \
failed("API returned error code."); \
} \
}
void checkPeer2PeerSupport() {
int gpuCount;
int canAccessPeer;
int gpuCount;
int canAccessPeer;
HIPCHECK(hipGetDeviceCount(&gpuCount));
HIPCHECK(hipGetDeviceCount(&gpuCount));
for (int currentGpu = 0; currentGpu < gpuCount; currentGpu++) {
HIPCHECK(hipSetDevice(currentGpu));
for (int currentGpu = 0; currentGpu < gpuCount; currentGpu++) {
HIPCHECK(hipSetDevice(currentGpu));
for (int peerGpu = 0; peerGpu < currentGpu; peerGpu++) {
if (currentGpu != peerGpu) {
HIPCHECK(hipDeviceCanAccessPeer(&canAccessPeer, currentGpu, peerGpu));
printf("currentGpu#%d canAccessPeer: peerGpu#%d=%d\n", currentGpu, peerGpu,
canAccessPeer);
}
for (int peerGpu = 0; peerGpu < currentGpu; peerGpu++) {
if (currentGpu != peerGpu) {
HIPCHECK(hipDeviceCanAccessPeer(&canAccessPeer, currentGpu, peerGpu));
printf("currentGpu#%d canAccessPeer: peerGpu#%d=%d\n", currentGpu, peerGpu, canAccessPeer);
}
HIPCHECK(hipSetDevice(peerGpu));
HIPCHECK(hipDeviceReset());
}
HIPCHECK(hipSetDevice(currentGpu));
HIPCHECK(hipDeviceReset());
HIPCHECK(hipSetDevice(peerGpu));
HIPCHECK(hipDeviceReset());
}
HIPCHECK(hipSetDevice(currentGpu));
HIPCHECK(hipDeviceReset());
}
}
void enablePeer2Peer(int currentGpu, int peerGpu) {
int canAccessPeer;
int canAccessPeer;
// Must be on a multi-gpu system:
assert(currentGpu != peerGpu);
// Must be on a multi-gpu system:
assert(currentGpu != peerGpu);
HIPCHECK(hipSetDevice(currentGpu));
hipDeviceCanAccessPeer(&canAccessPeer, currentGpu, peerGpu);
HIPCHECK(hipSetDevice(currentGpu));
hipDeviceCanAccessPeer(&canAccessPeer, currentGpu, peerGpu);
if (canAccessPeer == 1) {
HIPCHECK(hipDeviceEnablePeerAccess(peerGpu, 0));
} else
printf("peer2peer transfer not possible between the selected gpu devices");
if (canAccessPeer == 1) {
HIPCHECK(hipDeviceEnablePeerAccess(peerGpu, 0));
} else
printf("peer2peer transfer not possible between the selected gpu devices");
}
void disablePeer2Peer(int currentGpu, int peerGpu) {
int canAccessPeer;
int canAccessPeer;
// Must be on a multi-gpu system:
assert(currentGpu != peerGpu);
// Must be on a multi-gpu system:
assert(currentGpu != peerGpu);
HIPCHECK(hipSetDevice(currentGpu));
hipDeviceCanAccessPeer(&canAccessPeer, currentGpu, peerGpu);
HIPCHECK(hipSetDevice(currentGpu));
hipDeviceCanAccessPeer(&canAccessPeer, currentGpu, peerGpu);
if (canAccessPeer == 1) {
HIPCHECK(hipDeviceDisablePeerAccess(peerGpu));
} else
printf("peer2peer disable not required");
if (canAccessPeer == 1) {
HIPCHECK(hipDeviceDisablePeerAccess(peerGpu));
} else
printf("peer2peer disable not required");
}
__global__ void matrixTranspose_static_shared(float* out, float* in,
const int width) {
__shared__ float sharedMem[WIDTH * WIDTH];
__global__ void matrixTranspose_static_shared(float* out, float* in, const int width) {
__shared__ float sharedMem[WIDTH * WIDTH];
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
sharedMem[y * width + x] = in[x * width + y];
sharedMem[y * width + x] = in[x * width + y];
__syncthreads();
__syncthreads();
out[y * width + x] = sharedMem[y * width + x];
out[y * width + x] = sharedMem[y * width + x];
}
__global__ void matrixTranspose_dynamic_shared(float* out, float* in,
const int width) {
extern __shared__ float sharedMem[];
__global__ void matrixTranspose_dynamic_shared(float* out, float* in, const int width) {
extern __shared__ float sharedMem[];
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
int x = blockDim.x * blockIdx.x + threadIdx.x;
int y = blockDim.y * blockIdx.y + threadIdx.y;
sharedMem[y * width + x] = in[x * width + y];
sharedMem[y * width + x] = in[x * width + y];
__syncthreads();
__syncthreads();
out[y * width + x] = sharedMem[y * width + x];
out[y * width + x] = sharedMem[y * width + x];
}
int main() {
checkPeer2PeerSupport();
checkPeer2PeerSupport();
int gpuCount;
int currentGpu, peerGpu;
int gpuCount;
int currentGpu, peerGpu;
HIPCHECK(hipGetDeviceCount(&gpuCount));
if (gpuCount < 2) {
printf("Peer2Peer application requires atleast 2 gpu devices");
return 0;
}
currentGpu = 0;
peerGpu = (currentGpu + 1);
printf("currentGpu=%d peerGpu=%d (Total no. of gpu = %d)\n", currentGpu, peerGpu, gpuCount);
float *data[2], *TransposeMatrix[2], *gpuTransposeMatrix[2], *randArray;
int width = WIDTH;
randArray = (float*)malloc(NUM * sizeof(float));
for (int i = 0; i < NUM; i++) {
randArray[i] = (float)i * 1.0f;
}
enablePeer2Peer(currentGpu, peerGpu);
HIPCHECK(hipSetDevice(currentGpu));
TransposeMatrix[0] = (float*)malloc(NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix[0], NUM * sizeof(float));
hipMalloc((void**)&data[0], NUM * sizeof(float));
hipMemcpy(data[0], randArray, NUM * sizeof(float), hipMemcpyHostToDevice);
hipLaunchKernelGGL(matrixTranspose_static_shared,
dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix[0],
data[0], width);
HIPCHECK(hipDeviceSynchronize());
HIPCHECK(hipSetDevice(peerGpu));
TransposeMatrix[1] = (float*)malloc(NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix[1], NUM * sizeof(float));
hipMalloc((void**)&data[1], NUM * sizeof(float));
hipMemcpy(data[1], gpuTransposeMatrix[0], NUM * sizeof(float), hipMemcpyDeviceToDevice);
hipLaunchKernelGGL(matrixTranspose_dynamic_shared,
dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), sizeof(float) * WIDTH * WIDTH,
0, gpuTransposeMatrix[1], data[1], width);
hipMemcpy(TransposeMatrix[1], gpuTransposeMatrix[1], NUM * sizeof(float),
hipMemcpyDeviceToHost);
hipDeviceSynchronize();
disablePeer2Peer(currentGpu, peerGpu);
// verify the results
int errors = 0;
double eps = 1.0E-6;
for (int i = 0; i < NUM; i++) {
if (std::abs(randArray[i] - TransposeMatrix[1][i]) > eps) {
printf("%d cpu: %f gpu peered data %f\n", i, randArray[i], TransposeMatrix[1][i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("Peer2Peer PASSED!\n");
}
free(randArray);
for (int i = 0; i < 2; i++) {
hipFree(data[i]);
hipFree(gpuTransposeMatrix[i]);
free(TransposeMatrix[i]);
}
HIPCHECK(hipSetDevice(peerGpu));
HIPCHECK(hipDeviceReset());
HIPCHECK(hipSetDevice(currentGpu));
HIPCHECK(hipDeviceReset());
HIPCHECK(hipGetDeviceCount(&gpuCount));
if (gpuCount < 2) {
printf("Peer2Peer application requires atleast 2 gpu devices");
return 0;
}
currentGpu = 0;
peerGpu = (currentGpu + 1);
printf("currentGpu=%d peerGpu=%d (Total no. of gpu = %d)\n", currentGpu, peerGpu, gpuCount);
float *data[2], *TransposeMatrix[2], *gpuTransposeMatrix[2], *randArray;
int width = WIDTH;
randArray = (float*)malloc(NUM * sizeof(float));
for (int i = 0; i < NUM; i++) {
randArray[i] = (float)i * 1.0f;
}
enablePeer2Peer(currentGpu, peerGpu);
HIPCHECK(hipSetDevice(currentGpu));
TransposeMatrix[0] = (float*)malloc(NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix[0], NUM * sizeof(float));
hipMalloc((void**)&data[0], NUM * sizeof(float));
hipMemcpy(data[0], randArray, NUM * sizeof(float), hipMemcpyHostToDevice);
hipLaunchKernelGGL(
matrixTranspose_static_shared, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0, gpuTransposeMatrix[0], data[0], width);
HIPCHECK(hipDeviceSynchronize());
HIPCHECK(hipSetDevice(peerGpu));
TransposeMatrix[1] = (float*)malloc(NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix[1], NUM * sizeof(float));
hipMalloc((void**)&data[1], NUM * sizeof(float));
hipMemcpy(data[1], gpuTransposeMatrix[0], NUM * sizeof(float), hipMemcpyDeviceToDevice);
hipLaunchKernelGGL(matrixTranspose_dynamic_shared,
dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), sizeof(float) * WIDTH * WIDTH,
0, gpuTransposeMatrix[1], data[1], width);
hipMemcpy(TransposeMatrix[1], gpuTransposeMatrix[1], NUM * sizeof(float), hipMemcpyDeviceToHost);
hipDeviceSynchronize();
disablePeer2Peer(currentGpu, peerGpu);
// verify the results
int errors = 0;
double eps = 1.0E-6;
for (int i = 0; i < NUM; i++) {
if (std::abs(randArray[i] - TransposeMatrix[1][i]) > eps) {
printf("%d cpu: %f gpu peered data %f\n", i, randArray[i], TransposeMatrix[1][i]);
errors++;
}
}
if (errors != 0) {
printf("FAILED: %d errors\n", errors);
} else {
printf("Peer2Peer PASSED!\n");
}
free(randArray);
for (int i = 0; i < 2; i++) {
hipFree(data[i]);
hipFree(gpuTransposeMatrix[i]);
free(TransposeMatrix[i]);
}
HIPCHECK(hipSetDevice(peerGpu));
HIPCHECK(hipDeviceReset());
HIPCHECK(hipSetDevice(currentGpu));
HIPCHECK(hipDeviceReset());
return 0;
}
+67 -66
Ver ficheiro
@@ -35,89 +35,90 @@ THE SOFTWARE.
// CPU function - basically scan each row and save the output in array
void matrixRowSum(int* input, int* output, int width) {
for (int i = 0; i < width; i++) {
for (int j = 0; j < width; j++) {
output[i] += input[i * width + j];
}
for (int i = 0; i < width; i++) {
for (int j = 0; j < width; j++) {
output[i] += input[i * width + j];
}
}
}
// Device (kernel) function
__global__ void gpuMatrixRowSum(int* input, int* output, int width) {
int index = blockDim.x * blockIdx.x + threadIdx.x;
int index = blockDim.x * blockIdx.x + threadIdx.x;
#pragma unroll
for (int i = 0; i < width; i++) {
output[index] += input[index * width + i];
}
for (int i = 0; i < width; i++) {
output[index] += input[index * width + i];
}
}
int main() {
int* Matrix;
int* sumMatrix;
int* cpuSumMatrix;
int* Matrix;
int* sumMatrix;
int* cpuSumMatrix;
int* gpuMatrix;
int* gpuSumMatrix;
int* gpuMatrix;
int* gpuSumMatrix;
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipDeviceProp_t devProp;
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
std::cout << "Device name " << devProp.name << std::endl;
Matrix = (int*)malloc(sizeof(int) * SIZE);
sumMatrix = (int*)malloc(sizeof(int) * LENGTH);
cpuSumMatrix = (int*)malloc(sizeof(int) * LENGTH);
Matrix = (int*)malloc(sizeof(int) * SIZE);
sumMatrix = (int*)malloc(sizeof(int) * LENGTH);
cpuSumMatrix = (int*)malloc(sizeof(int) * LENGTH);
for (int i = 0; i < SIZE; i++) {
Matrix[i] = i * 2;
for (int i = 0; i < SIZE; i++) {
Matrix[i] = i * 2;
}
for (int i = 0; i < LENGTH; i++) {
cpuSumMatrix[i] = 0;
}
// Allocated Device Memory
checkHipErrors(hipMalloc((void**)&gpuMatrix, SIZE * sizeof(int)));
checkHipErrors(hipMalloc((void**)&gpuSumMatrix, LENGTH * sizeof(int)));
// Memory Copy to Device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, SIZE * sizeof(int), hipMemcpyHostToDevice));
checkHipErrors(
hipMemcpy(gpuSumMatrix, cpuSumMatrix, LENGTH * sizeof(float), hipMemcpyHostToDevice));
// Launch device kernels
hipLaunchKernelGGL(gpuMatrixRowSum, dim3(BLOCKS_PER_GRID), dim3(THREADS_PER_BLOCK), 0, 0,
gpuMatrix, gpuSumMatrix, LENGTH);
// Memory copy back to device
checkHipErrors(hipMemcpy(sumMatrix, gpuSumMatrix, LENGTH * sizeof(int), hipMemcpyDeviceToHost));
// Cpu implementation
matrixRowSum(Matrix, cpuSumMatrix, LENGTH);
// verify the results
int errors = 0;
for (int i = 0; i < LENGTH; i++) {
if (sumMatrix[i] != cpuSumMatrix[i]) {
printf("%d - cpu: %d gpu: %d\n", i, sumMatrix[i], cpuSumMatrix[i]);
errors++;
}
}
for (int i = 0; i < LENGTH; i++) {
cpuSumMatrix[i] = 0;
}
if (errors == 0) {
printf("PASSED\n");
} else {
printf("FAILED with %d errors\n", errors);
}
// Allocated Device Memory
checkHipErrors(hipMalloc((void**)&gpuMatrix, SIZE * sizeof(int)));
checkHipErrors(hipMalloc((void**)&gpuSumMatrix, LENGTH * sizeof(int)));
// GPU Free
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuSumMatrix));
// Memory Copy to Device
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, SIZE * sizeof(int), hipMemcpyHostToDevice));
checkHipErrors(hipMemcpy(gpuSumMatrix, cpuSumMatrix, LENGTH * sizeof(float), hipMemcpyHostToDevice));
// CPU Free
free(Matrix);
free(sumMatrix);
free(cpuSumMatrix);
// Launch device kernels
hipLaunchKernelGGL(gpuMatrixRowSum, dim3(BLOCKS_PER_GRID), dim3(THREADS_PER_BLOCK), 0, 0,
gpuMatrix, gpuSumMatrix, LENGTH);
// Memory copy back to device
checkHipErrors(hipMemcpy(sumMatrix, gpuSumMatrix, LENGTH * sizeof(int), hipMemcpyDeviceToHost));
// Cpu implementation
matrixRowSum(Matrix, cpuSumMatrix, LENGTH);
// verify the results
int errors = 0;
for (int i = 0; i < LENGTH; i++) {
if (sumMatrix[i] != cpuSumMatrix[i]) {
printf("%d - cpu: %d gpu: %d\n", i, sumMatrix[i], cpuSumMatrix[i]);
errors++;
}
}
if (errors == 0) {
printf("PASSED\n");
} else {
printf("FAILED with %d errors\n", errors);
}
// GPU Free
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuSumMatrix));
// CPU Free
free(Matrix);
free(sumMatrix);
free(cpuSumMatrix);
return errors;
return errors;
}