SWDEV-402381 - Add hipCheckErrors for HIP API calls in samples (#375)

Change-Id: I335d7e780362fc59fd2d90939b4c8b8a7231ffc7
This commit is contained in:
ROCm CI Service Account
2023-07-20 10:22:17 +05:30
committed by GitHub
parent b8fb6f88b9
commit 7cc53f992f
71 changed files with 460 additions and 448 deletions
@@ -40,5 +40,7 @@ set(CMAKE_BUILD_TYPE Release)
# Create the excutable
add_executable(MatrixTranspose MatrixTranspose.cpp)
target_include_directories(MatrixTranspose PRIVATE ../../common)
# Link with HIP
target_link_libraries(MatrixTranspose hip::host)
@@ -30,6 +30,7 @@ HIPCC=$(HIP_PATH)/bin/hipcc
TARGET=hcc
INCLUDES := -I../../common
SOURCES = MatrixTranspose.cpp
OBJECTS = $(SOURCES:.cpp=.o)
@@ -40,7 +41,7 @@ EXECUTABLE=./MatrixTranspose
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
@@ -24,6 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 1024
@@ -61,7 +62,7 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
@@ -78,11 +79,11 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
@@ -90,7 +91,7 @@ int main() {
gpuMatrix, WIDTH);
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
@@ -110,8 +111,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
@@ -40,5 +40,7 @@ set(CMAKE_BUILD_TYPE Release)
# Create the excutable
add_executable(inline_asm inline_asm.cpp)
target_include_directories(inline_asm PRIVATE ../../common)
# Link with HIP
target_link_libraries(inline_asm hip::host)
+2 -2
View File
@@ -32,7 +32,7 @@ TARGET=hcc
SOURCES = inline_asm.cpp
OBJECTS = $(SOURCES:.cpp=.o)
INCLUDES := -I../../common
EXECUTABLE=./inline_asm
.PHONY: test
@@ -40,7 +40,7 @@ EXECUTABLE=./inline_asm
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
+22 -21
View File
@@ -24,6 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 1024
@@ -59,13 +60,13 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
hipEvent_t start, stop;
hipEventCreate(&start);
hipEventCreate(&stop);
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
int i;
@@ -81,25 +82,25 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Record the start event
hipEventRecord(start, NULL);
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Record the stop event
hipEventRecord(stop, NULL);
hipEventSynchronize(stop);
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
hipEventElapsedTime(&eventMs, start, stop);
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyHostToDevice time taken = %6.3fms\n", eventMs);
// Record the start event
hipEventRecord(start, NULL);
checkHipErrors(hipEventRecord(start, NULL));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
@@ -107,24 +108,24 @@ int main() {
gpuMatrix, WIDTH);
// Record the stop event
hipEventRecord(stop, NULL);
hipEventSynchronize(stop);
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
hipEventElapsedTime(&eventMs, start, stop);
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
// Record the start event
hipEventRecord(start, NULL);
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// Record the stop event
hipEventRecord(stop, NULL);
hipEventSynchronize(stop);
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
hipEventElapsedTime(&eventMs, start, stop);
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyDeviceToHost time taken = %6.3fms\n", eventMs);
@@ -147,8 +148,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
@@ -50,5 +50,7 @@ add_custom_target(
add_dependencies(texture2dDrv codeobj)
target_include_directories(texture2dDrv PRIVATE ../../common)
# Link with HIP
target_link_libraries(texture2dDrv hip::host)
@@ -27,11 +27,12 @@ ifeq (,$(HIP_PATH))
endif
HIPCC=$(HIP_PATH)/bin/hipcc
HIP_PLATFORM=$(shell $(HIP_PATH)/bin/hipconfig --compiler)
INCLUDES := -I../../common
all: tex2dKernel.code texture2dDrv.out
texture2dDrv.out: texture2dDrv.cpp
$(HIPCC) $(HIPCC_FLAGS) $< -o $@
$(HIPCC) $(HIPCC_FLAGS) $(INCLUDES) $< -o $@
tex2dKernel.code: tex2dKernel.cpp
$(HIPCC) --genco $(GENCO_FLAGS) $^ -o $@
@@ -24,21 +24,12 @@ THE SOFTWARE.
#include <iostream>
#include <fstream>
#include <vector>
#include "hip_helper.h"
#define fileName "tex2dKernel.code"
bool testResult = true;
#define HIP_CHECK(cmd) \
{ \
hipError_t status = cmd; \
if (status != hipSuccess) { \
std::cout << "error: #" << status << " (" << hipGetErrorString(status) \
<< ") at line:" << __LINE__ << ": " << #cmd << std::endl; \
abort(); \
} \
}
template<typename T,
typename std::enable_if<std::is_arithmetic<T>::value>::type *t = nullptr>
static inline hipArray_Format getArrayFormat() {
@@ -154,11 +145,11 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
hipChannelFormatDesc channelDesc = hipCreateChannelDesc<T>();
hipArray_t array;
HIP_CHECK(hipMallocArray(&array, &channelDesc, width, height));
checkHipErrors(hipMallocArray(&array, &channelDesc, width, height));
const size_t spitch = width * sizeof(T);
HIP_CHECK(hipMemcpy2DToArray(array, 0, 0, hData, spitch, width * sizeof(T),
checkHipErrors(hipMemcpy2DToArray(array, 0, 0, hData, spitch, width * sizeof(T),
height, hipMemcpyHostToDevice));
hipResourceDesc resDesc;
@@ -175,10 +166,10 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
texDesc.normalizedCoords = 0;
hipTextureObject_t texObj;
HIP_CHECK(hipCreateTextureObject(&texObj, &resDesc, &texDesc, nullptr));
checkHipErrors(hipCreateTextureObject(&texObj, &resDesc, &texDesc, nullptr));
T *dData = NULL;
HIP_CHECK(hipMalloc((void** )&dData, size));
checkHipErrors(hipMalloc((void** )&dData, size));
struct {
void *_Ad;
@@ -197,18 +188,18 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
HIP_LAUNCH_PARAM_BUFFER_SIZE, &sizeTemp, HIP_LAUNCH_PARAM_END };
hipFunction_t Function;
HIP_CHECK(hipModuleGetFunction(&Function, module, funcName));
checkHipErrors(hipModuleGetFunction(&Function, module, funcName));
int temp1 = width / 16;
int temp2 = height / 16;
HIP_CHECK(
checkHipErrors(
hipModuleLaunchKernel(Function, 16, 16, 1, temp1, temp2, 1, 0, 0, NULL,
(void** )&config));
HIP_CHECK(hipDeviceSynchronize());
checkHipErrors(hipDeviceSynchronize());
T *hOutputData = (T*) malloc(size);
memset(hOutputData, 0, size);
HIP_CHECK(hipMemcpy(hOutputData, dData, size, hipMemcpyDeviceToHost));
checkHipErrors(hipMemcpy(hOutputData, dData, size, hipMemcpyDeviceToHost));
for (int i = 0; i < height; i++) {
for (int j = 0; j < width; j++) {
@@ -219,9 +210,9 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
}
}
}
HIP_CHECK(hipDestroyTextureObject(texObj));
HIP_CHECK(hipFree(dData));
HIP_CHECK(hipFreeArray(array));
checkHipErrors(hipDestroyTextureObject(texObj));
checkHipErrors(hipFree(dData));
checkHipErrors(hipFreeArray(array));
free(hOutputData);
free(hData);
printf("%s test %s ...\n", funcName, testResult ? "PASSED" : "FAILED");
@@ -231,7 +222,7 @@ bool runTest(hipModule_t &module, const char *refName, const char *funcName) {
inline bool isImageSupported() {
int imageSupport = 1;
#ifdef __HIP_PLATFORM_AMD__
HIP_CHECK(hipDeviceGetAttribute(&imageSupport, hipDeviceAttributeImageSupport,
checkHipErrors(hipDeviceGetAttribute(&imageSupport, hipDeviceAttributeImageSupport,
0));
#endif
return imageSupport != 0;
@@ -242,10 +233,10 @@ int main(int argc, char** argv) {
printf("Texture is not support on the device. Skipped.\n");
return 0;
}
HIP_CHECK(hipInit(0));
HIP_CHECK(hipSetDevice(0));
checkHipErrors(hipInit(0));
checkHipErrors(hipSetDevice(0));
hipModule_t module;
HIP_CHECK(hipModuleLoad(&module, fileName));
checkHipErrors(hipModuleLoad(&module, fileName));
testResult = testResult && runTest<char>(module, "texChar", "tex2dKernelChar");
testResult = testResult && runTest<short>(module, "texShort", "tex2dKernelShort");
testResult = testResult && runTest<int>(module, "texInt", "tex2dKernelInt");
@@ -255,7 +246,7 @@ int main(int argc, char** argv) {
testResult = testResult && runTest<int4>(module, "texInt4", "tex2dKernelInt4");
testResult = testResult && runTest<float4>(module, "texFloat4", "tex2dKernelFloat4");
HIP_CHECK(hipModuleUnload(module));
checkHipErrors(hipModuleUnload(module));
printf("texture2dDrv %s ...\n", testResult ? "PASSED" : "FAILED");
return testResult ? EXIT_SUCCESS : EXIT_FAILURE;
}
@@ -51,6 +51,8 @@ set(MY_NVCC_OPTIONS)
set_source_files_properties(${MY_SOURCE_FILES} PROPERTIES HIP_SOURCE_PROPERTY_FORMAT 1)
hip_add_executable(${MY_TARGET_NAME} ${MY_SOURCE_FILES} HIPCC_OPTIONS ${MY_HIPCC_OPTIONS} CLANG_OPTIONS ${MY_CLANG_OPTIONS} NVCC_OPTIONS ${MY_NVCC_OPTIONS})
target_include_directories(${MY_TARGET_NAME} PRIVATE ../../common)
# Search for rocm in common locations
list(APPEND CMAKE_PREFIX_PATH ${ROCM_PATH}/hip ${ROCM_PATH})
find_package(hip QUIET CONFIG)
@@ -24,6 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 1024
@@ -61,7 +62,7 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
@@ -78,11 +79,11 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
@@ -90,7 +91,7 @@ int main() {
gpuMatrix, WIDTH);
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
@@ -110,8 +111,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
@@ -40,5 +40,7 @@ set(CMAKE_BUILD_TYPE Release)
# Create the excutable
add_executable(occupancy occupancy.cpp)
target_include_directories(occupancy PRIVATE ../../common)
# Link with HIP
target_link_libraries(occupancy hip::host)
+2 -2
View File
@@ -26,7 +26,7 @@ ifeq (,$(HIP_PATH))
HIP_PATH=../../..
endif
HIPCC=$(HIP_PATH)/bin/hipcc
INCLUDES := -I../../common
EXE=./occupancy
.PHONY: test
@@ -34,7 +34,7 @@ EXE=./occupancy
all: test
$(EXE): occupancy.cpp
$(HIPCC) $^ -o $@
$(HIPCC) $(INCLUDES) $^ -o $@
test: $(EXE)
$(EXE)
+22 -27
View File
@@ -19,14 +19,9 @@ THE SOFTWARE.
#include "hip/hip_runtime.h"
#include <iostream>
#include "hip_helper.h"
#define NUM 1000000
#define HIP_CHECK(status) \
if (status != hipSuccess) { \
std::cout << "Got Status: " << status << " at Line: " << __LINE__ << std::endl; \
exit(0); \
}
// Device (Kernel) function
__global__ void multiply(float* C, float* A, float* B, int N){
@@ -47,11 +42,11 @@ void multiplyCPU(float* C, float* A, float* B, int N){
void launchKernel(float* C, float* A, float* B, bool manual){
hipDeviceProp_t devProp;
HIP_CHECK(hipGetDeviceProperties(&devProp, 0));
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
hipEvent_t start, stop;
HIP_CHECK(hipEventCreate(&start));
HIP_CHECK(hipEventCreate(&stop));
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
const unsigned threadsperblock = 32;
const unsigned blocks = (NUM/threadsperblock)+1;
@@ -66,28 +61,28 @@ void launchKernel(float* C, float* A, float* B, bool manual){
std::cout << std::endl << "Manual Configuration with block size " << blockSize << std::endl;
}
else{
HIP_CHECK(hipOccupancyMaxPotentialBlockSize(&mingridSize, &blockSize, multiply, 0, 0));
checkHipErrors(hipOccupancyMaxPotentialBlockSize(&mingridSize, &blockSize, multiply, 0, 0));
std::cout << std::endl << "Automatic Configuation based on hipOccupancyMaxPotentialBlockSize " << std::endl;
std::cout << "Suggested blocksize is " << blockSize << ", Minimum gridsize is " << mingridSize << std::endl;
gridSize = (NUM/blockSize)+1;
}
// Record the start event
HIP_CHECK(hipEventRecord(start, NULL));
checkHipErrors(hipEventRecord(start, NULL));
// Launching the Kernel from Host
hipLaunchKernelGGL(multiply, dim3(gridSize), dim3(blockSize), 0, 0, C, A, B, NUM);
// Record the stop event
HIP_CHECK(hipEventRecord(stop, NULL));
HIP_CHECK(hipEventSynchronize(stop));
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
HIP_CHECK(hipEventElapsedTime(&eventMs, start, stop));
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
//Calculate Occupancy
int numBlock = 0;
HIP_CHECK(hipOccupancyMaxActiveBlocksPerMultiprocessor(&numBlock, multiply, blockSize, 0));
checkHipErrors(hipOccupancyMaxActiveBlocksPerMultiprocessor(&numBlock, multiply, blockSize, 0));
if(devProp.maxThreadsPerMultiProcessor){
std::cout << "Theoretical Occupancy is " << (double)numBlock* blockSize/devProp.maxThreadsPerMultiProcessor * 100 << "%" << std::endl;
@@ -113,14 +108,14 @@ int main() {
}
// allocate the memory on the device side
HIP_CHECK(hipMalloc((void**)&Ad, NUM * sizeof(float)));
HIP_CHECK(hipMalloc((void**)&Bd, NUM * sizeof(float)));
HIP_CHECK(hipMalloc((void**)&C0d, NUM * sizeof(float)));
HIP_CHECK(hipMalloc((void**)&C1d, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&Ad, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&Bd, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&C0d, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&C1d, NUM * sizeof(float)));
// Memory transfer from host to device
HIP_CHECK(hipMemcpy(Ad,A,NUM * sizeof(float), hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(Bd,B,NUM * sizeof(float), hipMemcpyHostToDevice));
checkHipErrors(hipMemcpy(Ad,A,NUM * sizeof(float), hipMemcpyHostToDevice));
checkHipErrors(hipMemcpy(Bd,B,NUM * sizeof(float), hipMemcpyHostToDevice));
//Kernel launch with manual/default block size
launchKernel(C0d, Ad, Bd, 1);
@@ -129,8 +124,8 @@ int main() {
launchKernel(C1d, Ad, Bd, 0);
// Memory transfer from device to host
HIP_CHECK(hipMemcpy(C0,C0d, NUM * sizeof(float), hipMemcpyDeviceToHost));
HIP_CHECK(hipMemcpy(C1,C1d, NUM * sizeof(float), hipMemcpyDeviceToHost));
checkHipErrors(hipMemcpy(C0,C0d, NUM * sizeof(float), hipMemcpyDeviceToHost));
checkHipErrors(hipMemcpy(C1,C1d, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU computation
multiplyCPU(cpuC, A, B, NUM);
@@ -163,10 +158,10 @@ int main() {
printf("\nAutomatic Test PASSED!\n");
}
HIP_CHECK(hipFree(Ad));
HIP_CHECK(hipFree(Bd));
HIP_CHECK(hipFree(C0d));
HIP_CHECK(hipFree(C1d));
checkHipErrors(hipFree(Ad));
checkHipErrors(hipFree(Bd));
checkHipErrors(hipFree(C0d));
checkHipErrors(hipFree(C1d));
free(A);
free(B);
@@ -40,5 +40,7 @@ set(CMAKE_BUILD_TYPE Release)
# Create the excutable
add_executable(gpuarch gpuarch.cpp)
target_include_directories(gpuarch PRIVATE ../../common)
# Link with HIP
target_link_libraries(gpuarch hip::host)
+2 -2
View File
@@ -26,7 +26,7 @@ ifeq (,$(HIP_PATH))
HIP_PATH=../../..
endif
HIPCC=$(HIP_PATH)/bin/hipcc
INCLUDES := -I../../common
EXE=./gpuarch
.PHONY: test
@@ -34,7 +34,7 @@ EXE=./gpuarch
all: test
$(EXE): gpuarch.cpp
$(HIPCC) $^ -o $@
$(HIPCC) $(INCLUDES) $^ -o $@
test: $(EXE)
$(EXE)
+4 -10
View File
@@ -25,12 +25,6 @@ THE SOFTWARE.
#define SIZE (BLOCKS_PER_GRID * THREADS_PER_BLOCK)
#define NOT_SUPPORTED -99 // dummy number indicates unsupported operation
#define HIP_STATUS_CHECK(status) \
if (status != hipSuccess) { \
std::cout << "Got Status: " << status << " at Line: " << __LINE__ << std::endl; \
exit(0); \
}
// Using __gfx*__ macro one can have GPU architecture specific code flow
// For example: If below kernel runs on gfx908 it will increment 'in' by 'value' and store into
// 'out'
@@ -57,8 +51,8 @@ int main() {
int32_t* hInput = static_cast<int32_t*>(malloc(NBytes));
int32_t* hOutput = static_cast<int32_t*>(malloc(NBytes));
HIP_STATUS_CHECK(hipMalloc(&dInput, NBytes));
HIP_STATUS_CHECK(hipMalloc(&dOutput, NBytes));
checkHipErrors(hipMalloc(&dInput, NBytes));
checkHipErrors(hipMalloc(&dOutput, NBytes));
// Initialize host input/output buffers
for (int i = 0; i < SIZE; ++i) {
@@ -67,14 +61,14 @@ int main() {
}
// Initialize device input buffer
HIP_STATUS_CHECK(hipMemcpy(dInput, hInput, NBytes, hipMemcpyHostToDevice));
checkHipErrors(hipMemcpy(dInput, hInput, NBytes, hipMemcpyHostToDevice));
// Launch kernel
hipLaunchKernelGGL(incrementKernel, dim3(BLOCKS_PER_GRID), dim3(THREADS_PER_BLOCK), 0, 0, dInput,
dOutput, incrementValue, SIZE);
// Copy result back to host buffer
HIP_STATUS_CHECK(hipMemcpy(hOutput, dOutput, NBytes, hipMemcpyDeviceToHost));
checkHipErrors(hipMemcpy(hOutput, dOutput, NBytes, hipMemcpyDeviceToHost));
bool flag = true;
// verify data
@@ -30,6 +30,7 @@ HIPCC=$(HIP_PATH)/bin/hipcc
CLANG=$(HIP_PATH)/llvm/bin/clang
LLVM_MC=$(HIP_PATH)/llvm/bin/llvm-mc
CLANG_OFFLOAD_BUNDLER=$(HIP_PATH)/llvm/bin/clang-offload-bundler
INCLUDES := -I../../common
SRCS=square.cpp
@@ -57,8 +58,8 @@ GPU_ARCH9=gfx1103
all: src_to_asm asm_to_exec
src_to_asm:
$(HIPCC) -c -S --cuda-host-only -target x86_64-linux-gnu -o $(SQ_HOST_ASM) $(SRCS)
$(HIPCC) -c -S --cuda-device-only --offload-arch=$(GPU_ARCH1) --offload-arch=$(GPU_ARCH2) --offload-arch=$(GPU_ARCH3) --offload-arch=$(GPU_ARCH4) --offload-arch=$(GPU_ARCH5) --offload-arch=$(GPU_ARCH6) --offload-arch=$(GPU_ARCH7) --offload-arch=$(GPU_ARCH8) --offload-arch=$(GPU_ARCH9) $(SRCS)
$(HIPCC) -c -S $(INCLUDES) --cuda-host-only -target x86_64-linux-gnu -o $(SQ_HOST_ASM) $(SRCS)
$(HIPCC) -c -S $(INCLUDES) --cuda-device-only --offload-arch=$(GPU_ARCH1) --offload-arch=$(GPU_ARCH2) --offload-arch=$(GPU_ARCH3) --offload-arch=$(GPU_ARCH4) --offload-arch=$(GPU_ARCH5) --offload-arch=$(GPU_ARCH6) --offload-arch=$(GPU_ARCH7) --offload-arch=$(GPU_ARCH8) --offload-arch=$(GPU_ARCH9) $(SRCS)
# You may modify the .s assembly files before the next step
# By default, their names will be:
@@ -19,15 +19,7 @@ THE SOFTWARE.
#include <stdio.h>
#include <hip/hip_runtime.h>
#define CHECK(cmd) \
{\
hipError_t error = cmd;\
if (error != hipSuccess) { \
fprintf(stderr, "error: '%s'(%d) at %s:%d\n", hipGetErrorString(error), error,__FILE__, __LINE__); \
exit(EXIT_FAILURE);\
}\
}
#include "hip_helper.h"
/* This kernel is a placeholder for the kernel in assembly generated by this
* sample. It will be replaced by the kernel in assembly.
@@ -55,14 +47,14 @@ int main(int argc, char *argv[])
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
CHECK(hipGetDeviceProperties(&props, 0/*deviceID*/));
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
CHECK(A_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
checkHipErrors(A_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
C_h = (float*)malloc(Nbytes);
CHECK(C_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
checkHipErrors(C_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
@@ -70,12 +62,12 @@ int main(int argc, char *argv[])
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
CHECK(hipMalloc(&A_d, Nbytes));
CHECK(hipMalloc(&C_d, Nbytes));
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf ("info: copy Host2Device\n");
CHECK ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
@@ -84,12 +76,12 @@ int main(int argc, char *argv[])
vector_square <<<blocks, threadsPerBlock>>> (C_d, A_d, N);
printf ("info: copy Device2Host\n");
CHECK ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: check result\n");
printf ("info: checkHipErrors result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
CHECK(hipErrorUnknown);
checkHipErrors(hipErrorUnknown);
}
}
printf ("PASSED!\n");
@@ -32,6 +32,7 @@ LLVM_MC=$(HIP_PATH)/llvm/bin/llvm-mc
CLANG_OFFLOAD_BUNDLER=$(HIP_PATH)/llvm/bin/clang-offload-bundler
LLVM_AS=$(HIP_PATH)/llvm/bin/llvm-as
LLVM_DIS=$(HIP_PATH)/llvm/bin/llvm-dis
INCLUDES := -I../../common
SRCS=square.cpp
@@ -60,8 +61,8 @@ GPU_ARCH9=gfx1103
all: src_to_ir bc_to_ll ll_to_bc ir_to_exec
src_to_ir:
$(HIPCC) -c -emit-llvm --cuda-host-only -target x86_64-linux-gnu -o $(SQ_HOST_BC) $(SRCS)
$(HIPCC) -c -emit-llvm --cuda-device-only --offload-arch=$(GPU_ARCH1) --offload-arch=$(GPU_ARCH2) --offload-arch=$(GPU_ARCH3) --offload-arch=$(GPU_ARCH4) --offload-arch=$(GPU_ARCH5) --offload-arch=$(GPU_ARCH6) --offload-arch=$(GPU_ARCH7) --offload-arch=$(GPU_ARCH8) --offload-arch=$(GPU_ARCH9) $(SRCS)
$(HIPCC) $(INCLUDES) -c -emit-llvm --cuda-host-only -target x86_64-linux-gnu -o $(SQ_HOST_BC) $(SRCS)
$(HIPCC) $(INCLUDES) -c -emit-llvm --cuda-device-only --offload-arch=$(GPU_ARCH1) --offload-arch=$(GPU_ARCH2) --offload-arch=$(GPU_ARCH3) --offload-arch=$(GPU_ARCH4) --offload-arch=$(GPU_ARCH5) --offload-arch=$(GPU_ARCH6) --offload-arch=$(GPU_ARCH7) --offload-arch=$(GPU_ARCH8) --offload-arch=$(GPU_ARCH9) $(SRCS)
# By default, the LLVM IR Bitcode file names will be:
# square-hip-amdgcn-amd-amdhsa-gfx900.bc
@@ -19,15 +19,7 @@ THE SOFTWARE.
#include <stdio.h>
#include <hip/hip_runtime.h>
#define CHECK(cmd) \
{\
hipError_t error = cmd;\
if (error != hipSuccess) { \
fprintf(stderr, "error: '%s'(%d) at %s:%d\n", hipGetErrorString(error), error,__FILE__, __LINE__); \
exit(EXIT_FAILURE);\
}\
}
#include "hip_helper.h"
/* This kernel is a placeholder for the kernel in LLVM IR generated by this
* sample. It will be replaced by the kernel in LLVM IR.
@@ -55,14 +47,14 @@ int main(int argc, char *argv[])
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
CHECK(hipGetDeviceProperties(&props, 0/*deviceID*/));
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
CHECK(A_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
checkHipErrors(A_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
C_h = (float*)malloc(Nbytes);
CHECK(C_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
checkHipErrors(C_h == 0 ? hipErrorMemoryAllocation : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
@@ -70,12 +62,12 @@ int main(int argc, char *argv[])
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
CHECK(hipMalloc(&A_d, Nbytes));
CHECK(hipMalloc(&C_d, Nbytes));
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf ("info: copy Host2Device\n");
CHECK ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
@@ -84,12 +76,12 @@ int main(int argc, char *argv[])
vector_square <<<blocks, threadsPerBlock>>> (C_d, A_d, N);
printf ("info: copy Device2Host\n");
CHECK ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: check result\n");
printf ("info: checkHipErrors result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
CHECK(hipErrorUnknown);
checkHipErrors(hipErrorUnknown);
}
}
printf ("PASSED!\n");
@@ -22,11 +22,15 @@ project(cmake_hip_device_test)
cmake_minimum_required(VERSION 3.10.2)
include_directories(../../common)
# Find hip
find_package(hip REQUIRED)
# Create the excutable
add_executable(test_cpp square.cpp)
target_include_directories(test_cpp PRIVATE ../../common)
# Link with HIP
target_link_libraries(test_cpp hip::device)
@@ -22,16 +22,7 @@ THE SOFTWARE.
#include <stdio.h>
#include <hip/hip_runtime.h>
#define CHECK(cmd) \
{\
hipError_t error = cmd;\
if (error != hipSuccess) { \
fprintf(stderr, "error: '%s'(%d) at %s:%d\n", hipGetErrorString(error), error,__FILE__, __LINE__); \
exit(EXIT_FAILURE);\
}\
}
#include "hip_helper.h"
/*
* Square each element in the array A and write to array C.
@@ -57,14 +48,14 @@ int main(int argc, char *argv[])
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
CHECK(hipGetDeviceProperties(&props, 0/*deviceID*/));
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
CHECK(A_h == 0 ? hipErrorOutOfMemory : hipSuccess );
checkHipErrors(A_h == 0 ? hipErrorOutOfMemory : hipSuccess );
C_h = (float*)malloc(Nbytes);
CHECK(C_h == 0 ? hipErrorOutOfMemory : hipSuccess );
checkHipErrors(C_h == 0 ? hipErrorOutOfMemory : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
@@ -72,12 +63,12 @@ int main(int argc, char *argv[])
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
CHECK(hipMalloc(&A_d, Nbytes));
CHECK(hipMalloc(&C_d, Nbytes));
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf ("info: copy Host2Device\n");
CHECK ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
@@ -86,12 +77,12 @@ int main(int argc, char *argv[])
hipLaunchKernelGGL(vector_square, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d, N);
printf ("info: copy Device2Host\n");
CHECK ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: check result\n");
printf ("info: checkHipErrors result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
CHECK(hipErrorUnknown);
checkHipErrors(hipErrorUnknown);
}
}
printf ("PASSED!\n");
@@ -10,5 +10,8 @@ add_executable(test_fortran TestFortran.F90)
add_executable(test_cpp MatrixTranspose.cpp)
target_link_libraries(test_cpp PUBLIC hip::device)
target_include_directories(test_cpp PRIVATE ../../common)
# Assuming to build a C/C++-to-Fortran library binding.
target_link_libraries(test_fortran PUBLIC hip::device)
@@ -24,7 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 1024
@@ -61,7 +61,7 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
@@ -78,11 +78,11 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
@@ -90,7 +90,7 @@ int main() {
gpuMatrix, WIDTH);
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
@@ -110,8 +110,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
@@ -40,5 +40,7 @@ set(CMAKE_BUILD_TYPE Release)
# Create the excutable
add_executable(hipEvent hipEvent.cpp)
target_include_directories(hipEvent PRIVATE ../../common)
# Link with HIP
target_link_libraries(hipEvent hip::host)
+2 -1
View File
@@ -30,6 +30,7 @@ TARGET=hcc
SOURCES = hipEvent.cpp
OBJECTS = $(SOURCES:.cpp=.o)
INCLUDES := -I../../common
EXECUTABLE=./hipEvent
@@ -38,7 +39,7 @@ EXECUTABLE=./hipEvent
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
+22 -21
View File
@@ -24,6 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 1024
@@ -59,13 +60,13 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
hipEvent_t start, stop;
hipEventCreate(&start);
hipEventCreate(&stop);
checkHipErrors(hipEventCreate(&start));
checkHipErrors(hipEventCreate(&stop));
float eventMs = 1.0f;
int i;
@@ -81,25 +82,25 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Record the start event
hipEventRecord(start, NULL);
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Record the stop event
hipEventRecord(stop, NULL);
hipEventSynchronize(stop);
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
hipEventElapsedTime(&eventMs, start, stop);
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyHostToDevice time taken = %6.3fms\n", eventMs);
// Record the start event
hipEventRecord(start, NULL);
checkHipErrors(hipEventRecord(start, NULL));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
@@ -107,24 +108,24 @@ int main() {
gpuMatrix, WIDTH);
// Record the stop event
hipEventRecord(stop, NULL);
hipEventSynchronize(stop);
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
hipEventElapsedTime(&eventMs, start, stop);
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("kernel Execution time = %6.3fms\n", eventMs);
// Record the start event
hipEventRecord(start, NULL);
checkHipErrors(hipEventRecord(start, NULL));
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// Record the stop event
hipEventRecord(stop, NULL);
hipEventSynchronize(stop);
checkHipErrors(hipEventRecord(stop, NULL));
checkHipErrors(hipEventSynchronize(stop));
hipEventElapsedTime(&eventMs, start, stop);
checkHipErrors(hipEventElapsedTime(&eventMs, start, stop));
printf("hipMemcpyDeviceToHost time taken = %6.3fms\n", eventMs);
@@ -146,8 +147,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
@@ -30,5 +30,7 @@ find_package(hip REQUIRED)
# Create the excutable
add_executable(square square.cpp)
target_include_directories(square PRIVATE ../../common)
# Link with HIP
target_link_libraries(square hip::device)
@@ -22,16 +22,7 @@ THE SOFTWARE.
#include <stdio.h>
#include <hip/hip_runtime.h>
#define CHECK(cmd) \
{\
hipError_t error = cmd;\
if (error != hipSuccess) { \
fprintf(stderr, "error: '%s'(%d) at %s:%d\n", hipGetErrorString(error), error,__FILE__, __LINE__); \
exit(EXIT_FAILURE);\
}\
}
#include "hip_helper.h"
/*
* Square each element in the array A and write to array C.
@@ -57,14 +48,14 @@ int main(int argc, char *argv[])
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
CHECK(hipGetDeviceProperties(&props, 0/*deviceID*/));
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
CHECK(A_h == 0 ? hipErrorOutOfMemory : hipSuccess );
checkHipErrors(A_h == 0 ? hipErrorOutOfMemory : hipSuccess );
C_h = (float*)malloc(Nbytes);
CHECK(C_h == 0 ? hipErrorOutOfMemory : hipSuccess );
checkHipErrors(C_h == 0 ? hipErrorOutOfMemory : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
@@ -72,12 +63,12 @@ int main(int argc, char *argv[])
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
CHECK(hipMalloc(&A_d, Nbytes));
CHECK(hipMalloc(&C_d, Nbytes));
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf ("info: copy Host2Device\n");
CHECK ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
@@ -86,12 +77,12 @@ int main(int argc, char *argv[])
hipLaunchKernelGGL(vector_square, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d, N);
printf ("info: copy Device2Host\n");
CHECK ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: check result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
CHECK(hipErrorUnknown);
checkHipErrors(hipErrorUnknown);
}
}
printf ("PASSED!\n");
@@ -25,3 +25,5 @@ project(cmake_hip_lang_support VERSION 1.0
LANGUAGES HIP)
# Create the executable
add_executable(square square.hip)
target_include_directories(square PRIVATE ../../common)
@@ -22,16 +22,7 @@ THE SOFTWARE.
#include <stdio.h>
#include <hip/hip_runtime.h>
#define CHECK(cmd) \
{\
hipError_t error = cmd;\
if (error != hipSuccess) { \
fprintf(stderr, "error: '%s'(%d) at %s:%d\n", hipGetErrorString(error), error,__FILE__, __LINE__); \
exit(EXIT_FAILURE);\
}\
}
#include "hip_helper.h"
/*
* Square each element in the array A and write to array C.
@@ -57,14 +48,14 @@ int main(int argc, char *argv[])
size_t Nbytes = N * sizeof(float);
hipDeviceProp_t props;
CHECK(hipGetDeviceProperties(&props, 0/*deviceID*/));
checkHipErrors(hipGetDeviceProperties(&props, 0/*deviceID*/));
printf ("info: running on device %s\n", props.name);
printf ("info: allocate host mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
A_h = (float*)malloc(Nbytes);
CHECK(A_h == 0 ? hipErrorOutOfMemory : hipSuccess );
checkHipErrors(A_h == 0 ? hipErrorOutOfMemory : hipSuccess );
C_h = (float*)malloc(Nbytes);
CHECK(C_h == 0 ? hipErrorOutOfMemory : hipSuccess );
checkHipErrors(C_h == 0 ? hipErrorOutOfMemory : hipSuccess );
// Fill with Phi + i
for (size_t i=0; i<N; i++)
{
@@ -72,12 +63,12 @@ int main(int argc, char *argv[])
}
printf ("info: allocate device mem (%6.2f MB)\n", 2*Nbytes/1024.0/1024.0);
CHECK(hipMalloc(&A_d, Nbytes));
CHECK(hipMalloc(&C_d, Nbytes));
checkHipErrors(hipMalloc(&A_d, Nbytes));
checkHipErrors(hipMalloc(&C_d, Nbytes));
printf ("info: copy Host2Device\n");
CHECK ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
checkHipErrors ( hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
const unsigned blocks = 512;
const unsigned threadsPerBlock = 256;
@@ -86,12 +77,12 @@ int main(int argc, char *argv[])
hipLaunchKernelGGL(vector_square, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d, N);
printf ("info: copy Device2Host\n");
CHECK ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
checkHipErrors ( hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
printf ("info: check result\n");
for (size_t i=0; i<N; i++) {
if (C_h[i] != A_h[i] * A_h[i]) {
CHECK(hipErrorUnknown);
checkHipErrors(hipErrorUnknown);
}
}
printf ("PASSED!\n");
@@ -34,3 +34,5 @@ add_executable(test saxpy.cpp)
target_link_libraries(test hiprtc::hiprtc)
# Link with HIP
target_link_libraries(test hip::device)
target_include_directories(test PRIVATE ../../common)
+16 -15
View File
@@ -22,6 +22,7 @@ THE SOFTWARE.
#include <hip/hiprtc.h>
#include <hip/hip_runtime.h>
#include <hip_helper.h>
#include <cassert>
#include <cstddef>
@@ -69,7 +70,7 @@ int main()
hipDeviceProp_t props;
int device = 0;
hipGetDeviceProperties(&props, device);
checkHipErrors(hipGetDeviceProperties(&props, device));
const char* options[] = {};
@@ -100,8 +101,8 @@ int main()
hipModule_t module;
hipFunction_t kernel;
hipModuleLoadData(&module, code.data());
hipModuleGetFunction(&kernel, module, "saxpy");
checkHipErrors(hipModuleLoadData(&module, code.data()));
checkHipErrors(hipModuleGetFunction(&kernel, module, "saxpy"));
size_t n = NUM_THREADS * NUM_BLOCKS;
size_t bufferSize = n * sizeof(float);
@@ -117,11 +118,11 @@ int main()
}
hipDeviceptr_t dX, dY, dOut;
hipMalloc((void **)&dX, bufferSize);
hipMalloc((void **)&dY, bufferSize);
hipMalloc((void **)&dOut, bufferSize);
hipMemcpyHtoD(dX, hX.get(), bufferSize);
hipMemcpyHtoD(dY, hY.get(), bufferSize);
checkHipErrors(hipMalloc((void **)&dX, bufferSize));
checkHipErrors(hipMalloc((void **)&dY, bufferSize));
checkHipErrors(hipMalloc((void **)&dOut, bufferSize));
checkHipErrors(hipMemcpyHtoD(dX, hX.get(), bufferSize));
checkHipErrors(hipMemcpyHtoD(dY, hY.get(), bufferSize));
struct {
float a_;
@@ -136,9 +137,9 @@ int main()
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
hipModuleLaunchKernel(kernel, NUM_BLOCKS, 1, 1, NUM_THREADS, 1, 1,
0, nullptr, nullptr, config);
hipMemcpyDtoH(hOut.get(), dOut, bufferSize);
checkHipErrors(hipModuleLaunchKernel(kernel, NUM_BLOCKS, 1, 1, NUM_THREADS, 1, 1,
0, nullptr, nullptr, config));
checkHipErrors(hipMemcpyDtoH(hOut.get(), dOut, bufferSize));
for (size_t i = 0; i < n; ++i) {
if (fabs(a * hX[i] + hY[i] - hOut[i]) > fabs(hOut[i])* 1e-6) {
@@ -146,11 +147,11 @@ int main()
}
}
hipFree((void *)dX);
hipFree((void *)dY);
hipFree((void *)dOut);
checkHipErrors(hipFree((void *)dX));
checkHipErrors(hipFree((void *)dY));
checkHipErrors(hipFree((void *)dOut));
hipModuleUnload(module);
checkHipErrors(hipModuleUnload(module));
cout << "SAXPY test completed" << endl;
}
@@ -40,5 +40,7 @@ set(CMAKE_BUILD_TYPE Release)
# Create the excutable
add_executable(sharedMemory sharedMemory.cpp)
target_include_directories(sharedMemory PRIVATE ../../common)
# Link with HIP
target_link_libraries(sharedMemory hip::host)
+2 -1
View File
@@ -32,6 +32,7 @@ TARGET=hcc
SOURCES = sharedMemory.cpp
OBJECTS = $(SOURCES:.cpp=.o)
INCLUDES := -I../../common
EXECUTABLE=./sharedMemory
@@ -40,7 +41,7 @@ EXECUTABLE=./sharedMemory
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
@@ -24,7 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 64
@@ -66,7 +66,7 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
@@ -83,11 +83,11 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
@@ -95,7 +95,7 @@ int main() {
gpuMatrix, WIDTH);
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
@@ -116,8 +116,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
+2
View File
@@ -40,5 +40,7 @@ set(CMAKE_BUILD_TYPE Release)
# Create the excutable
add_executable(shfl shfl.cpp)
target_include_directories(shfl PRIVATE ../../common)
# Link with HIP
target_link_libraries(shfl hip::host)
+2 -1
View File
@@ -36,6 +36,7 @@ TARGET=hcc
SOURCES = shfl.cpp
OBJECTS = $(SOURCES:.cpp=.o)
INCLUDES := -I../../common
EXECUTABLE=./shfl
@@ -44,7 +45,7 @@ EXECUTABLE=./shfl
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
+8 -8
View File
@@ -24,7 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 4
@@ -63,7 +63,7 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
@@ -80,18 +80,18 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(1), dim3(THREADS_PER_BLOCK_X * THREADS_PER_BLOCK_Y), 0, 0,
gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
@@ -112,8 +112,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
+8 -7
View File
@@ -24,6 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 4
@@ -61,7 +62,7 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
@@ -78,18 +79,18 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(1), dim3(THREADS_PER_BLOCK_X, THREADS_PER_BLOCK_Y), 0, 0,
gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
@@ -110,8 +111,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
@@ -39,5 +39,7 @@ set(CMAKE_CXX_LINKER ${HIP_HIPCC_EXECUTABLE})
# Create the excutable
add_executable(2dshfl 2dshfl.cpp)
target_include_directories(2dshfl PRIVATE ../../common)
# Link with HIP
target_link_libraries(2dshfl hip::host)
+2 -1
View File
@@ -36,6 +36,7 @@ TARGET=hcc
SOURCES = 2dshfl.cpp
OBJECTS = $(SOURCES:.cpp=.o)
INCLUDES := -I../../common
EXECUTABLE=./2dshfl
@@ -44,7 +45,7 @@ EXECUTABLE=./2dshfl
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
@@ -22,6 +22,8 @@ project(dynamic_shared)
cmake_minimum_required(VERSION 3.10)
include_directories(../../common)
if (NOT DEFINED ROCM_PATH )
set ( ROCM_PATH "/opt/rocm" CACHE STRING "Default ROCM installation directory." )
endif ()
@@ -39,5 +41,7 @@ set(CMAKE_CXX_LINKER ${HIP_HIPCC_EXECUTABLE})
# Create the excutable
add_executable(dynamic_shared dynamic_shared.cpp)
target_include_directories(dynamic_shared PRIVATE ../../common)
# Link with HIP
target_link_libraries(dynamic_shared hip::host)
+2 -1
View File
@@ -32,6 +32,7 @@ TARGET=hcc
SOURCES = dynamic_shared.cpp
OBJECTS = $(SOURCES:.cpp=.o)
INCLUDES := -I../../common
EXECUTABLE=./dynamic_shared
@@ -40,7 +41,7 @@ EXECUTABLE=./dynamic_shared
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
@@ -24,6 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define WIDTH 16
@@ -65,7 +66,7 @@ int main() {
float* gpuTransposeMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
@@ -82,11 +83,11 @@ int main() {
}
// allocate the memory on the device side
hipMalloc((void**)&gpuMatrix, NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuMatrix, NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix, NUM * sizeof(float)));
// Memory transfer from host to device
hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, NUM * sizeof(float), hipMemcpyHostToDevice));
// Lauching kernel from host
hipLaunchKernelGGL(matrixTranspose, dim3(WIDTH / THREADS_PER_BLOCK_X, WIDTH / THREADS_PER_BLOCK_Y),
@@ -94,7 +95,7 @@ int main() {
0, gpuTransposeMatrix, gpuMatrix, WIDTH);
// Memory transfer from device to host
hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(TransposeMatrix, gpuTransposeMatrix, NUM * sizeof(float), hipMemcpyDeviceToHost));
// CPU MatrixTranspose computation
matrixTransposeCPUReference(cpuTransposeMatrix, Matrix, WIDTH);
@@ -115,8 +116,8 @@ int main() {
}
// free the resources on device side
hipFree(gpuMatrix);
hipFree(gpuTransposeMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuTransposeMatrix));
// free the resources on host side
free(Matrix);
@@ -39,5 +39,7 @@ set(CMAKE_CXX_LINKER ${HIP_HIPCC_EXECUTABLE})
# Create the excutable
add_executable(stream stream.cpp)
target_include_directories(stream PRIVATE ../../common)
# Link with HIP
target_link_libraries(stream hip::host)
+2 -1
View File
@@ -32,6 +32,7 @@ TARGET=hcc
SOURCES = stream.cpp
OBJECTS = $(SOURCES:.cpp=.o)
INCLUDES := -I../../common
EXECUTABLE=./stream
@@ -40,7 +41,7 @@ EXECUTABLE=./stream
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
+13 -12
View File
@@ -22,6 +22,7 @@ THE SOFTWARE.
#include <iostream>
#include <hip/hip_runtime.h>
#include "hip_helper.h"
#define WIDTH 32
@@ -66,11 +67,11 @@ void MultipleStream(float** data, float* randArray, float** gpuTransposeMatrix,
const int num_streams = 2;
hipStream_t streams[num_streams];
for (int i = 0; i < num_streams; i++) hipStreamCreate(&streams[i]);
for (int i = 0; i < num_streams; i++) checkHipErrors(hipStreamCreate(&streams[i]));
for (int i = 0; i < num_streams; i++) {
hipMalloc((void**)&data[i], NUM * sizeof(float));
hipMemcpyAsync(data[i], randArray, NUM * sizeof(float), hipMemcpyHostToDevice, streams[i]);
checkHipErrors(hipMalloc((void**)&data[i], NUM * sizeof(float)));
checkHipErrors(hipMemcpyAsync(data[i], randArray, NUM * sizeof(float), hipMemcpyHostToDevice, streams[i]));
}
hipLaunchKernelGGL(matrixTranspose_static_shared,
@@ -84,12 +85,12 @@ void MultipleStream(float** data, float* randArray, float** gpuTransposeMatrix,
streams[1], gpuTransposeMatrix[1], data[1], width);
for (int i = 0; i < num_streams; i++)
hipMemcpyAsync(TransposeMatrix[i], gpuTransposeMatrix[i], NUM * sizeof(float),
hipMemcpyDeviceToHost, streams[i]);
checkHipErrors(hipMemcpyAsync(TransposeMatrix[i], gpuTransposeMatrix[i], NUM * sizeof(float),
hipMemcpyDeviceToHost, streams[i]));
}
int main() {
hipSetDevice(0);
checkHipErrors(hipSetDevice(0));
float *data[2], *TransposeMatrix[2], *gpuTransposeMatrix[2], *randArray;
@@ -100,8 +101,8 @@ int main() {
TransposeMatrix[0] = (float*)malloc(NUM * sizeof(float));
TransposeMatrix[1] = (float*)malloc(NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix[0], NUM * sizeof(float));
hipMalloc((void**)&gpuTransposeMatrix[1], NUM * sizeof(float));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix[0], NUM * sizeof(float)));
checkHipErrors(hipMalloc((void**)&gpuTransposeMatrix[1], NUM * sizeof(float)));
for (int i = 0; i < NUM; i++) {
randArray[i] = (float)i * 1.0f;
@@ -109,7 +110,7 @@ int main() {
MultipleStream(data, randArray, gpuTransposeMatrix, TransposeMatrix, width);
hipDeviceSynchronize();
checkHipErrors(hipDeviceSynchronize());
// verify the results
int errors = 0;
@@ -128,11 +129,11 @@ int main() {
free(randArray);
for (int i = 0; i < 2; i++) {
hipFree(data[i]);
hipFree(gpuTransposeMatrix[i]);
checkHipErrors(hipFree(data[i]));
checkHipErrors(hipFree(gpuTransposeMatrix[i]));
free(TransposeMatrix[i]);
}
hipDeviceReset();
checkHipErrors(hipDeviceReset());
return 0;
}
@@ -39,5 +39,7 @@ set(CMAKE_CXX_LINKER ${HIP_HIPCC_EXECUTABLE})
# Create the excutable
add_executable(unroll unroll.cpp)
target_include_directories(unroll PRIVATE ../../common)
# Link with HIP
target_link_libraries(unroll hip::host)
+2 -1
View File
@@ -36,6 +36,7 @@ TARGET=hcc
SOURCES = unroll.cpp
OBJECTS = $(SOURCES:.cpp=.o)
INCLUDES := -I../../common
EXECUTABLE=./unroll
@@ -44,7 +45,7 @@ EXECUTABLE=./unroll
all: $(EXECUTABLE) test
CXXFLAGS =-g
CXXFLAGS =-g $(INCLUDES)
CXX=$(HIPCC)
+9 -8
View File
@@ -24,6 +24,7 @@ THE SOFTWARE.
// hip header file
#include "hip/hip_runtime.h"
#include "hip_helper.h"
#define LENGTH 4
@@ -59,7 +60,7 @@ int main() {
int* gpuSumMatrix;
hipDeviceProp_t devProp;
hipGetDeviceProperties(&devProp, 0);
checkHipErrors(hipGetDeviceProperties(&devProp, 0));
std::cout << "Device name " << devProp.name << std::endl;
@@ -76,19 +77,19 @@ int main() {
}
// Allocated Device Memory
hipMalloc((void**)&gpuMatrix, SIZE * sizeof(int));
hipMalloc((void**)&gpuSumMatrix, LENGTH * sizeof(int));
checkHipErrors(hipMalloc((void**)&gpuMatrix, SIZE * sizeof(int)));
checkHipErrors(hipMalloc((void**)&gpuSumMatrix, LENGTH * sizeof(int)));
// Memory Copy to Device
hipMemcpy(gpuMatrix, Matrix, SIZE * sizeof(int), hipMemcpyHostToDevice);
hipMemcpy(gpuSumMatrix, cpuSumMatrix, LENGTH * sizeof(float), hipMemcpyHostToDevice);
checkHipErrors(hipMemcpy(gpuMatrix, Matrix, SIZE * sizeof(int), hipMemcpyHostToDevice));
checkHipErrors(hipMemcpy(gpuSumMatrix, cpuSumMatrix, LENGTH * sizeof(float), hipMemcpyHostToDevice));
// Launch device kernels
hipLaunchKernelGGL(gpuMatrixRowSum, dim3(BLOCKS_PER_GRID), dim3(THREADS_PER_BLOCK), 0, 0,
gpuMatrix, gpuSumMatrix, LENGTH);
// Memory copy back to device
hipMemcpy(sumMatrix, gpuSumMatrix, LENGTH * sizeof(int), hipMemcpyDeviceToHost);
checkHipErrors(hipMemcpy(sumMatrix, gpuSumMatrix, LENGTH * sizeof(int), hipMemcpyDeviceToHost));
// Cpu implementation
matrixRowSum(Matrix, cpuSumMatrix, LENGTH);
@@ -110,8 +111,8 @@ int main() {
}
// GPU Free
hipFree(gpuMatrix);
hipFree(gpuSumMatrix);
checkHipErrors(hipFree(gpuMatrix));
checkHipErrors(hipFree(gpuSumMatrix));
// CPU Free
free(Matrix);