SWDEV-470698 - fix formatting, add format check workflow (#657)

このコミットが含まれているのは:
Danylo Lytovchenko
2025-08-20 16:28:06 +02:00
committed by GitHub
コミット f7338717ae
1574個のファイルの変更、162972行の追加、199346行の削除
+3 -4
ファイルの表示
@@ -21,15 +21,14 @@ THE SOFTWARE.
__device__ int globalDevData = 10;
extern "C" __global__ void addKernel(int *a, int size) {
extern "C" __global__ void addKernel(int* a, int size) {
int offset = blockDim.x * blockIdx.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
for (int i = offset; i < size; i+= stride) {
for (int i = offset; i < size; i += stride) {
a[i] += 2;
}
}
texture<float, 2> tex;
extern "C" __global__ void sampleModuleKernel() {
}
extern "C" __global__ void sampleModuleKernel() {}
+11 -12
ファイルの表示
@@ -24,15 +24,15 @@ THE SOFTWARE.
using namespace cooperative_groups;
extern "C" {
__global__ void cooperativeKernelEx(int* output, int totalThreads) {
grid_group grid = this_grid();
int tid = threadIdx.x + blockDim.x * blockIdx.x;
if (tid < totalThreads) {
output[tid] = tid * 3;
}
grid.sync();
if (tid == 0) {
output[0] = 2222;
}
grid_group grid = this_grid();
int tid = threadIdx.x + blockDim.x * blockIdx.x;
if (tid < totalThreads) {
output[tid] = tid * 3;
}
grid.sync();
if (tid == 0) {
output[0] = 2222;
}
}
/*
@@ -44,7 +44,7 @@ __global__ void emptyKernel() {}
* Kernel which doesn't use cooperative groups and takes an argument
* and updates the value with 100
*/
__global__ void argKernel(int *val) { *val = 100; }
__global__ void argKernel(int* val) { *val = 100; }
/*
* Kernel which uses cooperative groups and without any arguments
@@ -62,7 +62,7 @@ __global__ void coopEmptykernel() {
* 2) Wait for all the blocks completes it's operations
* 3) Fill each element in the output array with sum of elements in arr
*/
__global__ void coopFillArrayKernel(int *arr, int *output, int N) {
__global__ void coopFillArrayKernel(int* arr, int* output, int N) {
cooperative_groups::grid_group grid = cooperative_groups::this_grid();
if (blockIdx.x == 0)
@@ -93,4 +93,3 @@ __global__ void coopFillArrayKernel(int *arr, int *output, int N) {
}
}
}
+10 -10
ファイルの表示
@@ -19,15 +19,15 @@ THE SOFTWARE.
#include "hip/hip_runtime.h"
extern "C" __global__ void
kernelMultipleArgsSaxpy(int a1, int a2, int *x1, int b1, int b2, int *x2,
int c1, int c2, int *x3, int d1, int d2, int *x4, int e1, int e2, int *x5,
int f1, int f2, int *x6) {
extern "C" __global__ void kernelMultipleArgsSaxpy(int a1, int a2, int* x1, int b1, int b2, int* x2,
int c1, int c2, int* x3, int d1, int d2, int* x4,
int e1, int e2, int* x5, int f1, int f2,
int* x6) {
int id = threadIdx.x + blockIdx.x * blockDim.x;
x1[id] = a1*x1[id] + a2;
x2[id] = b1*x2[id] + b2;
x3[id] = c1*x3[id] + c2;
x4[id] = d1*x4[id] + d2;
x5[id] = e1*x5[id] + e2;
x6[id] = f1*x6[id] + f2;
x1[id] = a1 * x1[id] + a2;
x2[id] = b1 * x2[id] + b2;
x3[id] = c1 * x3[id] + c2;
x4[id] = d1 * x4[id] + d2;
x5[id] = e1 * x5[id] + e2;
x6[id] = f1 * x6[id] + f2;
}
+1 -1
ファイルの表示
@@ -21,7 +21,7 @@ THE SOFTWARE.
*/
#include "hip/hip_runtime.h"
extern "C" __global__ void copy_ker(int* Ad, int *Bd, size_t size) {
extern "C" __global__ void copy_ker(int* Ad, int* Bd, size_t size) {
int myId = threadIdx.x + blockDim.x * blockIdx.x;
if (myId < size) {
Bd[myId] = Ad[myId];
+1 -1
ファイルの表示
@@ -25,4 +25,4 @@ THE SOFTWARE.
texture<float, 2> tex;
#endif // CUDA_VERSION < CUDA_12000
#endif // CUDA_VERSION < CUDA_12000
+27 -34
ファイルの表示
@@ -55,7 +55,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
int blockSize = 16;
int numBlocks = (totalThreads + blockSize - 1) / blockSize;
int *d_output = nullptr;
int* d_output = nullptr;
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
@@ -68,7 +68,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
config.blockDimY = 1;
config.blockDimZ = 1;
config.sharedMemBytes = 0;
config.hStream = 0; // default stream
config.hStream = 0; // default stream
// Set up a cooperative launch attribute
hipDrvLaunchAttribute attr;
@@ -81,7 +81,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
config.numAttrs = 1;
// Kernel parameters: address of d_output and totalThreads.
void *kernelParams[] = {&d_output, &totalThreads};
void* kernelParams[] = {&d_output, &totalThreads};
hipModule_t module;
HIP_CHECK(hipModuleLoad(&module, CODE_OBJ_SINGLEARCH));
@@ -98,8 +98,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
hipErrorInvalidResourceHandle);
}
SECTION("Kernel parameter as nullptr") {
HIP_CHECK_ERROR(hipDrvLaunchKernelEx(&config, function, nullptr, NULL),
hipErrorInvalidValue);
HIP_CHECK_ERROR(hipDrvLaunchKernelEx(&config, function, nullptr, NULL), hipErrorInvalidValue);
}
HIP_LAUNCH_CONFIG invalidConfig = {};
invalidConfig.gridDimX = 0;
@@ -109,7 +108,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
invalidConfig.blockDimY = 1;
invalidConfig.blockDimZ = 1;
invalidConfig.sharedMemBytes = 0;
invalidConfig.hStream = 0; // default stream
invalidConfig.hStream = 0; // default stream
// Set up a cooperative launch attribute
hipDrvLaunchAttribute invalidAttr;
@@ -122,17 +121,16 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
invalidConfig.numAttrs = 1;
SECTION("Invalid Kernel config") {
HIP_CHECK_ERROR(
hipDrvLaunchKernelEx(&invalidConfig, function, kernelParams, NULL),
hipErrorInvalidConfiguration);
HIP_CHECK_ERROR(hipDrvLaunchKernelEx(&invalidConfig, function, kernelParams, NULL),
hipErrorInvalidConfiguration);
}
}
bool runTestDrvLaunch(const char *testName, std::string kernelFunc,
int totalThreads, int blockSize, int flagValue) {
bool runTestDrvLaunch(const char* testName, std::string kernelFunc, int totalThreads, int blockSize,
int flagValue) {
int numBlocks = (totalThreads + blockSize - 1) / blockSize;
int *d_output = nullptr;
int* d_output = nullptr;
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
@@ -145,7 +143,7 @@ bool runTestDrvLaunch(const char *testName, std::string kernelFunc,
config.blockDimY = 1;
config.blockDimZ = 1;
config.sharedMemBytes = 0;
config.hStream = 0; // default stream
config.hStream = 0; // default stream
// Set up a cooperative launch attribute
hipDrvLaunchAttribute attr;
@@ -158,7 +156,7 @@ bool runTestDrvLaunch(const char *testName, std::string kernelFunc,
config.numAttrs = 1;
// Kernel parameters: address of d_output and totalThreads.
void *kernelParams[] = {&d_output, &totalThreads};
void* kernelParams[] = {&d_output, &totalThreads};
hipModule_t module;
HIP_CHECK(hipModuleLoad(&module, CODE_OBJ_SINGLEARCH));
@@ -176,22 +174,21 @@ bool runTestDrvLaunch(const char *testName, std::string kernelFunc,
HIP_CHECK(hipDeviceSynchronize());
int *h_output = (int *)malloc(totalThreads * sizeof(int));
HIP_CHECK(hipMemcpy(h_output, d_output, totalThreads * sizeof(int),
hipMemcpyDeviceToHost));
int* h_output = (int*)malloc(totalThreads * sizeof(int));
HIP_CHECK(hipMemcpy(h_output, d_output, totalThreads * sizeof(int), hipMemcpyDeviceToHost));
// Verify results.
bool success = true;
if (h_output[0] != flagValue) {
printf("%s test failed: Expected flag %d at index 0, got %d\n", testName,
flagValue, h_output[0]);
printf("%s test failed: Expected flag %d at index 0, got %d\n", testName, flagValue,
h_output[0]);
success = false;
}
for (int i = 1; i < totalThreads; i++) {
int expectedValue = (flagValue == 1111) ? i : (i * 3);
if (h_output[i] != expectedValue) {
printf("%s test failed at index %d: Expected %d, got %d\n", testName, i,
expectedValue, h_output[i]);
printf("%s test failed at index %d: Expected %d, got %d\n", testName, i, expectedValue,
h_output[i]);
success = false;
break;
}
@@ -217,8 +214,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_Functional") {
HipTest::HIP_SKIP_TEST("CooperativeLaunch not supported");
return;
}
REQUIRE(runTestDrvLaunch("hipDrvLaunchKernelEx", kernel_name, 64, 16, 2222) ==
true);
REQUIRE(runTestDrvLaunch("hipDrvLaunchKernelEx", kernel_name, 64, 16, 2222) == true);
}
/**
@@ -271,10 +267,10 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_Different_Kernels") {
}
SECTION("Kernel with arguments using kernelParams") {
int *devMem = nullptr;
int* devMem = nullptr;
HIP_CHECK(hipMalloc(&devMem, sizeof(int)));
void *kernel_args[1] = {&devMem};
void* kernel_args[1] = {&devMem};
hipFunction_t argKernel;
HIP_CHECK(hipModuleGetFunction(&argKernel, module, "argKernel"));
@@ -343,15 +339,15 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_CooperativeKernelWithArgs") {
hostMem[i] = 0;
}
int *devMem1 = nullptr;
int* devMem1 = nullptr;
HIP_CHECK(hipMalloc(&devMem1, N * sizeof(int)));
HIP_CHECK(hipMemcpy(devMem1, hostMem, N * sizeof(int), hipMemcpyDefault));
int *devMem2 = nullptr;
int* devMem2 = nullptr;
HIP_CHECK(hipMalloc(&devMem2, N * sizeof(int)));
HIP_CHECK(hipMemcpy(devMem2, hostMem, N * sizeof(int), hipMemcpyDefault));
int size = N;
void *kernel_args[3] = {&devMem1, &devMem2, &size};
void* kernel_args[3] = {&devMem1, &devMem2, &size};
hipFunction_t argKernel;
HIP_CHECK(hipModuleGetFunction(&argKernel, module, "coopFillArrayKernel"));
@@ -416,8 +412,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_MaxBlockDims") {
config.numAttrs = 1;
SECTION("blockDim.x == maxBlockDimX") {
const unsigned int x =
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimX, 0);
const unsigned int x = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimX, 0);
config.blockDimX = x;
HIP_CHECK(hipDrvLaunchKernelEx(&config, kernel, nullptr, nullptr));
@@ -425,8 +420,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_MaxBlockDims") {
}
SECTION("blockDim.y == maxBlockDimY") {
const unsigned int y =
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimY, 0);
const unsigned int y = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimY, 0);
config.blockDimY = y;
HIP_CHECK(hipDrvLaunchKernelEx(&config, kernel, nullptr, nullptr));
@@ -434,8 +428,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_MaxBlockDims") {
}
SECTION("blockDim.z == maxBlockDimZ") {
const unsigned int z =
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimZ, 0);
const unsigned int z = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimZ, 0);
config.blockDimY = z;
HIP_CHECK(hipDrvLaunchKernelEx(&config, kernel, nullptr, nullptr));
+39 -46
ファイルの表示
@@ -48,25 +48,21 @@ THE SOFTWARE.
__device__ int globalvar = 1;
__device__ void Delay(uint32_t interval, const uint32_t ticks_per_ms) {
while (interval--) {
#if HT_AMD
#if HT_AMD
uint64_t start = wall_clock64();
while (wall_clock64() - start < ticks_per_ms) {
__builtin_amdgcn_s_sleep(10);
}
#endif
#if HT_NVIDIA
#endif
#if HT_NVIDIA
uint64_t start = clock64();
while (clock64() - start < ticks_per_ms) {
}
#endif
#endif
}
}
__global__ void TwoSecKernel(int clockrate) {
Delay(2000, clockrate);
}
__global__ void FourSecKernel(int clockrate) {
Delay(4000, clockrate);
}
__global__ void TwoSecKernel(int clockrate) { Delay(2000, clockrate); }
__global__ void FourSecKernel(int clockrate) { Delay(4000, clockrate); }
bool DisableTimeFlag() {
bool testStatus = true;
@@ -74,21 +70,19 @@ bool DisableTimeFlag() {
HIP_CHECK(hipSetDevice(0));
hipError_t e;
float time_2sec;
hipEvent_t start_event1, end_event1;
hipEvent_t start_event1, end_event1;
int clkRate = 0;
#if HT_AMD
#if HT_AMD
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeWallClockRate, 0));
#endif
#if HT_NVIDIA
#endif
#if HT_NVIDIA
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeClockRate, 0));
#endif
HIP_CHECK(hipEventCreateWithFlags(&start_event1,
hipEventDisableTiming));
HIP_CHECK(hipEventCreateWithFlags(&end_event1,
hipEventDisableTiming));
#endif
HIP_CHECK(hipEventCreateWithFlags(&start_event1, hipEventDisableTiming));
HIP_CHECK(hipEventCreateWithFlags(&end_event1, hipEventDisableTiming));
HIP_CHECK(hipStreamCreate(&stream1));
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0,
stream1, start_event1, end_event1, 0, clkRate);
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0, stream1, start_event1, end_event1, 0,
clkRate);
HIP_CHECK(hipStreamSynchronize(stream1));
e = hipEventElapsedTime(&time_2sec, start_event1, end_event1);
if (e == hipErrorInvalidHandle) {
@@ -108,12 +102,12 @@ bool ConcurencyCheck_GlobalVar(int conc_flag) {
int deviceGlobal_h = 0;
HIP_CHECK(hipSetDevice(0));
int clkRate = 0;
#if HT_AMD
#if HT_AMD
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeWallClockRate, 0));
#endif
#if HT_NVIDIA
#endif
#if HT_NVIDIA
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeClockRate, 0));
#endif
#endif
HIP_CHECK(hipStreamCreate(&stream1));
hipDeviceProp_t props{};
int device;
@@ -121,15 +115,14 @@ bool ConcurencyCheck_GlobalVar(int conc_flag) {
HIP_CHECK(hipGetDeviceProperties(&props, device));
if ((std::string(props.gcnArchName).find("gfx1101") != std::string::npos) ||
(std::string(props.gcnArchName).find("gfx1100") != std::string::npos)) {
hipExtLaunchKernelGGL((FourSecKernel), dim3(1), dim3(1), 0,
stream1, nullptr, nullptr, conc_flag, clkRate);
hipExtLaunchKernelGGL((FourSecKernel), dim3(1), dim3(1), 0, stream1, nullptr, nullptr,
conc_flag, clkRate);
} else {
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0,
stream1, nullptr, nullptr, conc_flag, clkRate);
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0, stream1, nullptr, nullptr, conc_flag,
clkRate);
}
HIP_CHECK(hipStreamSynchronize(stream1));
HIP_CHECK(hipMemcpyFromSymbol(&deviceGlobal_h, globalvar,
sizeof(int)));
HIP_CHECK(hipMemcpyFromSymbol(&deviceGlobal_h, globalvar, sizeof(int)));
if (conc_flag && deviceGlobal_h != 0x5555) {
testStatus = true;
@@ -148,15 +141,15 @@ bool KernelTimeExecution() {
bool testStatus = true;
hipStream_t stream1;
HIP_CHECK(hipSetDevice(0));
hipEvent_t start_event1, end_event1, start_event2, end_event2;
hipEvent_t start_event1, end_event1, start_event2, end_event2;
float time_4sec, time_2sec;
int clkRate = 0;
#if HT_AMD
#if HT_AMD
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeWallClockRate, 0));
#endif
#if HT_NVIDIA
#endif
#if HT_NVIDIA
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeClockRate, 0));
#endif
#endif
HIP_CHECK(hipEventCreate(&start_event1));
HIP_CHECK(hipEventCreate(&end_event1));
@@ -167,16 +160,16 @@ bool KernelTimeExecution() {
int device;
HIP_CHECK(hipGetDevice(&device));
HIP_CHECK(hipGetDeviceProperties(&props, device));
hipExtLaunchKernelGGL((FourSecKernel), dim3(1), dim3(1), 0,
stream1, start_event1, end_event1, 0, clkRate);
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0,
stream1, start_event2, end_event2, 0, clkRate);
hipExtLaunchKernelGGL((FourSecKernel), dim3(1), dim3(1), 0, stream1, start_event1, end_event1, 0,
clkRate);
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0, stream1, start_event2, end_event2, 0,
clkRate);
HIP_CHECK(hipStreamSynchronize(stream1));
HIP_CHECK(hipEventElapsedTime(&time_4sec, start_event1, end_event1));
HIP_CHECK(hipEventElapsedTime(&time_2sec, start_event2, end_event2));
if ( (time_4sec < static_cast<float>(FIVESEC_KERNEL)) &&
(time_2sec < static_cast<float>(THREESEC_KERNEL))) {
if ((time_4sec < static_cast<float>(FIVESEC_KERNEL)) &&
(time_2sec < static_cast<float>(THREESEC_KERNEL))) {
testStatus = true;
} else {
testStatus = false;
@@ -193,11 +186,11 @@ bool KernelTimeExecution() {
TEST_CASE("Unit_hipExtLaunchKernelGGL_Functional") {
bool testStatus = true;
// Disabled the concurency test as the firmware does not support concurrency
// in the same stream
#if 0
// Disabled the concurency test as the firmware does not support concurrency
// in the same stream
#if 0
testStatus &= ConcurencyCheck_GlobalVar(0);
#endif
#endif
SECTION("Kernel Execution Time") {
testStatus &= KernelTimeExecution();
REQUIRE(testStatus == true);
+20 -22
ファイルの表示
@@ -20,15 +20,15 @@ THE SOFTWARE.
#include <hip_test_defgroups.hh>
/**
* @addtogroup hipExtLaunchMultiKernelMultiDevice
* @{
* @ingroup ModuleTest
* `hipError_t hipExtLaunchMultiKernelMultiDevice(hipLaunchParams* launchParamsList,
* int numDevices, unsigned int flags)` -
* Launches kernels on multiple devices and guarantees all specified kernels are dispatched
* on respective streams before enqueuing any other work on the specified streams from any
* other threads
*/
* @addtogroup hipExtLaunchMultiKernelMultiDevice
* @{
* @ingroup ModuleTest
* `hipError_t hipExtLaunchMultiKernelMultiDevice(hipLaunchParams* launchParamsList,
* int numDevices, unsigned int flags)` -
* Launches kernels on multiple devices and guarantees all specified kernels are dispatched
* on respective streams before enqueuing any other work on the specified streams from any
* other threads
*/
/**
* Test Description
@@ -44,8 +44,7 @@ THE SOFTWARE.
// Square each element in the array A and write to array C.
#define NUM_KERNEL_ARGS 3
__global__ void
vector_square(float *C_d, float *A_d, size_t N) {
__global__ void vector_square(float* C_d, float* A_d, size_t N) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
@@ -76,7 +75,7 @@ TEST_CASE("Unit_hipExtLaunchMultiKernelMultiDevice_Functional") {
HIP_CHECK(C_h == 0 ? hipErrorOutOfMemory : hipSuccess);
// Fill with Phi + i
for (size_t i = 0; i < N; i++) {
A_h[i] = 1.618f + i;
A_h[i] = 1.618f + i;
}
const unsigned blocks = 512;
@@ -97,22 +96,21 @@ TEST_CASE("Unit_hipExtLaunchMultiKernelMultiDevice_Functional") {
HIP_CHECK(hipMemcpy(A_d[i], A_h, Nbytes, hipMemcpyHostToDevice));
}
hipLaunchParams *launchParamsList = reinterpret_cast<hipLaunchParams *>(
malloc(sizeof(hipLaunchParams)*nGpu));
hipLaunchParams* launchParamsList =
reinterpret_cast<hipLaunchParams*>(malloc(sizeof(hipLaunchParams) * nGpu));
void *args[MAX_GPUS * NUM_KERNEL_ARGS];
void* args[MAX_GPUS * NUM_KERNEL_ARGS];
for (int i = 0; i < nGpu; i++) {
args[i * NUM_KERNEL_ARGS] = &C_d[i];
args[i * NUM_KERNEL_ARGS] = &C_d[i];
args[i * NUM_KERNEL_ARGS + 1] = &A_d[i];
args[i * NUM_KERNEL_ARGS + 2] = &N;
launchParamsList[i].func =
reinterpret_cast<void *>(vector_square);
launchParamsList[i].gridDim = dim3(blocks);
launchParamsList[i].blockDim = dim3(threadsPerBlock);
launchParamsList[i].func = reinterpret_cast<void*>(vector_square);
launchParamsList[i].gridDim = dim3(blocks);
launchParamsList[i].blockDim = dim3(threadsPerBlock);
launchParamsList[i].sharedMem = 0;
launchParamsList[i].stream = stream[i];
launchParamsList[i].args = args + i * NUM_KERNEL_ARGS;
launchParamsList[i].stream = stream[i];
launchParamsList[i].args = args + i * NUM_KERNEL_ARGS;
}
INFO("info: launch vector_square kernel with")
+126 -215
ファイルの表示
@@ -156,14 +156,12 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_NonUniformWorkGroup") {
args.buffersize = arraylength;
size_t size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
// Memcpy from A to Ad
HIP_CHECK(hipMemcpy(Ad, A, sizeBytes, hipMemcpyDefault));
REQUIRE(hipErrorInvalidValue ==
hipExtModuleLaunchKernel(Function, arraylength, 1, 1, localWorkSize,
1, 1, 0, 0, NULL,
hipExtModuleLaunchKernel(Function, arraylength, 1, 1, localWorkSize, 1, 1, 0, 0, NULL,
reinterpret_cast<void**>(&config), 0));
HIP_CHECK(hipDeviceSynchronize());
HIP_CHECK(hipFree(Ad));
@@ -193,12 +191,8 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_UniformWorkGroup") {
// Get module and function from module
hipModule_t Module;
hipFunction_t Function;
SECTION("regular fatbin") {
HIP_CHECK(hipModuleLoad(&Module, fileName));
}
SECTION("compressed fatbin") {
HIP_CHECK(hipModuleLoad(&Module, fileNameCompressed));
}
SECTION("regular fatbin") { HIP_CHECK(hipModuleLoad(&Module, fileName)); }
SECTION("compressed fatbin") { HIP_CHECK(hipModuleLoad(&Module, fileNameCompressed)); }
SECTION("generic target in regular fatbin") {
if (!isGenericTargetSupported()) {
fprintf(stderr, "Generic target test is skipped\n");
@@ -237,13 +231,11 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_UniformWorkGroup") {
args.buffersize = arraylength;
size_t size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
// Memcpy from A to Ad
HIP_CHECK(hipMemcpy(Ad, A, sizeBytes, hipMemcpyDefault));
HIP_CHECK(hipExtModuleLaunchKernel(Function, arraylength, 1, 1, localWorkSize,
1, 1, 0, 0, NULL,
HIP_CHECK(hipExtModuleLaunchKernel(Function, arraylength, 1, 1, localWorkSize, 1, 1, 0, 0, NULL,
reinterpret_cast<void**>(&config), 0));
// Memcpy results back to host
HIP_CHECK(hipMemcpy(B, Bd, sizeBytes, hipMemcpyDefault));
@@ -266,8 +258,7 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_Positive_Parameters") {
hipEvent_t start_event = nullptr;
HIP_CHECK(hipEventCreate(&start_event));
const auto kernel = GetKernel(mg.module(), "NOPKernel");
HIP_CHECK(hipExtModuleLaunchKernel(kernel, 1, 1, 1, 1, 1, 1, 0, nullptr,
nullptr, nullptr,
HIP_CHECK(hipExtModuleLaunchKernel(kernel, 1, 1, 1, 1, 1, 1, 0, nullptr, nullptr, nullptr,
start_event, nullptr));
HIP_CHECK(hipDeviceSynchronize());
HIP_CHECK(hipEventQuery(start_event));
@@ -278,8 +269,7 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_Positive_Parameters") {
hipEvent_t stop_event = nullptr;
HIP_CHECK(hipEventCreate(&stop_event));
const auto kernel = GetKernel(mg.module(), "NOPKernel");
HIP_CHECK(hipExtModuleLaunchKernel(kernel, 1, 1, 1, 1, 1, 1, 0, nullptr,
nullptr, nullptr,
HIP_CHECK(hipExtModuleLaunchKernel(kernel, 1, 1, 1, 1, 1, 1, 0, nullptr, nullptr, nullptr,
nullptr, stop_event));
HIP_CHECK(hipDeviceSynchronize());
HIP_CHECK(hipEventQuery(stop_event));
@@ -294,9 +284,11 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_Negative_Parameters") {
* Test Description
* ------------------------
* - Test case to verify Negative tests of hipExtModuleLaunchKernel API.
* - Test case to verify kernel execution time of the particular kernel by using hipExtModuleLaunchKernel.
* - Test case to verify kernel execution time of the particular kernel by using
hipExtModuleLaunchKernel.
* - Test case to verify hipExtModuleLaunchKernel API by disabling time flag in event creation.
* - Test case to verify hipExtModuleLaunchKernel API's Corner Scenarios for Grid and Block dimensions.
* - Test case to verify hipExtModuleLaunchKernel API's Corner Scenarios for Grid and Block
dimensions.
* - Test case to verify different work groups of hipExtModuleLaunchKernel API.
* Test source
@@ -317,16 +309,16 @@ struct gridblockDim {
};
class ModuleLaunchKernel {
int N = 64;
int SIZE = N*N;
int SIZE = N * N;
int *A, *B, *C;
hipDeviceptr_t *Ad, *Bd;
hipStream_t stream1, stream2;
hipEvent_t start_event1, end_event1, start_event2, end_event2,
start_timingDisabled, end_timingDisabled;
hipEvent_t start_event1, end_event1, start_event2, end_event2, start_timingDisabled,
end_timingDisabled;
hipModule_t Module;
hipDeviceptr_t deviceGlobal;
hipFunction_t MultKernel, SixteenSecKernel, FourSecKernel,
TwoSecKernel, KernelandExtraParamKernel, DummyKernel;
hipFunction_t MultKernel, SixteenSecKernel, FourSecKernel, TwoSecKernel,
KernelandExtraParamKernel, DummyKernel;
struct {
int clockRate;
void* _Ad;
@@ -340,7 +332,8 @@ class ModuleLaunchKernel {
size_t size2;
size_t size3;
size_t deviceGlobalSize;
public :
public:
void AllocateMemory();
void DeAllocateMemory();
void ModuleLoad();
@@ -355,37 +348,37 @@ class ModuleLaunchKernel {
};
void ModuleLaunchKernel::AllocateMemory() {
A = new int[N*N*sizeof(int)];
B = new int[N*N*sizeof(int)];
for (int i=0; i < N; i++) {
for (int j=0; j < N; j++) {
A[i*N +j] = 1;
B[i*N +j] = 1;
A = new int[N * N * sizeof(int)];
B = new int[N * N * sizeof(int)];
for (int i = 0; i < N; i++) {
for (int j = 0; j < N; j++) {
A[i * N + j] = 1;
B[i * N + j] = 1;
}
}
HIP_CHECK(hipStreamCreate(&stream1));
HIP_CHECK(hipStreamCreate(&stream2));
HIP_CHECK(hipMalloc(&Ad, SIZE*sizeof(int)));
HIP_CHECK(hipMalloc(&Bd, SIZE*sizeof(int)));
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&C), SIZE*sizeof(int)));
HIP_CHECK(hipMemcpy(Ad, A, SIZE*sizeof(int), hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(Bd, B, SIZE*sizeof(int), hipMemcpyHostToDevice));
HIP_CHECK(hipMalloc(&Ad, SIZE * sizeof(int)));
HIP_CHECK(hipMalloc(&Bd, SIZE * sizeof(int)));
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&C), SIZE * sizeof(int)));
HIP_CHECK(hipMemcpy(Ad, A, SIZE * sizeof(int), hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(Bd, B, SIZE * sizeof(int), hipMemcpyHostToDevice));
int clkRate = 0;
#if HT_AMD
#if HT_AMD
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeWallClockRate, 0));
#endif
#if HT_NVIDIA
#endif
#if HT_NVIDIA
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeClockRate, 0));
#endif
#endif
args1._Ad = Ad;
args1._Bd = Bd;
args1._Cd = C;
args1._n = N;
args1._n = N;
args1.clockRate = clkRate;
args2._Ad = NULL;
args2._Bd = NULL;
args2._Cd = NULL;
args2._n = 0;
args2._n = 0;
args2.clockRate = clkRate;
size1 = sizeof(args1);
size2 = sizeof(args2);
@@ -394,16 +387,14 @@ void ModuleLaunchKernel::AllocateMemory() {
HIP_CHECK(hipEventCreate(&end_event1));
HIP_CHECK(hipEventCreate(&start_event2));
HIP_CHECK(hipEventCreate(&end_event2));
HIP_CHECK(hipEventCreateWithFlags(&start_timingDisabled,
hipEventDisableTiming));
HIP_CHECK(hipEventCreateWithFlags(&end_timingDisabled,
hipEventDisableTiming));
HIP_CHECK(hipEventCreateWithFlags(&start_timingDisabled, hipEventDisableTiming));
HIP_CHECK(hipEventCreateWithFlags(&end_timingDisabled, hipEventDisableTiming));
}
void ModuleLaunchKernel::ModuleLoad() {
constexpr auto matmulName = "matmul.code";
constexpr auto matmulK = "matmulK";
constexpr auto SixteenSec = "SixteenSecKernel";
constexpr auto matmulK = "matmulK";
constexpr auto SixteenSec = "SixteenSecKernel";
constexpr auto KernelandExtra = "KernelandExtraParams";
constexpr auto FourSec = "FourSecKernel";
constexpr auto TwoSec = "TwoSecKernel";
@@ -413,13 +404,11 @@ void ModuleLaunchKernel::ModuleLoad() {
HIP_CHECK(hipModuleLoad(&Module, matmulName));
HIP_CHECK(hipModuleGetFunction(&MultKernel, Module, matmulK));
HIP_CHECK(hipModuleGetFunction(&SixteenSecKernel, Module, SixteenSec));
HIP_CHECK(hipModuleGetFunction(&KernelandExtraParamKernel,
Module, KernelandExtra));
HIP_CHECK(hipModuleGetFunction(&KernelandExtraParamKernel, Module, KernelandExtra));
HIP_CHECK(hipModuleGetFunction(&FourSecKernel, Module, FourSec));
HIP_CHECK(hipModuleGetFunction(&TwoSecKernel, Module, TwoSec));
HIP_CHECK(hipModuleGetFunction(&DummyKernel, Module, dummyKernel));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize,
Module, globalDevVar));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, globalDevVar));
}
void ModuleLaunchKernel::DeAllocateMemory() {
@@ -452,15 +441,14 @@ bool ModuleLaunchKernel::ExtModule_KernelExecutionTime() {
ModuleLoad();
float time_4sec, time_2sec;
void *config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
HIP_LAUNCH_PARAM_END};
HIP_CHECK(hipExtModuleLaunchKernel(FourSecKernel, 1, 1, 1, 1, 1, 1, 0,
stream1, NULL, reinterpret_cast<void**>(&config2),
start_event1, end_event1, 0));
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
NULL, reinterpret_cast<void**>(&config2),
start_event2, end_event2, 0));
HIP_CHECK(hipExtModuleLaunchKernel(FourSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config2), start_event1, end_event1,
0));
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config2), start_event2, end_event2,
0));
HIP_CHECK(hipStreamSynchronize(stream1));
HIP_CHECK(hipEventElapsedTime(&time_4sec, start_event1, end_event1));
HIP_CHECK(hipEventElapsedTime(&time_2sec, start_event2, end_event2));
@@ -484,12 +472,11 @@ bool ModuleLaunchKernel::ExtModule_Disabled_Timingflag() {
ModuleLoad();
hipError_t e;
float time_2sec;
void *config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
HIP_LAUNCH_PARAM_END};
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
NULL, reinterpret_cast<void**>(&config2),
start_timingDisabled, end_timingDisabled, 0));
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
HIP_LAUNCH_PARAM_END};
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config2), start_timingDisabled,
end_timingDisabled, 0));
HIP_CHECK(hipStreamSynchronize(stream1));
e = hipEventElapsedTime(&time_2sec, start_timingDisabled, end_timingDisabled);
if (e == hipErrorInvalidHandle) {
@@ -516,18 +503,16 @@ bool ModuleLaunchKernel::ExtModule_ConcurencyCheck_GlobalVar(int conc_flag) {
int deviceGlobal_h = 0;
AllocateMemory();
ModuleLoad();
void *config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
HIP_LAUNCH_PARAM_END};
HIP_CHECK(hipExtModuleLaunchKernel(FourSecKernel, 1, 1, 1, 1, 1, 1, 0,
stream1, NULL, reinterpret_cast<void**>(&config2),
start_event1, end_event1, conc_flag));
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
NULL, reinterpret_cast<void**>(&config2),
start_event2, end_event2, conc_flag));
HIP_CHECK(hipExtModuleLaunchKernel(FourSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config2), start_event1, end_event1,
conc_flag));
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config2), start_event2, end_event2,
conc_flag));
HIP_CHECK(hipStreamSynchronize(stream1));
HIP_CHECK(hipMemcpyDtoH(&deviceGlobal_h, hipDeviceptr_t(deviceGlobal),
deviceGlobalSize));
HIP_CHECK(hipMemcpyDtoH(&deviceGlobal_h, hipDeviceptr_t(deviceGlobal), deviceGlobalSize));
if (conc_flag && deviceGlobal_h != 0x5555) {
testStatus = true;
} else if (!conc_flag && deviceGlobal_h == 0x5555) {
@@ -550,45 +535,32 @@ bool ModuleLaunchKernel::ExtModule_ConcurrencyCheck_TimeVer() {
AllocateMemory();
ModuleLoad();
int mismatch = 0;
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
HIP_LAUNCH_PARAM_END};
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
HIP_LAUNCH_PARAM_END};
auto start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipExtModuleLaunchKernel(SixteenSecKernel, 1, 1, 1, 1, 1, 1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config2),
NULL, NULL, 0));
HIP_CHECK(hipExtModuleLaunchKernel(MultKernel, N, N, 1, 32, 32 , 1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
NULL, NULL, 0));
HIP_CHECK(hipExtModuleLaunchKernel(SixteenSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config2), NULL, NULL, 0));
HIP_CHECK(hipExtModuleLaunchKernel(MultKernel, N, N, 1, 32, 32, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), NULL, NULL, 0));
HIP_CHECK(hipStreamSynchronize(stream1));
auto stop = std::chrono::high_resolution_clock::now();
auto duration1 = std::chrono::duration_cast<std::chrono::microseconds>
(stop-start);
auto duration1 = std::chrono::duration_cast<std::chrono::microseconds>(stop - start);
start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipExtModuleLaunchKernel(SixteenSecKernel, 1, 1, 1, 1, 1, 1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config2),
NULL, NULL, 1));
HIP_CHECK(hipExtModuleLaunchKernel(MultKernel, N, N, 1, 32, 32, 1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
NULL, NULL, 1));
HIP_CHECK(hipExtModuleLaunchKernel(SixteenSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config2), NULL, NULL, 1));
HIP_CHECK(hipExtModuleLaunchKernel(MultKernel, N, N, 1, 32, 32, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), NULL, NULL, 1));
HIP_CHECK(hipStreamSynchronize(stream1));
stop = std::chrono::high_resolution_clock::now();
auto duration2 = std::chrono::duration_cast<std::chrono::microseconds>
(stop-start);
auto duration2 = std::chrono::duration_cast<std::chrono::microseconds>(stop - start);
if (!(duration2.count() < duration1.count())) {
testStatus = false;
}
for (int i = 0; i < N; i++) {
for (int j = 0; j < N; j++) {
if (C[i*N + j] != N)
mismatch++;
if (C[i * N + j] != N) mismatch++;
}
}
if (mismatch) {
@@ -603,84 +575,57 @@ bool ModuleLaunchKernel::ExtModule_Negative_tests() {
hipError_t err;
AllocateMemory();
ModuleLoad();
void *config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
HIP_LAUNCH_PARAM_END};
void *params[] = {Ad};
void* params[] = {Ad};
// Passing nullptr to kernel function in hipExtModuleLaunchKernel API
err = hipExtModuleLaunchKernel(nullptr, 1, 1, 1, 1, 1, 1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(nullptr, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed nullptr to kernel function");
testStatus = false;
}
// Passing Max int value to block dimensions
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, std::numeric_limits<uint32_t>::max(),
std::numeric_limits<uint32_t>::max(),
std::numeric_limits<uint32_t>::max(),
std::numeric_limits<uint32_t>::max(), 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
std::numeric_limits<uint32_t>::max(), 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for max values to block dimension");
testStatus = false;
}
// Passing 0 as value for all dimensions
err = hipExtModuleLaunchKernel(MultKernel, 0, 0, 0,
0,
0,
0, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(MultKernel, 0, 0, 0, 0, 0, 0, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for 0 as value for all dimensions");
testStatus = false;
}
// Passing 0 as value for x dimension
err = hipExtModuleLaunchKernel(MultKernel, 0, 1, 1,
0,
1,
1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(MultKernel, 0, 1, 1, 0, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for 0 as value for x dimension");
testStatus = false;
}
// Passing 0 as value for y dimension
err = hipExtModuleLaunchKernel(MultKernel, 1, 0, 1,
1,
0,
1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(MultKernel, 1, 0, 1, 1, 0, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for 0 as value for y dimension");
testStatus = false;
}
// Passing 0 as value for z dimension
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 0,
1,
1,
0, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 0, 1, 1, 0, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for 0 as value for z dimension");
testStatus = false;
}
// Passing both kernel and extra params
err = hipExtModuleLaunchKernel(KernelandExtraParamKernel, 1, 1, 1, 1, 1, 1, 0,
stream1, reinterpret_cast<void**>(&params),
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(KernelandExtraParamKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
reinterpret_cast<void**>(&params),
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel fail when we pass both kernel,extra args");
testStatus = false;
@@ -688,58 +633,44 @@ bool ModuleLaunchKernel::ExtModule_Negative_tests() {
// Passing more than maxthreadsperblock to block dimensions
hipDeviceProp_t deviceProp;
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
deviceProp.maxThreadsPerBlock+1,
deviceProp.maxThreadsPerBlock+1,
deviceProp.maxThreadsPerBlock+1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, deviceProp.maxThreadsPerBlock + 1,
deviceProp.maxThreadsPerBlock + 1,
deviceProp.maxThreadsPerBlock + 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for max group size");
testStatus = false;
}
// Block dimension X = Max Allowed + 1
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
deviceProp.maxThreadsDim[0]+1,
1,
1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, deviceProp.maxThreadsDim[0] + 1, 1, 1, 0,
stream1, NULL, reinterpret_cast<void**>(&config1), nullptr,
nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for (MaxBlockDimX + 1)");
testStatus = false;
}
// Block dimension Y = Max Allowed + 1
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
1,
deviceProp.maxThreadsDim[1]+1,
1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, 1, deviceProp.maxThreadsDim[1] + 1, 1, 0,
stream1, NULL, reinterpret_cast<void**>(&config1), nullptr,
nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for (MaxBlockDimY + 1)");
testStatus = false;
}
// Block dimension Z = Max Allowed + 1
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
1,
1,
deviceProp.maxThreadsDim[2]+1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, 1, 1, deviceProp.maxThreadsDim[2] + 1, 0,
stream1, NULL, reinterpret_cast<void**>(&config1), nullptr,
nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for (MaxBlockDimZ + 1)");
testStatus = false;
}
// Passing invalid config data in extra params
void *config3[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
void* config3[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
HIP_LAUNCH_PARAM_END};
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config3),
nullptr, nullptr, 0);
reinterpret_cast<void**>(&config3), nullptr, nullptr, 0);
if (err == hipSuccess) {
INFO("hipExtModuleLaunchKernel failed for invalid conf");
testStatus = false;
@@ -754,8 +685,7 @@ bool ModuleLaunchKernel::ExtModule_Corner_tests() {
hipError_t err;
AllocateMemory();
ModuleLoad();
void *config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args3,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size3,
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args3, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size3,
HIP_LAUNCH_PARAM_END};
hipDeviceProp_t deviceProp;
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
@@ -765,25 +695,14 @@ bool ModuleLaunchKernel::ExtModule_Corner_tests() {
unsigned int maxgridX = deviceProp.maxGridSize[0];
unsigned int maxgridY = deviceProp.maxGridSize[1];
unsigned int maxgridZ = deviceProp.maxGridSize[2];
struct gridblockDim test[6] = {{1, 1, 1, maxblockX, 1, 1},
{1, 1, 1, 1, maxblockY, 1},
{1, 1, 1, 1, 1, maxblockZ},
{maxgridX, 1, 1, 1, 1, 1},
{1, maxgridY, 1, 1, 1, 1},
{1, 1, maxgridZ, 1, 1, 1}};
struct gridblockDim test[6] = {{1, 1, 1, maxblockX, 1, 1}, {1, 1, 1, 1, maxblockY, 1},
{1, 1, 1, 1, 1, maxblockZ}, {maxgridX, 1, 1, 1, 1, 1},
{1, maxgridY, 1, 1, 1, 1}, {1, 1, maxgridZ, 1, 1, 1}};
for (int i = 0; i < 6; i++) {
err = hipExtModuleLaunchKernel(DummyKernel,
test[i].gridX,
test[i].gridY,
test[i].gridZ,
test[i].blockX,
test[i].blockY,
test[i].blockZ,
0,
stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(DummyKernel, test[i].gridX, test[i].gridY, test[i].gridZ,
test[i].blockX, test[i].blockY, test[i].blockZ, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err != hipSuccess) {
testStatus = false;
}
@@ -798,34 +717,26 @@ bool ModuleLaunchKernel::Module_WorkGroup_Test() {
hipError_t err;
AllocateMemory();
ModuleLoad();
void *config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args3,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size3,
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args3, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size3,
HIP_LAUNCH_PARAM_END};
hipDeviceProp_t deviceProp;
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
double cuberootVal =
cbrt(static_cast<double>(deviceProp.maxThreadsPerBlock));
double cuberootVal = cbrt(static_cast<double>(deviceProp.maxThreadsPerBlock));
uint32_t cuberoot_floor = floor(cuberootVal);
uint32_t cuberoot_ceil = ceil(cuberootVal);
// Scenario: (block.x * block.y * block.z) <= Work Group Size where
// block.x < MaxBlockDimX , block.y < MaxBlockDimY and block.z < MaxBlockDimZ
err = hipExtModuleLaunchKernel(DummyKernel,
1, 1, 1,
cuberoot_floor, cuberoot_floor, cuberoot_floor,
0, stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(DummyKernel, 1, 1, 1, cuberoot_floor, cuberoot_floor,
cuberoot_floor, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err != hipSuccess) {
testStatus = false;
}
// Scenario: (block.x * block.y * block.z) > Work Group Size where
// block.x < MaxBlockDimX , block.y < MaxBlockDimY and block.z < MaxBlockDimZ
err = hipExtModuleLaunchKernel(DummyKernel,
1, 1, 1,
cuberoot_ceil, cuberoot_ceil, cuberoot_ceil + 1,
0, stream1, NULL,
reinterpret_cast<void**>(&config1),
nullptr, nullptr, 0);
err = hipExtModuleLaunchKernel(DummyKernel, 1, 1, 1, cuberoot_ceil, cuberoot_ceil,
cuberoot_ceil + 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
if (err == hipSuccess) {
testStatus = false;
}
@@ -862,6 +773,6 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_Functional") {
}
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+7 -8
ファイルの表示
@@ -20,12 +20,12 @@ THE SOFTWARE.
#include <hip_test_defgroups.hh>
/**
* @addtogroup hipFuncGetAttributes
* @{
* @ingroup ModuleTest
* `hipError_t hipFuncGetAttributes(struct hipFuncAttributes* attr, const void* func)` -
* Find out attributes for a given function
*/
* @addtogroup hipFuncGetAttributes
* @{
* @ingroup ModuleTest
* `hipError_t hipFuncGetAttributes(struct hipFuncAttributes* attr, const void* func)` -
* Find out attributes for a given function
*/
/**
* Test Description
@@ -48,8 +48,7 @@ __global__ void getAttrFn(float* px, float* py) {
TEST_CASE("Unit_hipFuncGetAttributes_basic") {
hipFuncAttributes attr{};
auto r = hipFuncGetAttributes(&attr,
reinterpret_cast<const void*>(&getAttrFn));
auto r = hipFuncGetAttributes(&attr, reinterpret_cast<const void*>(&getAttrFn));
REQUIRE(r == hipSuccess);
REQUIRE(attr.maxThreadsPerBlock != 0);
}
+8 -10
ファイルの表示
@@ -20,12 +20,12 @@ THE SOFTWARE.
#include <hip_test_defgroups.hh>
/**
* @addtogroup hipFuncSetAttribute
* @{
* @ingroup ModuleTest
* `hipError_t hipFuncSetAttribute(const void* func, hipFuncAttribute attr, int value)` -
* Set attributes for a specific function
*/
* @addtogroup hipFuncSetAttribute
* @{
* @ingroup ModuleTest
* `hipError_t hipFuncSetAttribute(const void* func, hipFuncAttribute attr, int value)` -
* Set attributes for a specific function
*/
/**
* Test Description
@@ -47,9 +47,7 @@ __global__ void fn(float* px, float* py) {
TEST_CASE("Unit_hipFuncSetAttribute_Basic") {
HIP_CHECK(hipFuncSetAttribute(reinterpret_cast<const void*>(&fn),
hipFuncAttributeMaxDynamicSharedMemorySize,
0));
hipFuncAttributeMaxDynamicSharedMemorySize, 0));
HIP_CHECK(hipFuncSetAttribute(reinterpret_cast<const void*>(&fn),
hipFuncAttributePreferredSharedMemoryCarveout,
0));
hipFuncAttributePreferredSharedMemoryCarveout, 0));
}
+13 -13
ファイルの表示
@@ -19,7 +19,7 @@ THE SOFTWARE.
#include <hip_test_common.hh>
#include <hip_test_defgroups.hh>
__global__ void ReverseSeq(int *A, int *B, int N) {
__global__ void ReverseSeq(int* A, int* B, int N) {
extern __shared__ int SMem[];
int offset = threadIdx.x;
int MirrorVal = N - offset - 1;
@@ -28,12 +28,12 @@ __global__ void ReverseSeq(int *A, int *B, int N) {
B[offset] = SMem[MirrorVal];
}
/**
* @addtogroup hipFuncSetSharedMemConfig
* @{
* @ingroup ModuleTest
* `hipError_t hipFuncSetSharedMemConfig(const void* func, hipSharedMemConfig config)` -
* Sets shared memory configuation for a specific function
*/
* @addtogroup hipFuncSetSharedMemConfig
* @{
* @ingroup ModuleTest
* `hipError_t hipFuncSetSharedMemConfig(const void* func, hipSharedMemConfig config)` -
* Sets shared memory configuation for a specific function
*/
/**
* Test Description
@@ -63,8 +63,8 @@ TEST_CASE("Unit_hipFuncSetSharedMemConfig_functional") {
// Testing hipFuncSetSharedMemConfig() with hipSharedMemBankSizeDefault flag
SECTION("Flag: hipSharedMemBankSizeDefault") {
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>
(&ReverseSeq), hipSharedMemBankSizeDefault));
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>(&ReverseSeq),
hipSharedMemBankSizeDefault));
// Kernel Launch with shared mem size of = NELMTS * sizeof(int)
ReverseSeq<<<1, NELMTS, NELMTS * sizeof(int)>>>(Ad, RAd, NELMTS);
memset(Ah, 0, NELMTS * sizeof(int));
@@ -77,8 +77,8 @@ TEST_CASE("Unit_hipFuncSetSharedMemConfig_functional") {
// Testing hipFuncSetSharedMemConfig() with hipSharedMemBankSizeFourBytes flag
SECTION("Flag: hipSharedMemBankSizeFourBytes") {
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>
(&ReverseSeq), hipSharedMemBankSizeFourByte));
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>(&ReverseSeq),
hipSharedMemBankSizeFourByte));
HIP_CHECK(hipMemset(RAd, 0, NELMTS * sizeof(int)));
// Kernel Launch with shared mem size of = NELMTS * sizeof(int)
ReverseSeq<<<1, NELMTS, NELMTS * sizeof(int)>>>(Ad, RAd, NELMTS);
@@ -91,8 +91,8 @@ TEST_CASE("Unit_hipFuncSetSharedMemConfig_functional") {
}
// Testing hipFuncSetSharedMemConfig() with hipSharedMemBankSizeEightBytes flg
SECTION("Flag: hipSharedMemBankSizeEightByte") {
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>
(&ReverseSeq), hipSharedMemBankSizeEightByte));
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>(&ReverseSeq),
hipSharedMemBankSizeEightByte));
HIP_CHECK(hipMemset(RAd, 0, NELMTS * sizeof(int)));
// Kernel Launch with shared mem size of = NELMTS * sizeof(int)
ReverseSeq<<<1, NELMTS, NELMTS * sizeof(int)>>>(Ad, RAd, NELMTS);
+35 -47
ファイルの表示
@@ -35,17 +35,16 @@ THE SOFTWARE.
#define LEN 64
#define SIZE LEN * sizeof(float)
#define ARR_SIZE (32*32)
#define SIZE_BYTES (ARR_SIZE*sizeof(int))
#define ARR_SIZE (32 * 32)
#define SIZE_BYTES (ARR_SIZE * sizeof(int))
extern "C" __global__ void bit_extract_kernel(uint32_t* C_d, const uint32_t*
A_d, size_t N) {
extern "C" __global__ void bit_extract_kernel(uint32_t* C_d, const uint32_t* A_d, size_t N) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
for (size_t i = offset; i < N; i += stride) {
#if HT_AMD
C_d[i] = __bitextract_u32(A_d[i], 8, 4);
#else /* defined __HIP_PLATFORM_NVIDIA__ or other path */
#else /* defined __HIP_PLATFORM_NVIDIA__ or other path */
C_d[i] = ((A_d[i] & 0xf00) >> 8);
#endif
}
@@ -54,18 +53,16 @@ extern "C" __global__ void bit_extract_kernel(uint32_t* C_d, const uint32_t*
/**
* Host Function to check for negative case.
*/
__host__ void hostFunction() {
printf("hostFunction\n");
}
__host__ void hostFunction() { printf("hostFunction\n"); }
/**
* Sample Kernel to be used for functional test cases
*/
__global__ void hipKernel(int *a) {
__global__ void hipKernel(int* a) {
int offset = blockDim.x * blockIdx.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
for (int i = offset; i < ARR_SIZE; i+= stride) {
for (int i = offset; i < ARR_SIZE; i += stride) {
a[i] += a[i];
}
}
@@ -73,7 +70,7 @@ __global__ void hipKernel(int *a) {
/**
* Local Function to validate the result
*/
bool verifyResult(int *a, int *output_ref, int arrSize) {
bool verifyResult(int* a, int* output_ref, int arrSize) {
for (int i = 0; i < arrSize; i++) {
if (a[i] != output_ref[i]) {
return false;
@@ -126,20 +123,19 @@ TEST_CASE("Unit_hipGetFuncBySymbol_PositiveTest") {
void* _Ad;
size_t _N;
} args;
args._Cd = reinterpret_cast<void**> (C_d);
args._Ad = reinterpret_cast<void**> (A_d);
args._N = static_cast<size_t> (N);
args._Cd = reinterpret_cast<void**>(C_d);
args._Ad = reinterpret_cast<void**>(A_d);
args._N = static_cast<size_t>(N);
size_t size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
hipFunction_t Function;
HIPCHECK(hipGetFuncBySymbol(&Function,
reinterpret_cast<void*>(bit_extract_kernel)));
HIPCHECK(hipGetFuncBySymbol(&Function, reinterpret_cast<void*>(bit_extract_kernel)));
HIPCHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0, 0, NULL,
reinterpret_cast<void**>(&config)));
reinterpret_cast<void**>(&config)));
HIPCHECK(hipMemcpyDtoH(C_h, (hipDeviceptr_t)(C_d), Nbytes));
@@ -177,8 +173,7 @@ TEST_CASE("Unit_hipGetFuncBySymbol_NegativeTests") {
REQUIRE(hipGetFuncBySymbol(&funcPointer, NULL) != hipSuccess);
// Passing hostFunction as second parameter
REQUIRE(hipGetFuncBySymbol(&funcPointer,
reinterpret_cast<const void*>(hostFunction)));
REQUIRE(hipGetFuncBySymbol(&funcPointer, reinterpret_cast<const void*>(hostFunction)));
}
/**
@@ -225,12 +220,12 @@ TEST_CASE("Unit_hipGetFuncBySymbol_MultiDev") {
for (int deviceId = 0; deviceId < deviceCount; deviceId++) {
HIP_CHECK(hipSetDevice(deviceId));
REQUIRE(hipGetFuncBySymbol(&funcPointer,
reinterpret_cast<const void*>(hipKernel))== hipSuccess);
REQUIRE(hipGetFuncBySymbol(&funcPointer, reinterpret_cast<const void*>(hipKernel)) ==
hipSuccess);
int *h_a = reinterpret_cast<int *>(malloc(SIZE_BYTES));
int* h_a = reinterpret_cast<int*>(malloc(SIZE_BYTES));
REQUIRE(h_a != nullptr);
int *output_ref = reinterpret_cast<int *>(malloc(SIZE_BYTES));
int* output_ref = reinterpret_cast<int*>(malloc(SIZE_BYTES));
REQUIRE(output_ref != nullptr);
for (int i = 0; i < ARR_SIZE; i++) {
@@ -238,7 +233,7 @@ TEST_CASE("Unit_hipGetFuncBySymbol_MultiDev") {
output_ref[i] = 4;
}
int *d_a = nullptr;
int* d_a = nullptr;
HIP_CHECK(hipMalloc(&d_a, SIZE_BYTES));
REQUIRE(d_a != nullptr);
HIP_CHECK(hipMemcpy(d_a, h_a, SIZE_BYTES, hipMemcpyHostToDevice));
@@ -249,13 +244,11 @@ TEST_CASE("Unit_hipGetFuncBySymbol_MultiDev") {
void* kernelParam[] = {d_a};
auto size = sizeof(kernelParam);
void* kernel_parameter[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
REQUIRE(hipModuleLaunchKernel(funcPointer,
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
0, 0, nullptr, kernel_parameter) == hipSuccess);
REQUIRE(hipModuleLaunchKernel(funcPointer, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z, 0, 0,
nullptr, kernel_parameter) == hipSuccess);
HIP_CHECK(hipMemcpy(h_a, d_a, SIZE_BYTES, hipMemcpyDeviceToHost));
@@ -273,9 +266,9 @@ TEST_CASE("Unit_hipGetFuncBySymbol_MultiDev") {
void MultiThreadMultiDevFunc(int DevId) {
HIP_CHECK(hipSetDevice(DevId));
int *h_a = reinterpret_cast<int *>(malloc(SIZE_BYTES));
int* h_a = reinterpret_cast<int*>(malloc(SIZE_BYTES));
REQUIRE(h_a != nullptr);
int *output_ref = reinterpret_cast<int *>(malloc(SIZE_BYTES));
int* output_ref = reinterpret_cast<int*>(malloc(SIZE_BYTES));
REQUIRE(output_ref != nullptr);
for (int i = 0; i < ARR_SIZE; i++) {
@@ -287,32 +280,27 @@ void MultiThreadMultiDevFunc(int DevId) {
HIP_CHECK(hipSetDevice(DevId));
HIP_CHECK(hipStreamCreate(&stream));
int *d_a = nullptr;
int* d_a = nullptr;
HIP_CHECK(hipMalloc(&d_a, SIZE_BYTES));
REQUIRE(d_a != nullptr);
HIP_CHECK(hipMemcpyAsync(d_a, h_a, SIZE_BYTES,
hipMemcpyHostToDevice, stream));
HIP_CHECK(hipMemcpyAsync(d_a, h_a, SIZE_BYTES, hipMemcpyHostToDevice, stream));
dim3 blocksPerGrid(1, 1, 1);
dim3 threadsPerBlock(1, 1, 64);
hipFunction_t funcPointer;
REQUIRE(hipGetFuncBySymbol(&funcPointer,
reinterpret_cast<const void*>(hipKernel))== hipSuccess);
REQUIRE(hipGetFuncBySymbol(&funcPointer, reinterpret_cast<const void*>(hipKernel)) == hipSuccess);
void* kernelParam[] = {d_a};
auto size = sizeof(kernelParam);
void* kernel_parameter[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
REQUIRE(hipModuleLaunchKernel(funcPointer,
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
0, stream, nullptr, kernel_parameter) == hipSuccess);
REQUIRE(hipModuleLaunchKernel(funcPointer, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z, 0, stream,
nullptr, kernel_parameter) == hipSuccess);
HIP_CHECK(hipMemcpyAsync(h_a, d_a, SIZE_BYTES,
hipMemcpyDeviceToHost, stream));
HIP_CHECK(hipMemcpyAsync(h_a, d_a, SIZE_BYTES, hipMemcpyDeviceToHost, stream));
REQUIRE(verifyResult(h_a, output_ref, ARR_SIZE) == true);
+23 -28
ファイルの表示
@@ -19,29 +19,27 @@ THE SOFTWARE.
#include "hip/hip_runtime.h"
#define ARR_SIZE (32*32)
#define SIZE (ARR_SIZE*sizeof(int))
#define ARR_SIZE (32 * 32)
#define SIZE (ARR_SIZE * sizeof(int))
#define HIP_CHECK(error) \
{ \
hipError_t localError = error; \
if ((localError != hipSuccess) && \
(localError != hipErrorPeerAccessAlreadyEnabled)) { \
printf("error: '%s'(%d) from %s at %s:%d\n", \
hipGetErrorString(localError), \
localError, #error, __FUNCTION__, __LINE__);\
exit(0); \
} \
}
#define HIP_CHECK(error) \
{ \
hipError_t localError = error; \
if ((localError != hipSuccess) && (localError != hipErrorPeerAccessAlreadyEnabled)) { \
printf("error: '%s'(%d) from %s at %s:%d\n", hipGetErrorString(localError), localError, \
#error, __FUNCTION__, __LINE__); \
exit(0); \
} \
}
/**
* Sample Kernel to be used for functional test cases
*/
__global__ void hipKernel(int *a) {
__global__ void hipKernel(int* a) {
int offset = blockDim.x * blockIdx.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
for (int i = offset; i < ARR_SIZE; i+= stride) {
for (int i = offset; i < ARR_SIZE; i += stride) {
a[i] += a[i];
}
}
@@ -53,17 +51,16 @@ __global__ void hipKernel(int *a) {
int main() {
hipFunction_t funcPointer;
if (hipGetFuncBySymbol(&funcPointer,
reinterpret_cast<const void*>(hipKernel)) != hipSuccess) {
return -1;
if (hipGetFuncBySymbol(&funcPointer, reinterpret_cast<const void*>(hipKernel)) != hipSuccess) {
return -1;
}
int *h_a = reinterpret_cast<int *>(malloc(SIZE));
int* h_a = reinterpret_cast<int*>(malloc(SIZE));
if (h_a == nullptr) {
return -1;
}
int *output_ref = reinterpret_cast<int *>(malloc(SIZE));
int* output_ref = reinterpret_cast<int*>(malloc(SIZE));
if (output_ref == nullptr) {
return -1;
}
@@ -73,7 +70,7 @@ int main() {
output_ref[i] = 4;
}
int *d_a = nullptr;
int* d_a = nullptr;
HIP_CHECK(hipMalloc(&d_a, SIZE));
if (d_a == nullptr) {
return -1;
@@ -86,14 +83,12 @@ int main() {
void* kernelParam[] = {d_a};
auto size = sizeof(kernelParam);
void* kernel_parameter[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
if (hipModuleLaunchKernel(funcPointer,
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
0, 0, nullptr, kernel_parameter) != hipSuccess) {
return -1;
if (hipModuleLaunchKernel(funcPointer, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z, 0, 0, nullptr,
kernel_parameter) != hipSuccess) {
return -1;
}
HIP_CHECK(hipMemcpy(h_a, d_a, SIZE, hipMemcpyDeviceToHost));
+247 -345
ファイルの表示
@@ -53,103 +53,65 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
int currentHipVersion = 0;
HIP_CHECK(hipRuntimeGetVersion(&currentHipVersion));
HIP_CHECK(hipGetProcAddress("hipModuleLoad",
&hipModuleLoad_ptr,
HIP_CHECK(hipGetProcAddress("hipModuleLoad", &hipModuleLoad_ptr, currentHipVersion, 0, nullptr));
HIP_CHECK(
hipGetProcAddress("hipModuleUnload", &hipModuleUnload_ptr, currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleGetFunction", &hipModuleGetFunction_ptr, currentHipVersion,
0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleLaunchKernel", &hipModuleLaunchKernel_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleUnload",
&hipModuleUnload_ptr,
HIP_CHECK(hipGetProcAddress("hipGetFuncBySymbol", &hipGetFuncBySymbol_ptr, currentHipVersion, 0,
nullptr));
HIP_CHECK(hipGetProcAddress("hipFuncGetAttributes", &hipFuncGetAttributes_ptr, currentHipVersion,
0, nullptr));
HIP_CHECK(hipGetProcAddress("hipFuncGetAttribute", &hipFuncGetAttribute_ptr, currentHipVersion, 0,
nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleGetGlobal", &hipModuleGetGlobal_ptr, currentHipVersion, 0,
nullptr));
HIP_CHECK(hipGetProcAddress("hipExtModuleLaunchKernel", &hipExtModuleLaunchKernel_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleGetFunction",
&hipModuleGetFunction_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleLaunchKernel",
&hipModuleLaunchKernel_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipGetFuncBySymbol",
&hipGetFuncBySymbol_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipFuncGetAttributes",
&hipFuncGetAttributes_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipFuncGetAttribute",
&hipFuncGetAttribute_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleGetGlobal",
&hipModuleGetGlobal_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipExtModuleLaunchKernel",
&hipExtModuleLaunchKernel_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipHccModuleLaunchKernel",
&hipHccModuleLaunchKernel_ptr,
HIP_CHECK(hipGetProcAddress("hipHccModuleLaunchKernel", &hipHccModuleLaunchKernel_ptr,
currentHipVersion, 0, nullptr));
hipError_t (*dyn_hipModuleLoad_ptr)(hipModule_t *, const char *) =
reinterpret_cast<hipError_t (*)(hipModule_t *, const char *)>
(hipModuleLoad_ptr);
hipError_t (*dyn_hipModuleLoad_ptr)(hipModule_t*, const char*) =
reinterpret_cast<hipError_t (*)(hipModule_t*, const char*)>(hipModuleLoad_ptr);
hipError_t (*dyn_hipModuleUnload_ptr)(hipModule_t) =
reinterpret_cast<hipError_t (*)(hipModule_t)>
(hipModuleUnload_ptr);
hipError_t (*dyn_hipModuleGetFunction_ptr)(
hipFunction_t *, hipModule_t, const char *) =
reinterpret_cast<hipError_t (*)(hipFunction_t *,
hipModule_t, const char *)>
(hipModuleGetFunction_ptr);
reinterpret_cast<hipError_t (*)(hipModule_t)>(hipModuleUnload_ptr);
hipError_t (*dyn_hipModuleGetFunction_ptr)(hipFunction_t*, hipModule_t, const char*) =
reinterpret_cast<hipError_t (*)(hipFunction_t*, hipModule_t, const char*)>(
hipModuleGetFunction_ptr);
hipError_t (*dyn_hipModuleLaunchKernel_ptr)(
hipFunction_t,
unsigned int, unsigned int, unsigned int,
unsigned int, unsigned int, unsigned int,
unsigned int, hipStream_t,
void **, void **) =
reinterpret_cast<hipError_t (*)(hipFunction_t,
unsigned int, unsigned int, unsigned int,
unsigned int, unsigned int, unsigned int,
unsigned int, hipStream_t,
void **, void **) > (hipModuleLaunchKernel_ptr);
hipError_t (*dyn_hipGetFuncBySymbol_ptr)(hipFunction_t *, const void *) =
reinterpret_cast<hipError_t (*)(hipFunction_t *, const void *)>
(hipGetFuncBySymbol_ptr);
hipError_t (*dyn_hipFuncGetAttributes_ptr)(
struct hipFuncAttributes *, const void *) =
reinterpret_cast<hipError_t (*)(struct hipFuncAttributes *, const void *)>
(hipFuncGetAttributes_ptr);
hipError_t (*dyn_hipFuncGetAttribute_ptr)(
int *, hipFunction_attribute, hipFunction_t) =
reinterpret_cast<hipError_t (*)(int *, hipFunction_attribute,
hipFunction_t)>(hipFuncGetAttribute_ptr);
hipError_t (*dyn_hipModuleGetGlobal_ptr)(
hipDeviceptr_t *, size_t *, hipModule_t, const char *) =
reinterpret_cast<hipError_t (*)(hipDeviceptr_t *, size_t *,
hipModule_t, const char *)>
(hipModuleGetGlobal_ptr);
hipFunction_t, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int,
unsigned int, unsigned int, hipStream_t, void**, void**) =
reinterpret_cast<hipError_t (*)(hipFunction_t, unsigned int, unsigned int, unsigned int,
unsigned int, unsigned int, unsigned int, unsigned int,
hipStream_t, void**, void**)>(hipModuleLaunchKernel_ptr);
hipError_t (*dyn_hipGetFuncBySymbol_ptr)(hipFunction_t*, const void*) =
reinterpret_cast<hipError_t (*)(hipFunction_t*, const void*)>(hipGetFuncBySymbol_ptr);
hipError_t (*dyn_hipFuncGetAttributes_ptr)(struct hipFuncAttributes*, const void*) =
reinterpret_cast<hipError_t (*)(struct hipFuncAttributes*, const void*)>(
hipFuncGetAttributes_ptr);
hipError_t (*dyn_hipFuncGetAttribute_ptr)(int*, hipFunction_attribute, hipFunction_t) =
reinterpret_cast<hipError_t (*)(int*, hipFunction_attribute, hipFunction_t)>(
hipFuncGetAttribute_ptr);
hipError_t (*dyn_hipModuleGetGlobal_ptr)(hipDeviceptr_t*, size_t*, hipModule_t, const char*) =
reinterpret_cast<hipError_t (*)(hipDeviceptr_t*, size_t*, hipModule_t, const char*)>(
hipModuleGetGlobal_ptr);
hipError_t (*dyn_hipExtModuleLaunchKernel_ptr)(hipFunction_t,
uint32_t, uint32_t, uint32_t,
uint32_t, uint32_t, uint32_t,
size_t, hipStream_t,
void **, void **,
hipEvent_t, hipEvent_t, uint32_t) =
reinterpret_cast<hipError_t (*)(hipFunction_t,
uint32_t, uint32_t, uint32_t,
uint32_t, uint32_t, uint32_t,
size_t, hipStream_t,
void **, void **,
hipEvent_t, hipEvent_t, uint32_t)>
(hipExtModuleLaunchKernel_ptr);
hipError_t (*dyn_hipExtModuleLaunchKernel_ptr)(hipFunction_t, uint32_t, uint32_t, uint32_t,
uint32_t, uint32_t, uint32_t, size_t, hipStream_t,
void**, void**, hipEvent_t, hipEvent_t, uint32_t) =
reinterpret_cast<hipError_t (*)(hipFunction_t, uint32_t, uint32_t, uint32_t, uint32_t,
uint32_t, uint32_t, size_t, hipStream_t, void**, void**,
hipEvent_t, hipEvent_t, uint32_t)>(
hipExtModuleLaunchKernel_ptr);
hipError_t (*dyn_hipHccModuleLaunchKernel_ptr)(hipFunction_t,
uint32_t, uint32_t, uint32_t,
uint32_t, uint32_t, uint32_t,
size_t, hipStream_t,
void **, void **,
hipEvent_t, hipEvent_t) =
reinterpret_cast<hipError_t (*)(hipFunction_t,
uint32_t, uint32_t, uint32_t,
uint32_t, uint32_t, uint32_t,
size_t, hipStream_t,
void **, void **,
hipEvent_t, hipEvent_t)>
(hipHccModuleLaunchKernel_ptr);
hipError_t (*dyn_hipHccModuleLaunchKernel_ptr)(hipFunction_t, uint32_t, uint32_t, uint32_t,
uint32_t, uint32_t, uint32_t, size_t, hipStream_t,
void**, void**, hipEvent_t, hipEvent_t) =
reinterpret_cast<hipError_t (*)(hipFunction_t, uint32_t, uint32_t, uint32_t, uint32_t,
uint32_t, uint32_t, size_t, hipStream_t, void**, void**,
hipEvent_t, hipEvent_t)>(hipHccModuleLaunchKernel_ptr);
// Validating hipModuleLoad API
hipModule_t module;
@@ -165,11 +127,11 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
const int N = 10;
const int Nbytes = 10 * sizeof(int);
int *hostArr = reinterpret_cast<int *>(malloc(Nbytes));
int* hostArr = reinterpret_cast<int*>(malloc(Nbytes));
REQUIRE(hostArr != nullptr);
fillHostArray(hostArr, N, 10);
int *devArr = nullptr;
int* devArr = nullptr;
HIP_CHECK(hipMalloc(&devArr, Nbytes));
REQUIRE(devArr != nullptr);
HIP_CHECK(hipMemcpy(devArr, hostArr, Nbytes, hipMemcpyHostToDevice));
@@ -178,7 +140,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
dim3 threadsPerBlock(1, 1, N);
struct kernelParameters {
void *arr;
void* arr;
int size;
};
kernelParameters kernelParam{};
@@ -186,48 +148,39 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
kernelParam.size = N;
auto size = sizeof(kernelParam);
void* kernel_parameter[] = { HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END };
void* kernel_parameter[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
HIP_CHECK(dyn_hipModuleLaunchKernel_ptr(function,
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
0, 0, nullptr, kernel_parameter));
HIP_CHECK(dyn_hipModuleLaunchKernel_ptr(function, blocksPerGrid.x, blocksPerGrid.y,
blocksPerGrid.z, threadsPerBlock.x, threadsPerBlock.y,
threadsPerBlock.z, 0, 0, nullptr, kernel_parameter));
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
REQUIRE(validateHostArray(hostArr, N, 12) == true);
// Validating hipExtModuleLaunchKernel API
HIP_CHECK(dyn_hipExtModuleLaunchKernel_ptr(function,
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
0, 0,
nullptr, kernel_parameter,
nullptr, nullptr, 0));
HIP_CHECK(dyn_hipExtModuleLaunchKernel_ptr(
function, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z, threadsPerBlock.x,
threadsPerBlock.y, threadsPerBlock.z, 0, 0, nullptr, kernel_parameter, nullptr, nullptr, 0));
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
REQUIRE(validateHostArray(hostArr, N, 14) == true);
// Validating hipHccModuleLaunchKernel API
HIP_CHECK(dyn_hipHccModuleLaunchKernel_ptr(function,
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
0, 0,
nullptr, kernel_parameter,
nullptr, nullptr));
HIP_CHECK(dyn_hipHccModuleLaunchKernel_ptr(
function, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z, threadsPerBlock.x,
threadsPerBlock.y, threadsPerBlock.z, 0, 0, nullptr, kernel_parameter, nullptr, nullptr));
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
REQUIRE(validateHostArray(hostArr, N, 16) == true);
// Validating hipGetFuncBySymbol API
hipFunction_t functionWithOrgApi, functionWithFuncPtr;
HIP_CHECK(hipGetFuncBySymbol(&functionWithOrgApi,
reinterpret_cast<const void*>(addOneKernel)));
HIP_CHECK(hipGetFuncBySymbol(&functionWithOrgApi, reinterpret_cast<const void*>(addOneKernel)));
REQUIRE(functionWithOrgApi != nullptr);
HIP_CHECK(dyn_hipGetFuncBySymbol_ptr(&functionWithFuncPtr,
reinterpret_cast<const void*>(addOneKernel)));
reinterpret_cast<const void*>(addOneKernel)));
REQUIRE(functionWithFuncPtr != nullptr);
REQUIRE(functionWithFuncPtr == functionWithOrgApi);
@@ -235,44 +188,38 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
// Validating hipFuncGetAttributes API
struct hipFuncAttributes attrWithOrgApi, attrWithFuncPtr;
HIP_CHECK(hipFuncGetAttributes(&attrWithOrgApi,
reinterpret_cast<const void*>(addOneKernel)));
HIP_CHECK(dyn_hipFuncGetAttributes_ptr(&attrWithFuncPtr,
reinterpret_cast<const void*>(addOneKernel)));
HIP_CHECK(hipFuncGetAttributes(&attrWithOrgApi, reinterpret_cast<const void*>(addOneKernel)));
HIP_CHECK(
dyn_hipFuncGetAttributes_ptr(&attrWithFuncPtr, reinterpret_cast<const void*>(addOneKernel)));
REQUIRE(attrWithFuncPtr.binaryVersion == attrWithOrgApi.binaryVersion);
REQUIRE(attrWithFuncPtr.cacheModeCA == attrWithOrgApi.cacheModeCA);
REQUIRE(attrWithFuncPtr.constSizeBytes == attrWithOrgApi.constSizeBytes);
REQUIRE(attrWithFuncPtr.localSizeBytes == attrWithOrgApi.localSizeBytes);
REQUIRE(attrWithFuncPtr.maxDynamicSharedSizeBytes ==
attrWithOrgApi.maxDynamicSharedSizeBytes);
REQUIRE(attrWithFuncPtr.maxThreadsPerBlock ==
attrWithOrgApi.maxThreadsPerBlock);
REQUIRE(attrWithFuncPtr.maxDynamicSharedSizeBytes == attrWithOrgApi.maxDynamicSharedSizeBytes);
REQUIRE(attrWithFuncPtr.maxThreadsPerBlock == attrWithOrgApi.maxThreadsPerBlock);
REQUIRE(attrWithFuncPtr.numRegs == attrWithOrgApi.numRegs);
REQUIRE(attrWithFuncPtr.preferredShmemCarveout ==
attrWithOrgApi.preferredShmemCarveout);
REQUIRE(attrWithFuncPtr.preferredShmemCarveout == attrWithOrgApi.preferredShmemCarveout);
REQUIRE(attrWithFuncPtr.ptxVersion == attrWithOrgApi.ptxVersion);
REQUIRE(attrWithFuncPtr.sharedSizeBytes == attrWithOrgApi.sharedSizeBytes);
// Validating hipFuncGetAttribute API
hipFunction_attribute attributes[] = {
HIP_FUNC_ATTRIBUTE_MAX_THREADS_PER_BLOCK,
HIP_FUNC_ATTRIBUTE_SHARED_SIZE_BYTES,
HIP_FUNC_ATTRIBUTE_CONST_SIZE_BYTES,
HIP_FUNC_ATTRIBUTE_LOCAL_SIZE_BYTES,
HIP_FUNC_ATTRIBUTE_NUM_REGS,
HIP_FUNC_ATTRIBUTE_PTX_VERSION,
HIP_FUNC_ATTRIBUTE_BINARY_VERSION,
HIP_FUNC_ATTRIBUTE_CACHE_MODE_CA,
HIP_FUNC_ATTRIBUTE_MAX_DYNAMIC_SHARED_SIZE_BYTES,
HIP_FUNC_ATTRIBUTE_PREFERRED_SHARED_MEMORY_CARVEOUT};
hipFunction_attribute attributes[] = {HIP_FUNC_ATTRIBUTE_MAX_THREADS_PER_BLOCK,
HIP_FUNC_ATTRIBUTE_SHARED_SIZE_BYTES,
HIP_FUNC_ATTRIBUTE_CONST_SIZE_BYTES,
HIP_FUNC_ATTRIBUTE_LOCAL_SIZE_BYTES,
HIP_FUNC_ATTRIBUTE_NUM_REGS,
HIP_FUNC_ATTRIBUTE_PTX_VERSION,
HIP_FUNC_ATTRIBUTE_BINARY_VERSION,
HIP_FUNC_ATTRIBUTE_CACHE_MODE_CA,
HIP_FUNC_ATTRIBUTE_MAX_DYNAMIC_SHARED_SIZE_BYTES,
HIP_FUNC_ATTRIBUTE_PREFERRED_SHARED_MEMORY_CARVEOUT};
for ( auto attribute : attributes ) {
for (auto attribute : attributes) {
int valuewithOrgAPI = 0, valueWithFuncPointer = 0;
HIP_CHECK(hipFuncGetAttribute(&valuewithOrgAPI, attribute, function));
HIP_CHECK(dyn_hipFuncGetAttribute_ptr(&valueWithFuncPointer, attribute,
function));
HIP_CHECK(dyn_hipFuncGetAttribute_ptr(&valueWithFuncPointer, attribute, function));
REQUIRE(valueWithFuncPointer == valuewithOrgAPI);
}
@@ -280,14 +227,13 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
// Validating hipModuleGetGlobal API
hipDeviceptr_t dptrWithOrgApi = nullptr;
size_t bytesWithOrgApi = 0;
HIP_CHECK(hipModuleGetGlobal(&dptrWithOrgApi, &bytesWithOrgApi,
module, "globalDevData"));
HIP_CHECK(hipModuleGetGlobal(&dptrWithOrgApi, &bytesWithOrgApi, module, "globalDevData"));
REQUIRE(dptrWithOrgApi != nullptr);
hipDeviceptr_t dptrWithFuncPtr = nullptr;
size_t bytesWithFuncPtr = 0;
HIP_CHECK(dyn_hipModuleGetGlobal_ptr(&dptrWithFuncPtr, &bytesWithFuncPtr,
module, "globalDevData") );
HIP_CHECK(
dyn_hipModuleGetGlobal_ptr(&dptrWithFuncPtr, &bytesWithFuncPtr, module, "globalDevData"));
REQUIRE(dptrWithFuncPtr != nullptr);
REQUIRE(bytesWithFuncPtr == 4);
@@ -323,24 +269,19 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisLoadData") {
int currentHipVersion = 0;
HIP_CHECK(hipRuntimeGetVersion(&currentHipVersion));
HIP_CHECK(hipGetProcAddress("hipModuleLoadData",
&hipModuleLoadData_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleLoadDataEx",
&hipModuleLoadDataEx_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleLoadData", &hipModuleLoadData_ptr, currentHipVersion, 0,
nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleLoadDataEx", &hipModuleLoadDataEx_ptr, currentHipVersion, 0,
nullptr));
hipError_t (*dyn_hipModuleLoadData_ptr)(hipModule_t *, const void *) =
reinterpret_cast<hipError_t (*)(hipModule_t *, const void *)>
(hipModuleLoadData_ptr);
hipError_t (*dyn_hipModuleLoadDataEx_ptr)(hipModule_t *, const void *,
unsigned int, hipJitOption *, void **) =
reinterpret_cast<hipError_t (*)(hipModule_t *, const void *,
unsigned int, hipJitOption *, void **)>
(hipModuleLoadDataEx_ptr);
hipError_t (*dyn_hipModuleLoadData_ptr)(hipModule_t*, const void*) =
reinterpret_cast<hipError_t (*)(hipModule_t*, const void*)>(hipModuleLoadData_ptr);
hipError_t (*dyn_hipModuleLoadDataEx_ptr)(hipModule_t*, const void*, unsigned int, hipJitOption*,
void**) =
reinterpret_cast<hipError_t (*)(hipModule_t*, const void*, unsigned int, hipJitOption*,
void**)>(hipModuleLoadDataEx_ptr);
const auto rtc = CreateRTCCharArray(
R"(extern "C" __global__ void simpleKernel() {})");
const auto rtc = CreateRTCCharArray(R"(extern "C" __global__ void simpleKernel() {})");
// Validating hipModuleLoadData API
{
@@ -352,9 +293,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisLoadData") {
hipFunction_t function;
HIP_CHECK(hipModuleGetFunction(&function, module, "simpleKernel"));
REQUIRE(function != nullptr);
HIP_CHECK(hipModuleLaunchKernel(function,
1, 1, 1, 1, 1, 1,
0, 0, nullptr, nullptr));
HIP_CHECK(hipModuleLaunchKernel(function, 1, 1, 1, 1, 1, 1, 0, 0, nullptr, nullptr));
HIP_CHECK(hipModuleUnload(module));
}
@@ -363,22 +302,19 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisLoadData") {
{
hipModule_t module = nullptr;
HIP_CHECK(dyn_hipModuleLoadDataEx_ptr(&module, rtc.data(),
0, nullptr, nullptr));
HIP_CHECK(dyn_hipModuleLoadDataEx_ptr(&module, rtc.data(), 0, nullptr, nullptr));
REQUIRE(module != nullptr);
hipFunction_t function;
HIP_CHECK(hipModuleGetFunction(&function, module, "simpleKernel"));
REQUIRE(function != nullptr);
HIP_CHECK(hipModuleLaunchKernel(function,
1, 1, 1, 1, 1, 1,
0, 0, nullptr, nullptr));
HIP_CHECK(hipModuleLaunchKernel(function, 1, 1, 1, 1, 1, 1, 0, 0, nullptr, nullptr));
HIP_CHECK(hipModuleUnload(module));
}
}
/**
/**
* Test Description
* ------------------------
* - This test will get the function pointer of different module management
@@ -398,77 +334,63 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
return;
}
void *hipModuleLaunchCooperativeKernel_ptr = nullptr;
void *hipModuleLaunchCooperativeKernelMultiDevice_ptr = nullptr;
void *hipLaunchCooperativeKernel_ptr = nullptr;
void *hipLaunchCooperativeKernelMultiDevice_ptr = nullptr;
void *hipExtLaunchMultiKernelMultiDevice_ptr = nullptr;
void* hipModuleLaunchCooperativeKernel_ptr = nullptr;
void* hipModuleLaunchCooperativeKernelMultiDevice_ptr = nullptr;
void* hipLaunchCooperativeKernel_ptr = nullptr;
void* hipLaunchCooperativeKernelMultiDevice_ptr = nullptr;
void* hipExtLaunchMultiKernelMultiDevice_ptr = nullptr;
int currentHipVersion = 0;
HIP_CHECK(hipRuntimeGetVersion(&currentHipVersion));
HIP_CHECK(hipGetProcAddress(
"hipModuleLaunchCooperativeKernel",
&hipModuleLaunchCooperativeKernel_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipModuleLaunchCooperativeKernelMultiDevice",
&hipModuleLaunchCooperativeKernelMultiDevice_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipLaunchCooperativeKernel",
&hipLaunchCooperativeKernel_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipLaunchCooperativeKernelMultiDevice",
&hipLaunchCooperativeKernelMultiDevice_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipExtLaunchMultiKernelMultiDevice",
&hipExtLaunchMultiKernelMultiDevice_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleLaunchCooperativeKernel",
&hipModuleLaunchCooperativeKernel_ptr, currentHipVersion, 0,
nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleLaunchCooperativeKernelMultiDevice",
&hipModuleLaunchCooperativeKernelMultiDevice_ptr, currentHipVersion,
0, nullptr));
HIP_CHECK(hipGetProcAddress("hipLaunchCooperativeKernel", &hipLaunchCooperativeKernel_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipLaunchCooperativeKernelMultiDevice",
&hipLaunchCooperativeKernelMultiDevice_ptr, currentHipVersion, 0,
nullptr));
HIP_CHECK(hipGetProcAddress("hipExtLaunchMultiKernelMultiDevice",
&hipExtLaunchMultiKernelMultiDevice_ptr, currentHipVersion, 0,
nullptr));
hipError_t (*dyn_hipModuleLaunchCooperativeKernel_ptr)(
hipFunction_t,
unsigned int, unsigned int, unsigned int,
unsigned int, unsigned int, unsigned int,
unsigned int, hipStream_t, void **) =
reinterpret_cast<hipError_t (*)(hipFunction_t,
unsigned int, unsigned int, unsigned int,
unsigned int, unsigned int, unsigned int,
unsigned int, hipStream_t, void **)>
(hipModuleLaunchCooperativeKernel_ptr);
hipFunction_t, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int,
unsigned int, unsigned int, hipStream_t, void**) =
reinterpret_cast<hipError_t (*)(hipFunction_t, unsigned int, unsigned int, unsigned int,
unsigned int, unsigned int, unsigned int, unsigned int,
hipStream_t, void**)>(hipModuleLaunchCooperativeKernel_ptr);
hipError_t (*dyn_hipModuleLaunchCooperativeKernelMultiDevice_ptr)(
hipFunctionLaunchParams *, unsigned int, unsigned int) =
reinterpret_cast<hipError_t (*)(hipFunctionLaunchParams *,
unsigned int, unsigned int)>
(hipModuleLaunchCooperativeKernelMultiDevice_ptr);
hipError_t (*dyn_hipModuleLaunchCooperativeKernelMultiDevice_ptr)(hipFunctionLaunchParams*,
unsigned int, unsigned int) =
reinterpret_cast<hipError_t (*)(hipFunctionLaunchParams*, unsigned int, unsigned int)>(
hipModuleLaunchCooperativeKernelMultiDevice_ptr);
hipError_t (*dyn_hipLaunchCooperativeKernel_ptr)(
const void *, dim3, dim3, void **, unsigned int, hipStream_t) =
reinterpret_cast<hipError_t (*)(const void *, dim3, dim3, void **,
unsigned int, hipStream_t)>
(hipLaunchCooperativeKernel_ptr);
hipError_t (*dyn_hipLaunchCooperativeKernel_ptr)(const void*, dim3, dim3, void**, unsigned int,
hipStream_t) =
reinterpret_cast<hipError_t (*)(const void*, dim3, dim3, void**, unsigned int, hipStream_t)>(
hipLaunchCooperativeKernel_ptr);
hipError_t (*dyn_hipLaunchCooperativeKernelMultiDevice_ptr)(
hipLaunchParams *, int, unsigned int) =
reinterpret_cast<hipError_t (*)(hipLaunchParams *, int, unsigned int)>
(hipLaunchCooperativeKernelMultiDevice_ptr);
hipError_t (*dyn_hipLaunchCooperativeKernelMultiDevice_ptr)(hipLaunchParams*, int, unsigned int) =
reinterpret_cast<hipError_t (*)(hipLaunchParams*, int, unsigned int)>(
hipLaunchCooperativeKernelMultiDevice_ptr);
hipError_t (*dyn_hipExtLaunchMultiKernelMultiDevice_ptr)(
hipLaunchParams *, int, unsigned int) =
reinterpret_cast<hipError_t (*)(hipLaunchParams *, int, unsigned int)>
(hipExtLaunchMultiKernelMultiDevice_ptr);
hipError_t (*dyn_hipExtLaunchMultiKernelMultiDevice_ptr)(hipLaunchParams*, int, unsigned int) =
reinterpret_cast<hipError_t (*)(hipLaunchParams*, int, unsigned int)>(
hipExtLaunchMultiKernelMultiDevice_ptr);
const int N = 10;
const int Nbytes = 10 * sizeof(int);
int *hostArr = reinterpret_cast<int *>(malloc(Nbytes));
int* hostArr = reinterpret_cast<int*>(malloc(Nbytes));
REQUIRE(hostArr != nullptr);
fillHostArray(hostArr, N, 10);
int *devArr = nullptr;
int* devArr = nullptr;
HIP_CHECK(hipMalloc(&devArr, Nbytes));
REQUIRE(devArr != nullptr);
HIP_CHECK(hipMemcpy(devArr, hostArr, Nbytes, hipMemcpyHostToDevice));
@@ -477,13 +399,13 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
dim3 threadsPerBlock(1, 1, N);
struct kernelParameters {
void *arr;
void* arr;
int size;
};
kernelParameters kernelParam;
kernelParam.arr = devArr;
kernelParam.size = N;
void *kernel_parameter[] = {&kernelParam.arr, &kernelParam.size};
void* kernel_parameter[] = {&kernelParam.arr, &kernelParam.size};
// Validating hipModuleLaunchCooperativeKernel API
{
@@ -495,10 +417,9 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
HIP_CHECK(hipModuleGetFunction(&function, module, "addKernel"));
REQUIRE(function != nullptr);
HIP_CHECK(dyn_hipModuleLaunchCooperativeKernel_ptr(function,
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
0, 0, kernel_parameter));
HIP_CHECK(dyn_hipModuleLaunchCooperativeKernel_ptr(
function, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z, threadsPerBlock.x,
threadsPerBlock.y, threadsPerBlock.z, 0, 0, kernel_parameter));
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
REQUIRE(validateHostArray(hostArr, N, 12) == true);
@@ -510,9 +431,9 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
int deviceCount = 0;
HIP_CHECK(hipGetDeviceCount(&deviceCount));
hipModule_t *module = new hipModule_t[deviceCount];
hipFunction_t *function = new hipFunction_t[deviceCount];
hipStream_t *streamArr = new hipStream_t[deviceCount];
hipModule_t* module = new hipModule_t[deviceCount];
hipFunction_t* function = new hipFunction_t[deviceCount];
hipStream_t* streamArr = new hipStream_t[deviceCount];
for (int i = 0; i < deviceCount; ++i) {
HIP_CHECK(hipSetDevice(i));
@@ -521,8 +442,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
HIP_CHECK(hipModuleLoad(&module[i], "addKernel.code"));
REQUIRE(module[i] != nullptr);
HIP_CHECK(hipModuleGetFunction(&function[i], module[i],
"sampleModuleKernel"));
HIP_CHECK(hipModuleGetFunction(&function[i], module[i], "sampleModuleKernel"));
REQUIRE(function[i] != nullptr);
}
@@ -543,8 +463,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
params[i].hStream = streamArr[i];
}
HIP_CHECK(dyn_hipModuleLaunchCooperativeKernelMultiDevice_ptr(
params.data(), deviceCount, 0));
HIP_CHECK(dyn_hipModuleLaunchCooperativeKernelMultiDevice_ptr(params.data(), deviceCount, 0));
for (int i = 0; i < deviceCount; ++i) {
HIP_CHECK(hipStreamSynchronize(params[i].hStream));
@@ -558,10 +477,9 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
// Validating hipLaunchCooperativeKernel API
{
HIP_CHECK(dyn_hipLaunchCooperativeKernel_ptr(
reinterpret_cast<void *>(addOneKernel),
dim3(1, 1, 1), dim3(1, 1, 1),
kernel_parameter, 0, 0));
HIP_CHECK(dyn_hipLaunchCooperativeKernel_ptr(reinterpret_cast<void*>(addOneKernel),
dim3(1, 1, 1), dim3(1, 1, 1), kernel_parameter, 0,
0));
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
REQUIRE(validateHostArray(hostArr, N, 13) == true);
}
@@ -571,7 +489,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
int deviceCount = 0;
HIP_CHECK(hipGetDeviceCount(&deviceCount));
hipStream_t *streamArr = new hipStream_t[deviceCount];
hipStream_t* streamArr = new hipStream_t[deviceCount];
for (int i = 0; i < deviceCount; ++i) {
HIP_CHECK(hipSetDevice(i));
@@ -581,7 +499,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
std::vector<hipLaunchParams> params(deviceCount);
for (int i = 0; i < deviceCount; ++i) {
params[i].func = reinterpret_cast<void *>(simpleKernel);
params[i].func = reinterpret_cast<void*>(simpleKernel);
params[i].gridDim = {1, 1, 1};
params[i].blockDim = {1, 1, 1};
params[i].args = nullptr;
@@ -589,8 +507,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
params[i].stream = streamArr[i];
}
HIP_CHECK(dyn_hipLaunchCooperativeKernelMultiDevice_ptr(
params.data(), deviceCount, 0));
HIP_CHECK(dyn_hipLaunchCooperativeKernelMultiDevice_ptr(params.data(), deviceCount, 0));
for (int i = 0; i < deviceCount; ++i) {
HIP_CHECK(hipStreamSynchronize(params[i].stream));
@@ -606,7 +523,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
int deviceCount = 0;
HIP_CHECK(hipGetDeviceCount(&deviceCount));
hipStream_t *streamArr = new hipStream_t[deviceCount];
hipStream_t* streamArr = new hipStream_t[deviceCount];
for (int i = 0; i < deviceCount; ++i) {
HIP_CHECK(hipSetDevice(i));
@@ -616,7 +533,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
std::vector<hipLaunchParams> params(deviceCount);
for (int i = 0; i < deviceCount; ++i) {
params[i].func = reinterpret_cast<void *>(simpleKernel);
params[i].func = reinterpret_cast<void*>(simpleKernel);
params[i].gridDim = {1, 1, 1};
params[i].blockDim = {1, 1, 1};
params[i].args = nullptr;
@@ -624,8 +541,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
params[i].stream = streamArr[i];
}
HIP_CHECK(dyn_hipExtLaunchMultiKernelMultiDevice_ptr(
params.data(), deviceCount, 0));
HIP_CHECK(dyn_hipExtLaunchMultiKernelMultiDevice_ptr(params.data(), deviceCount, 0));
for (int i = 0; i < deviceCount; ++i) {
HIP_CHECK(hipStreamSynchronize(params[i].stream));
@@ -658,8 +574,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
void* hipModuleOccupancyMaxPotentialBlockSize_ptr = nullptr;
void* hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr = nullptr;
void* hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr = nullptr;
void* hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr =
nullptr;
void* hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr = nullptr;
void* hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr = nullptr;
void* hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr = nullptr;
void* hipOccupancyMaxPotentialBlockSize_ptr = nullptr;
@@ -667,73 +582,61 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
int currentHipVersion = 0;
HIP_CHECK(hipRuntimeGetVersion(&currentHipVersion));
HIP_CHECK(hipGetProcAddress(
"hipModuleOccupancyMaxPotentialBlockSize",
&hipModuleOccupancyMaxPotentialBlockSize_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipModuleOccupancyMaxPotentialBlockSizeWithFlags",
&hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipModuleOccupancyMaxActiveBlocksPerMultiprocessor",
&hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags",
&hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipOccupancyMaxActiveBlocksPerMultiprocessor",
&hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags",
&hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress(
"hipOccupancyMaxPotentialBlockSize",
&hipOccupancyMaxPotentialBlockSize_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleOccupancyMaxPotentialBlockSize",
&hipModuleOccupancyMaxPotentialBlockSize_ptr, currentHipVersion, 0,
nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleOccupancyMaxPotentialBlockSizeWithFlags",
&hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleOccupancyMaxActiveBlocksPerMultiprocessor",
&hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags",
&hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipOccupancyMaxActiveBlocksPerMultiprocessor",
&hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr, currentHipVersion,
0, nullptr));
HIP_CHECK(hipGetProcAddress("hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags",
&hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr,
currentHipVersion, 0, nullptr));
HIP_CHECK(hipGetProcAddress("hipOccupancyMaxPotentialBlockSize",
&hipOccupancyMaxPotentialBlockSize_ptr, currentHipVersion, 0,
nullptr));
hipError_t(*dyn_hipModuleOccupancyMaxPotentialBlockSize_ptr)(
int *, int *, hipFunction_t, size_t, int) =
reinterpret_cast<hipError_t (*)(int *, int *, hipFunction_t, size_t, int)>
(hipModuleOccupancyMaxPotentialBlockSize_ptr);
hipError_t (*dyn_hipModuleOccupancyMaxPotentialBlockSize_ptr)(int*, int*, hipFunction_t, size_t,
int) =
reinterpret_cast<hipError_t (*)(int*, int*, hipFunction_t, size_t, int)>(
hipModuleOccupancyMaxPotentialBlockSize_ptr);
hipError_t(*dyn_hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr)(
int *, int *, hipFunction_t, size_t, int, unsigned int) =
reinterpret_cast<hipError_t (*)(int *, int *, hipFunction_t,
size_t, int, unsigned int)>
(hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr);
hipError_t (*dyn_hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr)(
int*, int*, hipFunction_t, size_t, int, unsigned int) =
reinterpret_cast<hipError_t (*)(int*, int*, hipFunction_t, size_t, int, unsigned int)>(
hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr);
hipError_t(*dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr)(
int *, hipFunction_t, int, size_t) =
reinterpret_cast<hipError_t (*)(int *, hipFunction_t, int, size_t)>
(hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr);
hipError_t (*dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr)(int*, hipFunction_t, int,
size_t) =
reinterpret_cast<hipError_t (*)(int*, hipFunction_t, int, size_t)>(
hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr);
hipError_t(
*dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr)(
int *, hipFunction_t, int, size_t, unsigned int) =
reinterpret_cast<hipError_t (*)(int *, hipFunction_t, int,
size_t, unsigned int)>
(hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr);
hipError_t (*dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr)(
int*, hipFunction_t, int, size_t, unsigned int) =
reinterpret_cast<hipError_t (*)(int*, hipFunction_t, int, size_t, unsigned int)>(
hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr);
hipError_t(*dyn_hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr)(
int *, const void *, int, size_t) =
reinterpret_cast<hipError_t (*)(int *, const void *, int, size_t)>
(hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr);
hipError_t (*dyn_hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr)(int*, const void*, int,
size_t) =
reinterpret_cast<hipError_t (*)(int*, const void*, int, size_t)>(
hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr);
hipError_t(*dyn_hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr)(
int *, const void *, int, size_t, unsigned int) =
reinterpret_cast<hipError_t (*)(int *, const void *,
int, size_t, unsigned int)>
(hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr);
hipError_t (*dyn_hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr)(
int*, const void*, int, size_t, unsigned int) =
reinterpret_cast<hipError_t (*)(int*, const void*, int, size_t, unsigned int)>(
hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr);
hipError_t(*dyn_hipOccupancyMaxPotentialBlockSize_ptr)(
int *, int *, const void *, size_t, int) =
reinterpret_cast<hipError_t (*)(int *, int *, const void *, size_t, int)>
(hipOccupancyMaxPotentialBlockSize_ptr);
hipError_t (*dyn_hipOccupancyMaxPotentialBlockSize_ptr)(int*, int*, const void*, size_t, int) =
reinterpret_cast<hipError_t (*)(int*, int*, const void*, size_t, int)>(
hipOccupancyMaxPotentialBlockSize_ptr);
hipModule_t module;
HIP_CHECK(hipModuleLoad(&module, "addKernel.code"));
@@ -747,10 +650,9 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
// Validating hipModuleOccupancyMaxPotentialBlockSize API
{
HIP_CHECK(hipModuleOccupancyMaxPotentialBlockSize(&gridSize, &blockSize,
function, 0, 0));
HIP_CHECK(hipModuleOccupancyMaxPotentialBlockSize(&gridSize, &blockSize, function, 0, 0));
HIP_CHECK(dyn_hipModuleOccupancyMaxPotentialBlockSize_ptr(
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr, function, 0, 0));
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr, function, 0, 0));
REQUIRE(gridSizeWithFuncPtr == gridSize);
REQUIRE(blockSizeWithFuncPtr == blockSize);
@@ -758,12 +660,14 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
// Validating hipModuleOccupancyMaxPotentialBlockSizeWithFlags API
{
gridSize = 0; blockSize = 0;
gridSizeWithFuncPtr = 0; blockSizeWithFuncPtr = 0;
HIP_CHECK(hipModuleOccupancyMaxPotentialBlockSizeWithFlags(
&gridSize, &blockSize, function, 0, 0, 0));
gridSize = 0;
blockSize = 0;
gridSizeWithFuncPtr = 0;
blockSizeWithFuncPtr = 0;
HIP_CHECK(
hipModuleOccupancyMaxPotentialBlockSizeWithFlags(&gridSize, &blockSize, function, 0, 0, 0));
HIP_CHECK(dyn_hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr(
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr, function, 0, 0, 0));
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr, function, 0, 0, 0));
REQUIRE(gridSizeWithFuncPtr == gridSize);
REQUIRE(blockSizeWithFuncPtr == blockSize);
@@ -772,63 +676,61 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
int numBlocks = 0, numBlocksWithFuncPtr = 0;
// Validating hipModuleOccupancyMaxActiveBlocksPerMultiprocessor API
{
HIP_CHECK(hipModuleOccupancyMaxActiveBlocksPerMultiprocessor(
&numBlocks, function, blockSize, 0));
HIP_CHECK(dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr(
&numBlocksWithFuncPtr, function, blockSize, 0));
HIP_CHECK(
hipModuleOccupancyMaxActiveBlocksPerMultiprocessor(&numBlocks, function, blockSize, 0));
HIP_CHECK(dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr(&numBlocksWithFuncPtr,
function, blockSize, 0));
REQUIRE(numBlocksWithFuncPtr == numBlocks);
}
// Validating hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags API
{
numBlocks = 0; numBlocksWithFuncPtr = 0;
HIP_CHECK(hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(
&numBlocks, function, blockSize, 0, 0));
HIP_CHECK(
dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr(
&numBlocksWithFuncPtr, function, blockSize, 0, 0));
numBlocks = 0;
numBlocksWithFuncPtr = 0;
HIP_CHECK(hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(&numBlocks, function,
blockSize, 0, 0));
HIP_CHECK(dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr(
&numBlocksWithFuncPtr, function, blockSize, 0, 0));
REQUIRE(numBlocksWithFuncPtr == numBlocks);
}
// Validating hipOccupancyMaxActiveBlocksPerMultiprocessor API
{
numBlocks = 0; numBlocksWithFuncPtr = 0;
numBlocks = 0;
numBlocksWithFuncPtr = 0;
HIP_CHECK(hipOccupancyMaxActiveBlocksPerMultiprocessor(
&numBlocks, reinterpret_cast<const void *>(addOneKernel),
blockSize, 0));
&numBlocks, reinterpret_cast<const void*>(addOneKernel), blockSize, 0));
HIP_CHECK(dyn_hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr(
&numBlocksWithFuncPtr,
reinterpret_cast<const void *>(addOneKernel), blockSize, 0));
&numBlocksWithFuncPtr, reinterpret_cast<const void*>(addOneKernel), blockSize, 0));
REQUIRE(numBlocksWithFuncPtr == numBlocks);
}
// Validating hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags API
{
numBlocks = 0; numBlocksWithFuncPtr = 0;
numBlocks = 0;
numBlocksWithFuncPtr = 0;
HIP_CHECK(hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(
&numBlocks, reinterpret_cast<const void *>(addOneKernel),
blockSize, 0, 0));
&numBlocks, reinterpret_cast<const void*>(addOneKernel), blockSize, 0, 0));
HIP_CHECK(dyn_hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr(
&numBlocksWithFuncPtr,
reinterpret_cast<const void *>(addOneKernel),
blockSize, 0, 0));
&numBlocksWithFuncPtr, reinterpret_cast<const void*>(addOneKernel), blockSize, 0, 0));
REQUIRE(numBlocksWithFuncPtr == numBlocks);
}
// Validating hipOccupancyMaxPotentialBlockSize API
{
gridSize = 0; blockSize = 0;
gridSizeWithFuncPtr = 0; blockSizeWithFuncPtr = 0;
HIP_CHECK(hipOccupancyMaxPotentialBlockSize(
&gridSize, &blockSize,
reinterpret_cast<const void *>(addOneKernel), 0, 0));
HIP_CHECK(dyn_hipOccupancyMaxPotentialBlockSize_ptr(
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr,
reinterpret_cast<const void *>(addOneKernel), 0, 0));
gridSize = 0;
blockSize = 0;
gridSizeWithFuncPtr = 0;
blockSizeWithFuncPtr = 0;
HIP_CHECK(hipOccupancyMaxPotentialBlockSize(&gridSize, &blockSize,
reinterpret_cast<const void*>(addOneKernel), 0, 0));
HIP_CHECK(dyn_hipOccupancyMaxPotentialBlockSize_ptr(&gridSizeWithFuncPtr, &blockSizeWithFuncPtr,
reinterpret_cast<const void*>(addOneKernel),
0, 0));
REQUIRE(gridSizeWithFuncPtr == gridSize);
REQUIRE(blockSizeWithFuncPtr == blockSize);
+18 -19
ファイルの表示
@@ -61,22 +61,22 @@ TEST_CASE("Unit_hipHccModuleLaunchKernel_basic") {
size_t width = GENERATE(3, 4, 100);
size_t widthInBytes = width * sizeof(int);
int *A_d, *B_d;
int *A_h = reinterpret_cast<int*>(malloc(widthInBytes));
int *B_h = reinterpret_cast<int*>(malloc(widthInBytes));
int* A_h = reinterpret_cast<int*>(malloc(widthInBytes));
int* B_h = reinterpret_cast<int*>(malloc(widthInBytes));
for (int i = 0; i < width; i++) {
A_h[i] = i;
}
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&A_d), widthInBytes));
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&B_d), widthInBytes));
void *kernelArgs[3] = {&A_d, &B_d, &widthInBytes};
void* kernelArgs[3] = {&A_d, &B_d, &widthInBytes};
HIP_CHECK(hipMemcpyHtoD((hipDeviceptr_t)A_d, A_h, widthInBytes));
hipModule_t module;
HIP_CHECK(hipModuleLoad(&module, fileName));
hipFunction_t kernelFunc;
HIP_CHECK(hipModuleGetFunction(&kernelFunc, module, kernel_name));
HIP_CHECK(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, 1, 0, 0,
kernelArgs, nullptr, nullptr, nullptr));
HIP_CHECK(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, 1, 0, 0, kernelArgs,
nullptr, nullptr, nullptr));
HIP_CHECK(hipMemcpyDtoH(B_h, (hipDeviceptr_t)B_d, widthInBytes));
for (int i = 0; i < width; i++) {
REQUIRE(A_h[i] == B_h[i]);
@@ -105,43 +105,42 @@ TEST_CASE("Unit_hipHccModuleLaunchKernel_NegTst") {
int *A_d, *B_d;
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&A_d), widthInBytes));
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&B_d), widthInBytes));
void *kernelArgs[3] = {&A_d, &B_d, &widthInBytes};
void* kernelArgs[3] = {&A_d, &B_d, &widthInBytes};
hipModule_t module;
HIP_CHECK(hipModuleLoad(&module, fileName));
hipFunction_t kernelFunc;
HIP_CHECK(hipModuleGetFunction(&kernelFunc, module, kernel_name));
SECTION("nullptr to f(first argument)") {
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(nullptr, width, 1, 1, width, 1, 1,
0, 0, kernelArgs, nullptr, nullptr, nullptr),
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(nullptr, width, 1, 1, width, 1, 1, 0, 0, kernelArgs,
nullptr, nullptr, nullptr),
hipErrorInvalidHandle);
}
SECTION("-1 to localWorkSizeX(fifth argument)") {
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, -1, 1, 1,
0, 0, kernelArgs, nullptr, nullptr, nullptr),
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, -1, 1, 1, 0, 0, kernelArgs,
nullptr, nullptr, nullptr),
hipErrorInvalidConfiguration);
}
SECTION("-1 to localWorkSizeY(sixth argument)") {
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width,
-1, 1, 0, 0, kernelArgs, nullptr, nullptr, nullptr),
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, -1, 1, 0, 0,
kernelArgs, nullptr, nullptr, nullptr),
hipErrorInvalidConfiguration);
}
SECTION("-1 to localWorkSizeZ(seventh argument)") {
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1,
-1, 0, 0, kernelArgs, nullptr, nullptr, nullptr),
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, -1, 0, 0,
kernelArgs, nullptr, nullptr, nullptr),
hipErrorInvalidConfiguration);
}
SECTION("-1 to sharedMemBytes(eighth argument)") {
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1,
1, -1, 0, kernelArgs, nullptr, nullptr, nullptr),
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, 1, -1, 0,
kernelArgs, nullptr, nullptr, nullptr),
hipErrorInvalidValue);
}
SECTION("nullptr to kernelParams(10th argument)") {
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1,
1, 0, 0, nullptr, nullptr, nullptr, nullptr),
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, 1, 0, 0, nullptr,
nullptr, nullptr, nullptr),
hipErrorInvalidValue);
}
HIP_CHECK(hipModuleUnload(module));
HIP_CHECK(hipFree(A_d));
HIP_CHECK(hipFree(B_d));
}
+22 -28
ファイルの表示
@@ -29,25 +29,21 @@ static constexpr auto fileName3 = "copiousArgKernel3.code";
static constexpr auto fileName16 = "copiousArgKernel16.code";
static constexpr auto fileName17 = "copiousArgKernel17.code";
static constexpr int coeff[12] =
{2, 3, 5, 7, 11, 13, 17, 19, 23, 29, 31, 37};
static constexpr int coeff[12] = {2, 3, 5, 7, 11, 13, 17, 19, 23, 29, 31, 37};
static void fillDataTransfer2Dev(int *hostBuf, int *devBuf, size_t len) {
static void fillDataTransfer2Dev(int* hostBuf, int* devBuf, size_t len) {
unsigned int seed = time(nullptr);
for (size_t i = 0; i < len; i++) {
hostBuf[i] = (HipTest::RAND_R(&seed) & 0xFF);
}
HIP_CHECK(hipMemcpy(devBuf, hostBuf, len*sizeof(int),
hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(devBuf, hostBuf, len * sizeof(int), hipMemcpyHostToDevice));
}
static void verifyDevResult(int *hostBuf, int *devBuf, int coef1, int coef2,
size_t len) {
int *buf = new int[len];
HIP_CHECK(hipMemcpy(buf, devBuf, len*sizeof(int),
hipMemcpyDeviceToHost));
static void verifyDevResult(int* hostBuf, int* devBuf, int coef1, int coef2, size_t len) {
int* buf = new int[len];
HIP_CHECK(hipMemcpy(buf, devBuf, len * sizeof(int), hipMemcpyDeviceToHost));
for (size_t i = 0; i < len; i++) {
REQUIRE(buf[i] == (coef1*hostBuf[i] + coef2));
REQUIRE(buf[i] == (coef1 * hostBuf[i] + coef2));
}
delete[] buf;
}
@@ -57,17 +53,17 @@ TEST_CASE("Unit_KerArgOptimization_Saxpy") {
constexpr size_t arraylenBytes = arraylen * sizeof(int);
constexpr auto blocksize = 256;
// Allocate host resources
int *x1_h = new int[arraylen];
int* x1_h = new int[arraylen];
REQUIRE(x1_h != nullptr);
int *x2_h = new int[arraylen];
int* x2_h = new int[arraylen];
REQUIRE(x2_h != nullptr);
int *x3_h = new int[arraylen];
int* x3_h = new int[arraylen];
REQUIRE(x3_h != nullptr);
int *x4_h = new int[arraylen];
int* x4_h = new int[arraylen];
REQUIRE(x4_h != nullptr);
int *x5_h = new int[arraylen];
int* x5_h = new int[arraylen];
REQUIRE(x5_h != nullptr);
int *x6_h = new int[arraylen];
int* x6_h = new int[arraylen];
REQUIRE(x6_h != nullptr);
// Allocate device resources
int *x1_d, *x2_d, *x3_d, *x4_d, *x5_d, *x6_d;
@@ -89,22 +85,22 @@ TEST_CASE("Unit_KerArgOptimization_Saxpy") {
struct {
int a1;
int a2;
void *x1;
void* x1;
int b1;
int b2;
void *x2;
void* x2;
int c1;
int c2;
void *x3;
void* x3;
int d1;
int d2;
void *x4;
void* x4;
int e1;
int e2;
void *x5;
void* x5;
int f1;
int f2;
void *x6;
void* x6;
} args;
args.a1 = coeff[0];
args.a2 = coeff[1];
@@ -125,8 +121,7 @@ TEST_CASE("Unit_KerArgOptimization_Saxpy") {
args.f2 = coeff[11];
args.x6 = x6_d;
size_t size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
// Get module and function from module
@@ -160,9 +155,8 @@ TEST_CASE("Unit_KerArgOptimization_Saxpy") {
HIP_CHECK(hipModuleLoad(&Module, fileName17));
HIP_CHECK(hipModuleGetFunction(&Function, Module, kernel_name));
}
HIP_CHECK(hipExtModuleLaunchKernel(Function, arraylen, 1, 1,
blocksize, 1, 1, 0, 0, NULL,
reinterpret_cast<void**>(&config), 0));
HIP_CHECK(hipExtModuleLaunchKernel(Function, arraylen, 1, 1, blocksize, 1, 1, 0, 0, NULL,
reinterpret_cast<void**>(&config), 0));
HIP_CHECK(hipDeviceSynchronize());
// Verify results
verifyDevResult(x1_h, x1_d, coeff[0], coeff[1], arraylen);
+10 -14
ファイルの表示
@@ -20,15 +20,15 @@ THE SOFTWARE.
#include <hip_test_defgroups.hh>
constexpr int MANAGED_VAR_INIT_VALUE = 10;
constexpr auto fileName = "managed_kernel.code";
constexpr auto fileName = "managed_kernel.code";
/**
* @addtogroup hipModuleGetGlobal
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleGetGlobal(hipDeviceptr_t* dptr, size_t* bytes, hipModule_t hmod, const char* name)` -
* Returns a global pointer from a module
*/
* @addtogroup hipModuleGetGlobal
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleGetGlobal(hipDeviceptr_t* dptr, size_t* bytes, hipModule_t hmod, const char*
* name)` - Returns a global pointer from a module
*/
/**
* Test Description
@@ -52,9 +52,7 @@ TEST_CASE("Unit_hipModuleGetGlobal_Functional") {
HIP_CHECK(hipGetDeviceCount(&numDevices));
for (int i = 0; i < numDevices; i++) {
int managed_memory = 0;
HIPCHECK(hipDeviceGetAttribute(&managed_memory,
hipDeviceAttributeManagedMemory,
i));
HIPCHECK(hipDeviceGetAttribute(&managed_memory, hipDeviceAttributeManagedMemory, i));
if (!managed_memory) {
HipTest::HIP_SKIP_TEST("managed memory access not supported on device");
return;
@@ -70,11 +68,9 @@ TEST_CASE("Unit_hipModuleGetGlobal_Functional") {
HIP_CHECK(hipModuleLoad(&Module, fileName));
hipFunction_t Function;
HIP_CHECK(hipModuleGetFunction(&Function, Module, "GPU_func"));
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, 1, 1, 1, 0, 0,
NULL, NULL));
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, 1, 1, 1, 0, 0, NULL, NULL));
HIP_CHECK(hipDeviceSynchronize());
HIP_CHECK(hipModuleGetGlobal(reinterpret_cast<hipDeviceptr_t*>(&x),
&xSize, Module, "x"));
HIP_CHECK(hipModuleGetGlobal(reinterpret_cast<hipDeviceptr_t*>(&x), &xSize, Module, "x"));
HIP_CHECK(hipMemcpyDtoH(&data, hipDeviceptr_t(x), xSize));
if (data != (1 + MANAGED_VAR_INIT_VALUE)) {
HIP_CHECK(hipModuleUnload(Module));
+19 -25
ファイルの表示
@@ -33,18 +33,19 @@ constexpr auto CODE_OBJ_MULTIARCH = "vcpy_kernel_multarch.code";
#endif
/**
* @addtogroup hipModuleLoad
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
* Loads code object from file into a module
*/
* @addtogroup hipModuleLoad
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
* Loads code object from file into a module
*/
/**
* Test Description
* ------------------------
* - Test case to load and execute a code object file for the current GPU architecture.
* - Test case to load and execute a code object file for the multiple GPU architectures including the current
* - Test case to load and execute a code object file for the multiple GPU architectures including
the current
* Test source
* ------------------------
@@ -54,7 +55,7 @@ constexpr auto CODE_OBJ_MULTIARCH = "vcpy_kernel_multarch.code";
* - HIP_VERSION >= 5.6
*/
bool testCodeObjFile(const char *codeObjFile) {
bool testCodeObjFile(const char* codeObjFile) {
float *A, *B, *Ad, *Bd;
A = new float[LEN];
B = new float[LEN];
@@ -85,12 +86,10 @@ bool testCodeObjFile(const char *codeObjFile) {
args._Bd = reinterpret_cast<void*>(Bd);
size_t size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0,
stream, NULL,
reinterpret_cast<void**>(&config)));
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0, stream, NULL,
reinterpret_cast<void**>(&config)));
HIP_CHECK(hipStreamDestroy(stream));
@@ -114,8 +113,8 @@ bool testCodeObjFile(const char *codeObjFile) {
#ifdef __linux__
// Check if environment variable $ROCM_PATH is defined
bool isRocmPathSet() {
FILE *fpipe;
char const *command = "echo $ROCM_PATH";
FILE* fpipe;
char const* command = "echo $ROCM_PATH";
fpipe = popen(command, "r");
if (fpipe == nullptr) {
@@ -143,14 +142,12 @@ bool testMultiTargArchCodeObj() {
HIP_CHECK(hipGetDeviceProperties(&props, 0));
// Hardcoding the codeobject lines in multiple string to avoid cpplint warning
std::string CodeObjL1 = "#include \"hip/hip_runtime.h\"\n";
std::string CodeObjL2 =
"extern \"C\" __global__ void hello_world(float* a, float* b) {\n";
std::string CodeObjL2 = "extern \"C\" __global__ void hello_world(float* a, float* b) {\n";
std::string CodeObjL3 = " int tx = threadIdx.x;\n";
std::string CodeObjL4 = " b[tx] = a[tx];\n";
std::string CodeObjL5 = "}";
// Creating the full code object string
static std::string CodeObj = CodeObjL1 + CodeObjL2 + CodeObjL3 +
CodeObjL4 + CodeObjL5;
static std::string CodeObj = CodeObjL1 + CodeObjL2 + CodeObjL3 + CodeObjL4 + CodeObjL5;
std::ofstream ofs("/tmp/vcpy_kernel.cpp", std::ofstream::out);
ofs << CodeObj;
ofs.close();
@@ -172,15 +169,12 @@ bool testMultiTargArchCodeObj() {
const char* genco_option = "--offload-arch";
const char* input_codeobj = "/tmp/vcpy_kernel.cpp";
const char* rocm_enumerator = "${ROCM_PATH}/bin/rocm_agent_enumerator";
snprintf(command, COMMAND_LEN,
rocm_enumerator,
hipcc_path, genco_option, props.gcnArchName, input_codeobj,
CODE_OBJ_MULTIARCH);
snprintf(command, COMMAND_LEN, rocm_enumerator, hipcc_path, genco_option, props.gcnArchName,
input_codeobj, CODE_OBJ_MULTIARCH);
system((const char*)command);
// Check if the code object file is created
snprintf(command, COMMAND_LEN, "./%s",
CODE_OBJ_MULTIARCH);
snprintf(command, COMMAND_LEN, "./%s", CODE_OBJ_MULTIARCH);
if (access(command, F_OK) == -1) {
INFO("Code Object File not found \n");
+3 -3
ファイルの表示
@@ -211,6 +211,6 @@ TEST_CASE("Unit_hipModuleLaunchCooperativeKernel_Negative_Parameters") {
}
/**
* End doxygen group ModuleTest.
* @}
*/
* End doxygen group ModuleTest.
* @}
*/
+43 -91
ファイルの表示
@@ -101,13 +101,13 @@ bool Module_Negative_tests() {
args1._Ad = nullptr;
args1._Bd = nullptr;
args1._Cd = nullptr;
args1._n = 0;
args1._n = 0;
hipFunction_t MultKernel, KernelandExtraParamKernel;
size_t size1;
size1 = sizeof(args1);
hipModule_t Module;
hipStream_t stream1;
hipDeviceptr_t *Ad = nullptr;
hipDeviceptr_t* Ad = nullptr;
#ifdef HT_NVIDIA
HIP_CHECK(hipInit(0));
hipCtx_t context;
@@ -116,122 +116,88 @@ bool Module_Negative_tests() {
HIP_CHECK(hipModuleLoad(&Module, fileName));
HIP_CHECK(hipModuleGetFunction(&MultKernel, Module, matmulK));
HIP_CHECK(hipModuleGetFunction(&KernelandExtraParamKernel,
Module, KernelandExtra));
void *config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
HIP_LAUNCH_PARAM_END};
void *params[] = {Ad};
HIP_CHECK(hipModuleGetFunction(&KernelandExtraParamKernel, Module, KernelandExtra));
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
HIP_LAUNCH_PARAM_END};
void* params[] = {Ad};
HIP_CHECK(hipStreamCreate(&stream1));
// Passing nullptr to kernel function
err = hipModuleLaunchKernel(nullptr, 1, 1, 1, 1, 1, 1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(nullptr, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Passing Max int value to block dimensions
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, std::numeric_limits<uint32_t>::max(),
std::numeric_limits<uint32_t>::max(),
std::numeric_limits<uint32_t>::max(),
std::numeric_limits<uint32_t>::max(),
0, stream1, NULL,
std::numeric_limits<uint32_t>::max(), 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Passing 0 as value for all dimensions
err = hipModuleLaunchKernel(MultKernel, 0, 0, 0,
0,
0,
0, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(MultKernel, 0, 0, 0, 0, 0, 0, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Passing 0 as value for x dimension
err = hipModuleLaunchKernel(MultKernel, 0, 1, 1,
0,
1,
1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(MultKernel, 0, 1, 1, 0, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Passing 0 as value for y dimension
err = hipModuleLaunchKernel(MultKernel, 1, 0, 1,
1,
0,
1, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(MultKernel, 1, 0, 1, 1, 0, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Passing 0 as value for z dimension
err = hipModuleLaunchKernel(MultKernel, 1, 1, 0,
1,
1,
0, 0,
stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(MultKernel, 1, 1, 0, 1, 1, 0, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Passing both kernel and extra params
err = hipModuleLaunchKernel(KernelandExtraParamKernel, 1, 1, 1, 1,
1, 1, 0, stream1,
reinterpret_cast<void**>(&params),
reinterpret_cast<void**>(&config1));
err =
hipModuleLaunchKernel(KernelandExtraParamKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
reinterpret_cast<void**>(&params), reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Passing more than maxthreadsperblock to block dimensions
hipDeviceProp_t deviceProp;
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
deviceProp.maxThreadsPerBlock+1,
deviceProp.maxThreadsPerBlock+1,
deviceProp.maxThreadsPerBlock+1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, deviceProp.maxThreadsPerBlock + 1,
deviceProp.maxThreadsPerBlock + 1, deviceProp.maxThreadsPerBlock + 1,
0, stream1, NULL, reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Block dimension X = Max Allowed + 1
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
deviceProp.maxThreadsDim[0]+1,
1,
1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, deviceProp.maxThreadsDim[0] + 1, 1, 1, 0,
stream1, NULL, reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Block dimension Y = Max Allowed + 1
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
1,
deviceProp.maxThreadsDim[1]+1,
1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, 1, deviceProp.maxThreadsDim[1] + 1, 1, 0,
stream1, NULL, reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Block dimension Z = Max Allowed + 1
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
1,
1,
deviceProp.maxThreadsDim[2]+1, 0, stream1, NULL,
reinterpret_cast<void**>(&config1));
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, 1, 1, deviceProp.maxThreadsDim[2] + 1, 0,
stream1, NULL, reinterpret_cast<void**>(&config1));
if (err == hipSuccess) {
testStatus = false;
}
// Passing invalid config data to extra params
void *config3[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
void* config3[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
HIP_LAUNCH_PARAM_END};
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
reinterpret_cast<void**>(&config3));
reinterpret_cast<void**>(&config3));
if (err == hipSuccess) {
testStatus = false;
}
@@ -275,22 +241,13 @@ bool Module_GridBlock_Corner_Tests() {
unsigned int maxgridY = deviceProp.maxGridSize[1];
unsigned int maxgridZ = deviceProp.maxGridSize[2];
#endif
struct gridblockDim test[6] = {{1, 1, 1, maxblockX, 1, 1},
{1, 1, 1, 1, maxblockY, 1},
{1, 1, 1, 1, 1, maxblockZ},
{maxgridX, 1, 1, 1, 1, 1},
{1, maxgridY, 1, 1, 1, 1},
{1, 1, maxgridZ, 1, 1, 1}};
struct gridblockDim test[6] = {{1, 1, 1, maxblockX, 1, 1}, {1, 1, 1, 1, maxblockY, 1},
{1, 1, 1, 1, 1, maxblockZ}, {maxgridX, 1, 1, 1, 1, 1},
{1, maxgridY, 1, 1, 1, 1}, {1, 1, maxgridZ, 1, 1, 1}};
for (int i = 0; i < 6; i++) {
err = hipModuleLaunchKernel(DummyKernel,
test[i].gridX,
test[i].gridY,
test[i].gridZ,
test[i].blockX,
test[i].blockY,
test[i].blockZ,
0,
stream1, NULL, NULL);
err = hipModuleLaunchKernel(DummyKernel, test[i].gridX, test[i].gridY, test[i].gridZ,
test[i].blockX, test[i].blockY, test[i].blockZ, 0, stream1, NULL,
NULL);
if (err != hipSuccess) {
testStatus = false;
}
@@ -321,25 +278,20 @@ bool Module_WorkGroup_Test() {
// Passing Max int value to block dimensions
hipDeviceProp_t deviceProp;
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
double cuberootVal =
cbrt(static_cast<double>(deviceProp.maxThreadsPerBlock));
double cuberootVal = cbrt(static_cast<double>(deviceProp.maxThreadsPerBlock));
uint32_t cuberoot_floor = floor(cuberootVal);
uint32_t cuberoot_ceil = ceil(cuberootVal);
// Scenario: (block.x * block.y * block.z) <= Work Group Size where
// block.x < MaxBlockDimX , block.y < MaxBlockDimY and block.z < MaxBlockDimZ
err = hipModuleLaunchKernel(DummyKernel,
1, 1, 1,
cuberoot_floor, cuberoot_floor, cuberoot_floor,
0, stream1, NULL, NULL);
err = hipModuleLaunchKernel(DummyKernel, 1, 1, 1, cuberoot_floor, cuberoot_floor, cuberoot_floor,
0, stream1, NULL, NULL);
if (err != hipSuccess) {
testStatus = false;
}
// Scenario: (block.x * block.y * block.z) > Work Group Size where
// block.x < MaxBlockDimX , block.y < MaxBlockDimY and block.z < MaxBlockDimZ
err = hipModuleLaunchKernel(DummyKernel,
1, 1, 1,
cuberoot_ceil, cuberoot_ceil, cuberoot_ceil + 1,
0, stream1, NULL, NULL);
err = hipModuleLaunchKernel(DummyKernel, 1, 1, 1, cuberoot_ceil, cuberoot_ceil, cuberoot_ceil + 1,
0, stream1, NULL, NULL);
if (err == hipSuccess) {
testStatus = false;
}
+13 -14
ファイルの表示
@@ -107,14 +107,14 @@ TEST_CASE("Unit_hipModuleLoadData_Negative_Image_Is_An_Empty_String") {
}
/**
* @addtogroup hipModuleLoad hipModuleGetFunction
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
* Loads code object from file into a module
* `hipError_t hipModuleGetFunction(hipFunction_t* function, hipModule_t module, const char* kname)` -
* Function with kname will be extracted if present in module
*/
* @addtogroup hipModuleLoad hipModuleGetFunction
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
* Loads code object from file into a module
* `hipError_t hipModuleGetFunction(hipFunction_t* function, hipModule_t module, const char* kname)`
* - Function with kname will be extracted if present in module
*/
/**
* Test Description
@@ -172,11 +172,10 @@ TEST_CASE("Unit_hipModuleLoadData_Functional") {
args._Bd = reinterpret_cast<void*>(Bd);
size_t size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0,
stream, NULL, reinterpret_cast<void**>(&config)));
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0, stream, NULL,
reinterpret_cast<void**>(&config)));
HIP_CHECK(hipStreamDestroy(stream));
@@ -185,8 +184,8 @@ TEST_CASE("Unit_hipModuleLoadData_Functional") {
for (uint32_t i = 0; i < LEN; i++) {
REQUIRE(A[i] == B[i]);
}
delete [] A;
delete [] B;
delete[] A;
delete[] B;
HIP_CHECK(hipModuleUnload(Module));
HIP_CHECK(hipFree(Ad));
HIP_CHECK(hipFree(Bd));
+12 -12
ファイルの表示
@@ -20,18 +20,18 @@ THE SOFTWARE.
#include <hip_test_defgroups.hh>
#include <hip_test_process.hh>
/**
* @addtogroup hipModuleLoad hipModuleLoadData hipModuleLoadDataEx
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
* Loads code object from file into a module
* `hipError_t hipModuleLoadData (hipModule_t *module, const void *image)` -
* Builds module from code object which resides in host memory. Image is pointer to that location.
* `hipError_t hipModuleLoadDataEx (hipModule_t *module, const void *image,
* unsigned int numOptions, hipJitOption *options, void **optionValues)` -
* Builds module from code object which resides in host memory. Image is pointer to that
* location. Options are not used.
*/
* @addtogroup hipModuleLoad hipModuleLoadData hipModuleLoadDataEx
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
* Loads code object from file into a module
* `hipError_t hipModuleLoadData (hipModule_t *module, const void *image)` -
* Builds module from code object which resides in host memory. Image is pointer to that location.
* `hipError_t hipModuleLoadDataEx (hipModule_t *module, const void *image,
* unsigned int numOptions, hipJitOption *options, void **optionValues)` -
* Builds module from code object which resides in host memory. Image is pointer to that
* location. Options are not used.
*/
/**
* Test Description
+12 -12
ファイルの表示
@@ -35,12 +35,12 @@ TEST_CASE("Unit_hipModuleUnload_Negative_Double_Unload") {
HIP_CHECK_ERROR(hipModuleUnload(module), hipErrorNotFound);
}
/**
* @addtogroup hipModuleUnload
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleUnload(hipModule_t module)` -
* Frees the module
*/
* @addtogroup hipModuleUnload
* @{
* @ingroup ModuleTest
* `hipError_t hipModuleUnload(hipModule_t module)` -
* Frees the module
*/
/**
* Test Description
@@ -52,11 +52,11 @@ TEST_CASE("Unit_hipModuleUnload_Negative_Double_Unload") {
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.6
*/
*/
TEST_CASE("Unit_hipModuleLoad_basic") {
constexpr auto fileName = "vcpy_kernel.code";
hipModule_t module;
HIP_CHECK(hipModuleLoad(&module, fileName));
REQUIRE(module != nullptr);
HIP_CHECK(hipModuleUnload(module));
constexpr auto fileName = "vcpy_kernel.code";
hipModule_t module;
HIP_CHECK(hipModuleLoad(&module, fileName));
REQUIRE(module != nullptr);
HIP_CHECK(hipModuleUnload(module));
}
+7 -8
ファイルの表示
@@ -20,18 +20,17 @@ THE SOFTWARE.
constexpr int GLOBAL_BUF_SIZE = 2048;
__device__ float deviceGlobalFloat;
__device__ int deviceGlobalInt1;
__device__ int deviceGlobalInt2;
__device__ short deviceGlobalShort; //NOLINT
__device__ char deviceGlobalChar;
__device__ int deviceGlobalInt1;
__device__ int deviceGlobalInt2;
__device__ short deviceGlobalShort; // NOLINT
__device__ char deviceGlobalChar;
__device__ int getSquareOfGlobalFloat() {
return static_cast<int>(deviceGlobalFloat*deviceGlobalFloat);
return static_cast<int>(deviceGlobalFloat * deviceGlobalFloat);
}
extern "C" __global__ void testWeightedCopy(int* a, int* b) {
int tx = threadIdx.x;
b[tx] = deviceGlobalInt1 * a[tx] + deviceGlobalInt2 +
static_cast<int>(deviceGlobalShort) + static_cast<int>(deviceGlobalChar)
+ getSquareOfGlobalFloat();
b[tx] = deviceGlobalInt1 * a[tx] + deviceGlobalInt2 + static_cast<int>(deviceGlobalShort) +
static_cast<int>(deviceGlobalChar) + getSquareOfGlobalFloat();
}
+1 -3
ファイルの表示
@@ -19,6 +19,4 @@ THE SOFTWARE.
#include "hip/hip_runtime.h"
__managed__ int x = 10;
extern "C" __global__ void GPU_func() {
x++;
}
extern "C" __global__ void GPU_func() { x++; }
+11 -20
ファイルの表示
@@ -16,11 +16,10 @@ LIABILITY, WHETHER INN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include"hip/hip_runtime.h"
#include "hip/hip_runtime.h"
__device__ int deviceGlobal = 1;
extern "C" __global__ void matmulK(int clockrate, int* A, int* B, int* C,
int N) {
extern "C" __global__ void matmulK(int clockrate, int* A, int* B, int* C, int N) {
int ROW = blockIdx.y * blockDim.y + threadIdx.y;
int COL = blockIdx.x * blockDim.x + threadIdx.x;
int tmpSum = 0;
@@ -33,8 +32,7 @@ extern "C" __global__ void matmulK(int clockrate, int* A, int* B, int* C,
}
}
extern "C" __global__ void KernelandExtraParams(int* A, int* B, int* C,
int *D, int N) {
extern "C" __global__ void KernelandExtraParams(int* A, int* B, int* C, int* D, int N) {
int ROW = blockIdx.y * blockDim.y + threadIdx.y;
int COL = blockIdx.x * blockDim.x + threadIdx.x;
int tmpSum = 0;
@@ -50,31 +48,24 @@ extern "C" __global__ void KernelandExtraParams(int* A, int* B, int* C,
__device__ void Delay(uint32_t interval, const uint32_t ticks_per_ms) {
while (interval--) {
#if HT_AMD
#if HT_AMD
uint64_t start = wall_clock64();
while (wall_clock64() - start < ticks_per_ms) {
__builtin_amdgcn_s_sleep(10);
}
#endif
#if HT_NVIDIA
#endif
#if HT_NVIDIA
uint64_t start = clock64();
while (clock64() - start < ticks_per_ms) {
}
#endif
#endif
}
}
extern "C" __global__ void SixteenSecKernel(int clockrate) {
Delay(16000, clockrate);
}
extern "C" __global__ void SixteenSecKernel(int clockrate) { Delay(16000, clockrate); }
extern "C" __global__ void TwoSecKernel(int clockrate) {
Delay(2000, clockrate);
}
extern "C" __global__ void TwoSecKernel(int clockrate) { Delay(2000, clockrate); }
extern "C" __global__ void FourSecKernel(int clockrate) {
Delay(4000, clockrate);
}
extern "C" __global__ void FourSecKernel(int clockrate) { Delay(4000, clockrate); }
extern "C" __global__ void dummyKernel() {
}
extern "C" __global__ void dummyKernel() {}
+46 -72
ファイルの表示
@@ -20,24 +20,21 @@ THE SOFTWARE.
#include <fstream>
#include <cstddef>
#include <vector>
#include<iostream>
#define HIP_CHECK(error)\
{\
hipError_t localError = error;\
if ((localError != hipSuccess) && \
(localError != hipErrorPeerAccessAlreadyEnabled)) {\
printf("error: '%s'(%d) from %s at %s:%d\n", \
hipGetErrorString(localError), \
localError, #error, __FUNCTION__, __LINE__);\
exit(0);\
}\
}
#include <iostream>
#define HIP_CHECK(error) \
{ \
hipError_t localError = error; \
if ((localError != hipSuccess) && (localError != hipErrorPeerAccessAlreadyEnabled)) { \
printf("error: '%s'(%d) from %s at %s:%d\n", hipGetErrorString(localError), localError, \
#error, __FUNCTION__, __LINE__); \
exit(0); \
} \
}
constexpr auto CODEOBJ_FILE = "kernel_composite_test.code";
bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
char* globTestID) {
bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer, char* globTestID) {
constexpr auto CODEOBJ_GLOB_KERNEL1 = "testWeightedCopy";
size_t N = 16*16;
size_t N = 16 * 16;
size_t Nbytes = N * sizeof(int);
int *A_d, *B_d;
int *A_h, *B_h;
@@ -47,8 +44,8 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
HIP_CHECK(hipMalloc(&A_d, Nbytes));
HIP_CHECK(hipMalloc(&B_d, Nbytes));
A_h = reinterpret_cast<int *>(malloc(Nbytes));
B_h = reinterpret_cast<int *>(malloc(Nbytes));
A_h = reinterpret_cast<int*>(malloc(Nbytes));
B_h = reinterpret_cast<int*>(malloc(Nbytes));
// set host buffers
for (size_t idx = 0; idx < N; idx++) {
A_h[idx] = deviceid;
@@ -58,56 +55,37 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
hipModule_t Module;
hipFunction_t Function;
int check = atoi(globTestID);
/**
* Validates hipModuleLoadUnload if globTestID = 1
* Validates hipModuleLoadDataUnload if globTestID = 2
* Validates hipModuleLoadDataExUnload if globTestID = 3
*/
/**
* Validates hipModuleLoadUnload if globTestID = 1
* Validates hipModuleLoadDataUnload if globTestID = 2
* Validates hipModuleLoadDataExUnload if globTestID = 3
*/
switch (check) {
case 1:
HIP_CHECK(hipModuleLoad(&Module, CODEOBJ_FILE));
case 2:
HIP_CHECK(hipModuleLoadData(&Module, &buffer[0]));
case 3:
HIP_CHECK(hipModuleLoadDataEx(&Module,
&buffer[0], 0, nullptr, nullptr));
HIP_CHECK(hipModuleLoadDataEx(&Module, &buffer[0], 0, nullptr, nullptr));
}
HIP_CHECK(hipModuleGetFunction(&Function, Module,
CODEOBJ_GLOB_KERNEL1));
HIP_CHECK(hipModuleGetFunction(&Function, Module, CODEOBJ_GLOB_KERNEL1));
float deviceGlobalFloatH = 3.14;
int deviceGlobalInt1H = 100*deviceid;
int deviceGlobalInt2H = 50*deviceid;
uint32_t deviceGlobalShortH = 25*deviceid;
char deviceGlobalCharH = 13*deviceid;
int deviceGlobalInt1H = 100 * deviceid;
int deviceGlobalInt2H = 50 * deviceid;
uint32_t deviceGlobalShortH = 25 * deviceid;
char deviceGlobalCharH = 13 * deviceid;
hipDeviceptr_t deviceGlobal;
size_t deviceGlobalSize;
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
&deviceGlobalSize,
Module, "deviceGlobalFloat"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
&deviceGlobalFloatH,
deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
&deviceGlobalSize,
Module, "deviceGlobalInt1"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
&deviceGlobalInt1H,
deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
&deviceGlobalSize,
Module,
"deviceGlobalInt2"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
&deviceGlobalInt2H, deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
&deviceGlobalSize,
Module, "deviceGlobalShort"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
&deviceGlobalShortH, deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
&deviceGlobalSize, Module, "deviceGlobalChar"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
&deviceGlobalCharH, deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalFloat"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalFloatH, deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalInt1"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalInt1H, deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalInt2"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalInt2H, deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalShort"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalShortH, deviceGlobalSize));
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalChar"));
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalCharH, deviceGlobalSize));
// Launch Function kernel function
hipStream_t stream;
@@ -121,12 +99,10 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
args._Bd = reinterpret_cast<void*>(B_d);
size_t size = sizeof(args);
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
HIP_LAUNCH_PARAM_END};
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1,
N, 1, 1, 0, stream, NULL,
reinterpret_cast<void**>(&config)));
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, N, 1, 1, 0, stream, NULL,
reinterpret_cast<void**>(&config)));
// Copy buffer from decice to host
HIP_CHECK(hipMemcpyAsync(B_h, B_d, Nbytes, hipMemcpyDeviceToHost, stream));
HIP_CHECK(hipDeviceSynchronize());
@@ -134,13 +110,12 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
// Check the results
for (size_t idx = 0; idx < N; idx++) {
if (B_h[idx] != (deviceGlobalInt1H*A_h[idx]
+ deviceGlobalInt2H
+ static_cast<int>(deviceGlobalShortH) +
+ static_cast<int>(deviceGlobalCharH)
+ static_cast<int>(deviceGlobalFloatH*deviceGlobalFloatH))) {
// exit the current process with failure
return false;
if (B_h[idx] !=
(deviceGlobalInt1H * A_h[idx] + deviceGlobalInt2H + static_cast<int>(deviceGlobalShortH) +
+static_cast<int>(deviceGlobalCharH) +
static_cast<int>(deviceGlobalFloatH * deviceGlobalFloatH))) {
// exit the current process with failure
return false;
}
}
HIP_CHECK(hipModuleUnload(Module));
@@ -153,10 +128,9 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
return true;
}
int main(int argc, char* argv[]) {
if(argc > 0) {
if (argc > 0) {
bool value = false;
std::ifstream file(CODEOBJ_FILE,
std::ios::binary | std::ios::ate);
std::ifstream file(CODEOBJ_FILE, std::ios::binary | std::ios::ate);
std::streamsize fsize = file.tellg();
file.seekg(0, std::ios::beg);
std::vector<char> buffer(fsize);
+2 -2
ファイルの表示
@@ -19,6 +19,6 @@ THE SOFTWARE.
#include "hip/hip_runtime.h"
extern "C" __global__ void hello_world(float* a, float* b) {
int tx = threadIdx.x;
b[tx] = a[tx];
int tx = threadIdx.x;
b[tx] = a[tx];
}