SWDEV-470698 - fix formatting, add format check workflow (#657)
このコミットが含まれているのは:
@@ -21,15 +21,14 @@ THE SOFTWARE.
|
||||
|
||||
__device__ int globalDevData = 10;
|
||||
|
||||
extern "C" __global__ void addKernel(int *a, int size) {
|
||||
extern "C" __global__ void addKernel(int* a, int size) {
|
||||
int offset = blockDim.x * blockIdx.x + threadIdx.x;
|
||||
int stride = blockDim.x * gridDim.x;
|
||||
for (int i = offset; i < size; i+= stride) {
|
||||
for (int i = offset; i < size; i += stride) {
|
||||
a[i] += 2;
|
||||
}
|
||||
}
|
||||
|
||||
texture<float, 2> tex;
|
||||
|
||||
extern "C" __global__ void sampleModuleKernel() {
|
||||
}
|
||||
extern "C" __global__ void sampleModuleKernel() {}
|
||||
|
||||
@@ -24,15 +24,15 @@ THE SOFTWARE.
|
||||
using namespace cooperative_groups;
|
||||
extern "C" {
|
||||
__global__ void cooperativeKernelEx(int* output, int totalThreads) {
|
||||
grid_group grid = this_grid();
|
||||
int tid = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if (tid < totalThreads) {
|
||||
output[tid] = tid * 3;
|
||||
}
|
||||
grid.sync();
|
||||
if (tid == 0) {
|
||||
output[0] = 2222;
|
||||
}
|
||||
grid_group grid = this_grid();
|
||||
int tid = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if (tid < totalThreads) {
|
||||
output[tid] = tid * 3;
|
||||
}
|
||||
grid.sync();
|
||||
if (tid == 0) {
|
||||
output[0] = 2222;
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
@@ -44,7 +44,7 @@ __global__ void emptyKernel() {}
|
||||
* Kernel which doesn't use cooperative groups and takes an argument
|
||||
* and updates the value with 100
|
||||
*/
|
||||
__global__ void argKernel(int *val) { *val = 100; }
|
||||
__global__ void argKernel(int* val) { *val = 100; }
|
||||
|
||||
/*
|
||||
* Kernel which uses cooperative groups and without any arguments
|
||||
@@ -62,7 +62,7 @@ __global__ void coopEmptykernel() {
|
||||
* 2) Wait for all the blocks completes it's operations
|
||||
* 3) Fill each element in the output array with sum of elements in arr
|
||||
*/
|
||||
__global__ void coopFillArrayKernel(int *arr, int *output, int N) {
|
||||
__global__ void coopFillArrayKernel(int* arr, int* output, int N) {
|
||||
cooperative_groups::grid_group grid = cooperative_groups::this_grid();
|
||||
|
||||
if (blockIdx.x == 0)
|
||||
@@ -93,4 +93,3 @@ __global__ void coopFillArrayKernel(int *arr, int *output, int N) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -19,15 +19,15 @@ THE SOFTWARE.
|
||||
|
||||
#include "hip/hip_runtime.h"
|
||||
|
||||
extern "C" __global__ void
|
||||
kernelMultipleArgsSaxpy(int a1, int a2, int *x1, int b1, int b2, int *x2,
|
||||
int c1, int c2, int *x3, int d1, int d2, int *x4, int e1, int e2, int *x5,
|
||||
int f1, int f2, int *x6) {
|
||||
extern "C" __global__ void kernelMultipleArgsSaxpy(int a1, int a2, int* x1, int b1, int b2, int* x2,
|
||||
int c1, int c2, int* x3, int d1, int d2, int* x4,
|
||||
int e1, int e2, int* x5, int f1, int f2,
|
||||
int* x6) {
|
||||
int id = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
x1[id] = a1*x1[id] + a2;
|
||||
x2[id] = b1*x2[id] + b2;
|
||||
x3[id] = c1*x3[id] + c2;
|
||||
x4[id] = d1*x4[id] + d2;
|
||||
x5[id] = e1*x5[id] + e2;
|
||||
x6[id] = f1*x6[id] + f2;
|
||||
x1[id] = a1 * x1[id] + a2;
|
||||
x2[id] = b1 * x2[id] + b2;
|
||||
x3[id] = c1 * x3[id] + c2;
|
||||
x4[id] = d1 * x4[id] + d2;
|
||||
x5[id] = e1 * x5[id] + e2;
|
||||
x6[id] = f1 * x6[id] + f2;
|
||||
}
|
||||
|
||||
@@ -21,7 +21,7 @@ THE SOFTWARE.
|
||||
*/
|
||||
#include "hip/hip_runtime.h"
|
||||
|
||||
extern "C" __global__ void copy_ker(int* Ad, int *Bd, size_t size) {
|
||||
extern "C" __global__ void copy_ker(int* Ad, int* Bd, size_t size) {
|
||||
int myId = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if (myId < size) {
|
||||
Bd[myId] = Ad[myId];
|
||||
|
||||
@@ -25,4 +25,4 @@ THE SOFTWARE.
|
||||
|
||||
texture<float, 2> tex;
|
||||
|
||||
#endif // CUDA_VERSION < CUDA_12000
|
||||
#endif // CUDA_VERSION < CUDA_12000
|
||||
|
||||
@@ -55,7 +55,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
|
||||
int blockSize = 16;
|
||||
int numBlocks = (totalThreads + blockSize - 1) / blockSize;
|
||||
|
||||
int *d_output = nullptr;
|
||||
int* d_output = nullptr;
|
||||
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
|
||||
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
|
||||
|
||||
@@ -68,7 +68,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
|
||||
config.blockDimY = 1;
|
||||
config.blockDimZ = 1;
|
||||
config.sharedMemBytes = 0;
|
||||
config.hStream = 0; // default stream
|
||||
config.hStream = 0; // default stream
|
||||
|
||||
// Set up a cooperative launch attribute
|
||||
hipDrvLaunchAttribute attr;
|
||||
@@ -81,7 +81,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
|
||||
config.numAttrs = 1;
|
||||
|
||||
// Kernel parameters: address of d_output and totalThreads.
|
||||
void *kernelParams[] = {&d_output, &totalThreads};
|
||||
void* kernelParams[] = {&d_output, &totalThreads};
|
||||
|
||||
hipModule_t module;
|
||||
HIP_CHECK(hipModuleLoad(&module, CODE_OBJ_SINGLEARCH));
|
||||
@@ -98,8 +98,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
|
||||
hipErrorInvalidResourceHandle);
|
||||
}
|
||||
SECTION("Kernel parameter as nullptr") {
|
||||
HIP_CHECK_ERROR(hipDrvLaunchKernelEx(&config, function, nullptr, NULL),
|
||||
hipErrorInvalidValue);
|
||||
HIP_CHECK_ERROR(hipDrvLaunchKernelEx(&config, function, nullptr, NULL), hipErrorInvalidValue);
|
||||
}
|
||||
HIP_LAUNCH_CONFIG invalidConfig = {};
|
||||
invalidConfig.gridDimX = 0;
|
||||
@@ -109,7 +108,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
|
||||
invalidConfig.blockDimY = 1;
|
||||
invalidConfig.blockDimZ = 1;
|
||||
invalidConfig.sharedMemBytes = 0;
|
||||
invalidConfig.hStream = 0; // default stream
|
||||
invalidConfig.hStream = 0; // default stream
|
||||
|
||||
// Set up a cooperative launch attribute
|
||||
hipDrvLaunchAttribute invalidAttr;
|
||||
@@ -122,17 +121,16 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_NegTsts") {
|
||||
invalidConfig.numAttrs = 1;
|
||||
|
||||
SECTION("Invalid Kernel config") {
|
||||
HIP_CHECK_ERROR(
|
||||
hipDrvLaunchKernelEx(&invalidConfig, function, kernelParams, NULL),
|
||||
hipErrorInvalidConfiguration);
|
||||
HIP_CHECK_ERROR(hipDrvLaunchKernelEx(&invalidConfig, function, kernelParams, NULL),
|
||||
hipErrorInvalidConfiguration);
|
||||
}
|
||||
}
|
||||
|
||||
bool runTestDrvLaunch(const char *testName, std::string kernelFunc,
|
||||
int totalThreads, int blockSize, int flagValue) {
|
||||
bool runTestDrvLaunch(const char* testName, std::string kernelFunc, int totalThreads, int blockSize,
|
||||
int flagValue) {
|
||||
int numBlocks = (totalThreads + blockSize - 1) / blockSize;
|
||||
|
||||
int *d_output = nullptr;
|
||||
int* d_output = nullptr;
|
||||
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
|
||||
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
|
||||
|
||||
@@ -145,7 +143,7 @@ bool runTestDrvLaunch(const char *testName, std::string kernelFunc,
|
||||
config.blockDimY = 1;
|
||||
config.blockDimZ = 1;
|
||||
config.sharedMemBytes = 0;
|
||||
config.hStream = 0; // default stream
|
||||
config.hStream = 0; // default stream
|
||||
|
||||
// Set up a cooperative launch attribute
|
||||
hipDrvLaunchAttribute attr;
|
||||
@@ -158,7 +156,7 @@ bool runTestDrvLaunch(const char *testName, std::string kernelFunc,
|
||||
config.numAttrs = 1;
|
||||
|
||||
// Kernel parameters: address of d_output and totalThreads.
|
||||
void *kernelParams[] = {&d_output, &totalThreads};
|
||||
void* kernelParams[] = {&d_output, &totalThreads};
|
||||
|
||||
hipModule_t module;
|
||||
HIP_CHECK(hipModuleLoad(&module, CODE_OBJ_SINGLEARCH));
|
||||
@@ -176,22 +174,21 @@ bool runTestDrvLaunch(const char *testName, std::string kernelFunc,
|
||||
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
int *h_output = (int *)malloc(totalThreads * sizeof(int));
|
||||
HIP_CHECK(hipMemcpy(h_output, d_output, totalThreads * sizeof(int),
|
||||
hipMemcpyDeviceToHost));
|
||||
int* h_output = (int*)malloc(totalThreads * sizeof(int));
|
||||
HIP_CHECK(hipMemcpy(h_output, d_output, totalThreads * sizeof(int), hipMemcpyDeviceToHost));
|
||||
|
||||
// Verify results.
|
||||
bool success = true;
|
||||
if (h_output[0] != flagValue) {
|
||||
printf("%s test failed: Expected flag %d at index 0, got %d\n", testName,
|
||||
flagValue, h_output[0]);
|
||||
printf("%s test failed: Expected flag %d at index 0, got %d\n", testName, flagValue,
|
||||
h_output[0]);
|
||||
success = false;
|
||||
}
|
||||
for (int i = 1; i < totalThreads; i++) {
|
||||
int expectedValue = (flagValue == 1111) ? i : (i * 3);
|
||||
if (h_output[i] != expectedValue) {
|
||||
printf("%s test failed at index %d: Expected %d, got %d\n", testName, i,
|
||||
expectedValue, h_output[i]);
|
||||
printf("%s test failed at index %d: Expected %d, got %d\n", testName, i, expectedValue,
|
||||
h_output[i]);
|
||||
success = false;
|
||||
break;
|
||||
}
|
||||
@@ -217,8 +214,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_Functional") {
|
||||
HipTest::HIP_SKIP_TEST("CooperativeLaunch not supported");
|
||||
return;
|
||||
}
|
||||
REQUIRE(runTestDrvLaunch("hipDrvLaunchKernelEx", kernel_name, 64, 16, 2222) ==
|
||||
true);
|
||||
REQUIRE(runTestDrvLaunch("hipDrvLaunchKernelEx", kernel_name, 64, 16, 2222) == true);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -271,10 +267,10 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_Different_Kernels") {
|
||||
}
|
||||
|
||||
SECTION("Kernel with arguments using kernelParams") {
|
||||
int *devMem = nullptr;
|
||||
int* devMem = nullptr;
|
||||
HIP_CHECK(hipMalloc(&devMem, sizeof(int)));
|
||||
|
||||
void *kernel_args[1] = {&devMem};
|
||||
void* kernel_args[1] = {&devMem};
|
||||
|
||||
hipFunction_t argKernel;
|
||||
HIP_CHECK(hipModuleGetFunction(&argKernel, module, "argKernel"));
|
||||
@@ -343,15 +339,15 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_CooperativeKernelWithArgs") {
|
||||
hostMem[i] = 0;
|
||||
}
|
||||
|
||||
int *devMem1 = nullptr;
|
||||
int* devMem1 = nullptr;
|
||||
HIP_CHECK(hipMalloc(&devMem1, N * sizeof(int)));
|
||||
HIP_CHECK(hipMemcpy(devMem1, hostMem, N * sizeof(int), hipMemcpyDefault));
|
||||
int *devMem2 = nullptr;
|
||||
int* devMem2 = nullptr;
|
||||
HIP_CHECK(hipMalloc(&devMem2, N * sizeof(int)));
|
||||
HIP_CHECK(hipMemcpy(devMem2, hostMem, N * sizeof(int), hipMemcpyDefault));
|
||||
|
||||
int size = N;
|
||||
void *kernel_args[3] = {&devMem1, &devMem2, &size};
|
||||
void* kernel_args[3] = {&devMem1, &devMem2, &size};
|
||||
|
||||
hipFunction_t argKernel;
|
||||
HIP_CHECK(hipModuleGetFunction(&argKernel, module, "coopFillArrayKernel"));
|
||||
@@ -416,8 +412,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_MaxBlockDims") {
|
||||
config.numAttrs = 1;
|
||||
|
||||
SECTION("blockDim.x == maxBlockDimX") {
|
||||
const unsigned int x =
|
||||
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimX, 0);
|
||||
const unsigned int x = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimX, 0);
|
||||
config.blockDimX = x;
|
||||
|
||||
HIP_CHECK(hipDrvLaunchKernelEx(&config, kernel, nullptr, nullptr));
|
||||
@@ -425,8 +420,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_MaxBlockDims") {
|
||||
}
|
||||
|
||||
SECTION("blockDim.y == maxBlockDimY") {
|
||||
const unsigned int y =
|
||||
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimY, 0);
|
||||
const unsigned int y = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimY, 0);
|
||||
config.blockDimY = y;
|
||||
|
||||
HIP_CHECK(hipDrvLaunchKernelEx(&config, kernel, nullptr, nullptr));
|
||||
@@ -434,8 +428,7 @@ TEST_CASE("Unit_hipDrvLaunchKernelEx_With_MaxBlockDims") {
|
||||
}
|
||||
|
||||
SECTION("blockDim.z == maxBlockDimZ") {
|
||||
const unsigned int z =
|
||||
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimZ, 0);
|
||||
const unsigned int z = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimZ, 0);
|
||||
config.blockDimY = z;
|
||||
|
||||
HIP_CHECK(hipDrvLaunchKernelEx(&config, kernel, nullptr, nullptr));
|
||||
|
||||
@@ -48,25 +48,21 @@ THE SOFTWARE.
|
||||
__device__ int globalvar = 1;
|
||||
__device__ void Delay(uint32_t interval, const uint32_t ticks_per_ms) {
|
||||
while (interval--) {
|
||||
#if HT_AMD
|
||||
#if HT_AMD
|
||||
uint64_t start = wall_clock64();
|
||||
while (wall_clock64() - start < ticks_per_ms) {
|
||||
__builtin_amdgcn_s_sleep(10);
|
||||
}
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
uint64_t start = clock64();
|
||||
while (clock64() - start < ticks_per_ms) {
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
}
|
||||
__global__ void TwoSecKernel(int clockrate) {
|
||||
Delay(2000, clockrate);
|
||||
}
|
||||
__global__ void FourSecKernel(int clockrate) {
|
||||
Delay(4000, clockrate);
|
||||
}
|
||||
__global__ void TwoSecKernel(int clockrate) { Delay(2000, clockrate); }
|
||||
__global__ void FourSecKernel(int clockrate) { Delay(4000, clockrate); }
|
||||
|
||||
bool DisableTimeFlag() {
|
||||
bool testStatus = true;
|
||||
@@ -74,21 +70,19 @@ bool DisableTimeFlag() {
|
||||
HIP_CHECK(hipSetDevice(0));
|
||||
hipError_t e;
|
||||
float time_2sec;
|
||||
hipEvent_t start_event1, end_event1;
|
||||
hipEvent_t start_event1, end_event1;
|
||||
int clkRate = 0;
|
||||
#if HT_AMD
|
||||
#if HT_AMD
|
||||
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeWallClockRate, 0));
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeClockRate, 0));
|
||||
#endif
|
||||
HIP_CHECK(hipEventCreateWithFlags(&start_event1,
|
||||
hipEventDisableTiming));
|
||||
HIP_CHECK(hipEventCreateWithFlags(&end_event1,
|
||||
hipEventDisableTiming));
|
||||
#endif
|
||||
HIP_CHECK(hipEventCreateWithFlags(&start_event1, hipEventDisableTiming));
|
||||
HIP_CHECK(hipEventCreateWithFlags(&end_event1, hipEventDisableTiming));
|
||||
HIP_CHECK(hipStreamCreate(&stream1));
|
||||
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0,
|
||||
stream1, start_event1, end_event1, 0, clkRate);
|
||||
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0, stream1, start_event1, end_event1, 0,
|
||||
clkRate);
|
||||
HIP_CHECK(hipStreamSynchronize(stream1));
|
||||
e = hipEventElapsedTime(&time_2sec, start_event1, end_event1);
|
||||
if (e == hipErrorInvalidHandle) {
|
||||
@@ -108,12 +102,12 @@ bool ConcurencyCheck_GlobalVar(int conc_flag) {
|
||||
int deviceGlobal_h = 0;
|
||||
HIP_CHECK(hipSetDevice(0));
|
||||
int clkRate = 0;
|
||||
#if HT_AMD
|
||||
#if HT_AMD
|
||||
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeWallClockRate, 0));
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeClockRate, 0));
|
||||
#endif
|
||||
#endif
|
||||
HIP_CHECK(hipStreamCreate(&stream1));
|
||||
hipDeviceProp_t props{};
|
||||
int device;
|
||||
@@ -121,15 +115,14 @@ bool ConcurencyCheck_GlobalVar(int conc_flag) {
|
||||
HIP_CHECK(hipGetDeviceProperties(&props, device));
|
||||
if ((std::string(props.gcnArchName).find("gfx1101") != std::string::npos) ||
|
||||
(std::string(props.gcnArchName).find("gfx1100") != std::string::npos)) {
|
||||
hipExtLaunchKernelGGL((FourSecKernel), dim3(1), dim3(1), 0,
|
||||
stream1, nullptr, nullptr, conc_flag, clkRate);
|
||||
hipExtLaunchKernelGGL((FourSecKernel), dim3(1), dim3(1), 0, stream1, nullptr, nullptr,
|
||||
conc_flag, clkRate);
|
||||
} else {
|
||||
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0,
|
||||
stream1, nullptr, nullptr, conc_flag, clkRate);
|
||||
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0, stream1, nullptr, nullptr, conc_flag,
|
||||
clkRate);
|
||||
}
|
||||
HIP_CHECK(hipStreamSynchronize(stream1));
|
||||
HIP_CHECK(hipMemcpyFromSymbol(&deviceGlobal_h, globalvar,
|
||||
sizeof(int)));
|
||||
HIP_CHECK(hipMemcpyFromSymbol(&deviceGlobal_h, globalvar, sizeof(int)));
|
||||
|
||||
if (conc_flag && deviceGlobal_h != 0x5555) {
|
||||
testStatus = true;
|
||||
@@ -148,15 +141,15 @@ bool KernelTimeExecution() {
|
||||
bool testStatus = true;
|
||||
hipStream_t stream1;
|
||||
HIP_CHECK(hipSetDevice(0));
|
||||
hipEvent_t start_event1, end_event1, start_event2, end_event2;
|
||||
hipEvent_t start_event1, end_event1, start_event2, end_event2;
|
||||
float time_4sec, time_2sec;
|
||||
int clkRate = 0;
|
||||
#if HT_AMD
|
||||
#if HT_AMD
|
||||
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeWallClockRate, 0));
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeClockRate, 0));
|
||||
#endif
|
||||
#endif
|
||||
|
||||
HIP_CHECK(hipEventCreate(&start_event1));
|
||||
HIP_CHECK(hipEventCreate(&end_event1));
|
||||
@@ -167,16 +160,16 @@ bool KernelTimeExecution() {
|
||||
int device;
|
||||
HIP_CHECK(hipGetDevice(&device));
|
||||
HIP_CHECK(hipGetDeviceProperties(&props, device));
|
||||
hipExtLaunchKernelGGL((FourSecKernel), dim3(1), dim3(1), 0,
|
||||
stream1, start_event1, end_event1, 0, clkRate);
|
||||
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0,
|
||||
stream1, start_event2, end_event2, 0, clkRate);
|
||||
hipExtLaunchKernelGGL((FourSecKernel), dim3(1), dim3(1), 0, stream1, start_event1, end_event1, 0,
|
||||
clkRate);
|
||||
hipExtLaunchKernelGGL((TwoSecKernel), dim3(1), dim3(1), 0, stream1, start_event2, end_event2, 0,
|
||||
clkRate);
|
||||
HIP_CHECK(hipStreamSynchronize(stream1));
|
||||
HIP_CHECK(hipEventElapsedTime(&time_4sec, start_event1, end_event1));
|
||||
HIP_CHECK(hipEventElapsedTime(&time_2sec, start_event2, end_event2));
|
||||
|
||||
if ( (time_4sec < static_cast<float>(FIVESEC_KERNEL)) &&
|
||||
(time_2sec < static_cast<float>(THREESEC_KERNEL))) {
|
||||
if ((time_4sec < static_cast<float>(FIVESEC_KERNEL)) &&
|
||||
(time_2sec < static_cast<float>(THREESEC_KERNEL))) {
|
||||
testStatus = true;
|
||||
} else {
|
||||
testStatus = false;
|
||||
@@ -193,11 +186,11 @@ bool KernelTimeExecution() {
|
||||
|
||||
TEST_CASE("Unit_hipExtLaunchKernelGGL_Functional") {
|
||||
bool testStatus = true;
|
||||
// Disabled the concurency test as the firmware does not support concurrency
|
||||
// in the same stream
|
||||
#if 0
|
||||
// Disabled the concurency test as the firmware does not support concurrency
|
||||
// in the same stream
|
||||
#if 0
|
||||
testStatus &= ConcurencyCheck_GlobalVar(0);
|
||||
#endif
|
||||
#endif
|
||||
SECTION("Kernel Execution Time") {
|
||||
testStatus &= KernelTimeExecution();
|
||||
REQUIRE(testStatus == true);
|
||||
|
||||
@@ -20,15 +20,15 @@ THE SOFTWARE.
|
||||
#include <hip_test_defgroups.hh>
|
||||
|
||||
/**
|
||||
* @addtogroup hipExtLaunchMultiKernelMultiDevice
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipExtLaunchMultiKernelMultiDevice(hipLaunchParams* launchParamsList,
|
||||
* int numDevices, unsigned int flags)` -
|
||||
* Launches kernels on multiple devices and guarantees all specified kernels are dispatched
|
||||
* on respective streams before enqueuing any other work on the specified streams from any
|
||||
* other threads
|
||||
*/
|
||||
* @addtogroup hipExtLaunchMultiKernelMultiDevice
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipExtLaunchMultiKernelMultiDevice(hipLaunchParams* launchParamsList,
|
||||
* int numDevices, unsigned int flags)` -
|
||||
* Launches kernels on multiple devices and guarantees all specified kernels are dispatched
|
||||
* on respective streams before enqueuing any other work on the specified streams from any
|
||||
* other threads
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
@@ -44,8 +44,7 @@ THE SOFTWARE.
|
||||
|
||||
// Square each element in the array A and write to array C.
|
||||
#define NUM_KERNEL_ARGS 3
|
||||
__global__ void
|
||||
vector_square(float *C_d, float *A_d, size_t N) {
|
||||
__global__ void vector_square(float* C_d, float* A_d, size_t N) {
|
||||
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
|
||||
size_t stride = blockDim.x * gridDim.x;
|
||||
|
||||
@@ -76,7 +75,7 @@ TEST_CASE("Unit_hipExtLaunchMultiKernelMultiDevice_Functional") {
|
||||
HIP_CHECK(C_h == 0 ? hipErrorOutOfMemory : hipSuccess);
|
||||
// Fill with Phi + i
|
||||
for (size_t i = 0; i < N; i++) {
|
||||
A_h[i] = 1.618f + i;
|
||||
A_h[i] = 1.618f + i;
|
||||
}
|
||||
|
||||
const unsigned blocks = 512;
|
||||
@@ -97,22 +96,21 @@ TEST_CASE("Unit_hipExtLaunchMultiKernelMultiDevice_Functional") {
|
||||
HIP_CHECK(hipMemcpy(A_d[i], A_h, Nbytes, hipMemcpyHostToDevice));
|
||||
}
|
||||
|
||||
hipLaunchParams *launchParamsList = reinterpret_cast<hipLaunchParams *>(
|
||||
malloc(sizeof(hipLaunchParams)*nGpu));
|
||||
hipLaunchParams* launchParamsList =
|
||||
reinterpret_cast<hipLaunchParams*>(malloc(sizeof(hipLaunchParams) * nGpu));
|
||||
|
||||
void *args[MAX_GPUS * NUM_KERNEL_ARGS];
|
||||
void* args[MAX_GPUS * NUM_KERNEL_ARGS];
|
||||
|
||||
for (int i = 0; i < nGpu; i++) {
|
||||
args[i * NUM_KERNEL_ARGS] = &C_d[i];
|
||||
args[i * NUM_KERNEL_ARGS] = &C_d[i];
|
||||
args[i * NUM_KERNEL_ARGS + 1] = &A_d[i];
|
||||
args[i * NUM_KERNEL_ARGS + 2] = &N;
|
||||
launchParamsList[i].func =
|
||||
reinterpret_cast<void *>(vector_square);
|
||||
launchParamsList[i].gridDim = dim3(blocks);
|
||||
launchParamsList[i].blockDim = dim3(threadsPerBlock);
|
||||
launchParamsList[i].func = reinterpret_cast<void*>(vector_square);
|
||||
launchParamsList[i].gridDim = dim3(blocks);
|
||||
launchParamsList[i].blockDim = dim3(threadsPerBlock);
|
||||
launchParamsList[i].sharedMem = 0;
|
||||
launchParamsList[i].stream = stream[i];
|
||||
launchParamsList[i].args = args + i * NUM_KERNEL_ARGS;
|
||||
launchParamsList[i].stream = stream[i];
|
||||
launchParamsList[i].args = args + i * NUM_KERNEL_ARGS;
|
||||
}
|
||||
|
||||
INFO("info: launch vector_square kernel with")
|
||||
|
||||
@@ -156,14 +156,12 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_NonUniformWorkGroup") {
|
||||
args.buffersize = arraylength;
|
||||
size_t size = sizeof(args);
|
||||
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
// Memcpy from A to Ad
|
||||
HIP_CHECK(hipMemcpy(Ad, A, sizeBytes, hipMemcpyDefault));
|
||||
REQUIRE(hipErrorInvalidValue ==
|
||||
hipExtModuleLaunchKernel(Function, arraylength, 1, 1, localWorkSize,
|
||||
1, 1, 0, 0, NULL,
|
||||
hipExtModuleLaunchKernel(Function, arraylength, 1, 1, localWorkSize, 1, 1, 0, 0, NULL,
|
||||
reinterpret_cast<void**>(&config), 0));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
HIP_CHECK(hipFree(Ad));
|
||||
@@ -193,12 +191,8 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_UniformWorkGroup") {
|
||||
// Get module and function from module
|
||||
hipModule_t Module;
|
||||
hipFunction_t Function;
|
||||
SECTION("regular fatbin") {
|
||||
HIP_CHECK(hipModuleLoad(&Module, fileName));
|
||||
}
|
||||
SECTION("compressed fatbin") {
|
||||
HIP_CHECK(hipModuleLoad(&Module, fileNameCompressed));
|
||||
}
|
||||
SECTION("regular fatbin") { HIP_CHECK(hipModuleLoad(&Module, fileName)); }
|
||||
SECTION("compressed fatbin") { HIP_CHECK(hipModuleLoad(&Module, fileNameCompressed)); }
|
||||
SECTION("generic target in regular fatbin") {
|
||||
if (!isGenericTargetSupported()) {
|
||||
fprintf(stderr, "Generic target test is skipped\n");
|
||||
@@ -237,13 +231,11 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_UniformWorkGroup") {
|
||||
args.buffersize = arraylength;
|
||||
size_t size = sizeof(args);
|
||||
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
// Memcpy from A to Ad
|
||||
HIP_CHECK(hipMemcpy(Ad, A, sizeBytes, hipMemcpyDefault));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(Function, arraylength, 1, 1, localWorkSize,
|
||||
1, 1, 0, 0, NULL,
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(Function, arraylength, 1, 1, localWorkSize, 1, 1, 0, 0, NULL,
|
||||
reinterpret_cast<void**>(&config), 0));
|
||||
// Memcpy results back to host
|
||||
HIP_CHECK(hipMemcpy(B, Bd, sizeBytes, hipMemcpyDefault));
|
||||
@@ -266,8 +258,7 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_Positive_Parameters") {
|
||||
hipEvent_t start_event = nullptr;
|
||||
HIP_CHECK(hipEventCreate(&start_event));
|
||||
const auto kernel = GetKernel(mg.module(), "NOPKernel");
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(kernel, 1, 1, 1, 1, 1, 1, 0, nullptr,
|
||||
nullptr, nullptr,
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(kernel, 1, 1, 1, 1, 1, 1, 0, nullptr, nullptr, nullptr,
|
||||
start_event, nullptr));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
HIP_CHECK(hipEventQuery(start_event));
|
||||
@@ -278,8 +269,7 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_Positive_Parameters") {
|
||||
hipEvent_t stop_event = nullptr;
|
||||
HIP_CHECK(hipEventCreate(&stop_event));
|
||||
const auto kernel = GetKernel(mg.module(), "NOPKernel");
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(kernel, 1, 1, 1, 1, 1, 1, 0, nullptr,
|
||||
nullptr, nullptr,
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(kernel, 1, 1, 1, 1, 1, 1, 0, nullptr, nullptr, nullptr,
|
||||
nullptr, stop_event));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
HIP_CHECK(hipEventQuery(stop_event));
|
||||
@@ -294,9 +284,11 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_Negative_Parameters") {
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Test case to verify Negative tests of hipExtModuleLaunchKernel API.
|
||||
* - Test case to verify kernel execution time of the particular kernel by using hipExtModuleLaunchKernel.
|
||||
* - Test case to verify kernel execution time of the particular kernel by using
|
||||
hipExtModuleLaunchKernel.
|
||||
* - Test case to verify hipExtModuleLaunchKernel API by disabling time flag in event creation.
|
||||
* - Test case to verify hipExtModuleLaunchKernel API's Corner Scenarios for Grid and Block dimensions.
|
||||
* - Test case to verify hipExtModuleLaunchKernel API's Corner Scenarios for Grid and Block
|
||||
dimensions.
|
||||
* - Test case to verify different work groups of hipExtModuleLaunchKernel API.
|
||||
|
||||
* Test source
|
||||
@@ -317,16 +309,16 @@ struct gridblockDim {
|
||||
};
|
||||
class ModuleLaunchKernel {
|
||||
int N = 64;
|
||||
int SIZE = N*N;
|
||||
int SIZE = N * N;
|
||||
int *A, *B, *C;
|
||||
hipDeviceptr_t *Ad, *Bd;
|
||||
hipStream_t stream1, stream2;
|
||||
hipEvent_t start_event1, end_event1, start_event2, end_event2,
|
||||
start_timingDisabled, end_timingDisabled;
|
||||
hipEvent_t start_event1, end_event1, start_event2, end_event2, start_timingDisabled,
|
||||
end_timingDisabled;
|
||||
hipModule_t Module;
|
||||
hipDeviceptr_t deviceGlobal;
|
||||
hipFunction_t MultKernel, SixteenSecKernel, FourSecKernel,
|
||||
TwoSecKernel, KernelandExtraParamKernel, DummyKernel;
|
||||
hipFunction_t MultKernel, SixteenSecKernel, FourSecKernel, TwoSecKernel,
|
||||
KernelandExtraParamKernel, DummyKernel;
|
||||
struct {
|
||||
int clockRate;
|
||||
void* _Ad;
|
||||
@@ -340,7 +332,8 @@ class ModuleLaunchKernel {
|
||||
size_t size2;
|
||||
size_t size3;
|
||||
size_t deviceGlobalSize;
|
||||
public :
|
||||
|
||||
public:
|
||||
void AllocateMemory();
|
||||
void DeAllocateMemory();
|
||||
void ModuleLoad();
|
||||
@@ -355,37 +348,37 @@ class ModuleLaunchKernel {
|
||||
};
|
||||
|
||||
void ModuleLaunchKernel::AllocateMemory() {
|
||||
A = new int[N*N*sizeof(int)];
|
||||
B = new int[N*N*sizeof(int)];
|
||||
for (int i=0; i < N; i++) {
|
||||
for (int j=0; j < N; j++) {
|
||||
A[i*N +j] = 1;
|
||||
B[i*N +j] = 1;
|
||||
A = new int[N * N * sizeof(int)];
|
||||
B = new int[N * N * sizeof(int)];
|
||||
for (int i = 0; i < N; i++) {
|
||||
for (int j = 0; j < N; j++) {
|
||||
A[i * N + j] = 1;
|
||||
B[i * N + j] = 1;
|
||||
}
|
||||
}
|
||||
HIP_CHECK(hipStreamCreate(&stream1));
|
||||
HIP_CHECK(hipStreamCreate(&stream2));
|
||||
HIP_CHECK(hipMalloc(&Ad, SIZE*sizeof(int)));
|
||||
HIP_CHECK(hipMalloc(&Bd, SIZE*sizeof(int)));
|
||||
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&C), SIZE*sizeof(int)));
|
||||
HIP_CHECK(hipMemcpy(Ad, A, SIZE*sizeof(int), hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(Bd, B, SIZE*sizeof(int), hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMalloc(&Ad, SIZE * sizeof(int)));
|
||||
HIP_CHECK(hipMalloc(&Bd, SIZE * sizeof(int)));
|
||||
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&C), SIZE * sizeof(int)));
|
||||
HIP_CHECK(hipMemcpy(Ad, A, SIZE * sizeof(int), hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(Bd, B, SIZE * sizeof(int), hipMemcpyHostToDevice));
|
||||
int clkRate = 0;
|
||||
#if HT_AMD
|
||||
#if HT_AMD
|
||||
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeWallClockRate, 0));
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
HIP_CHECK(hipDeviceGetAttribute(&clkRate, hipDeviceAttributeClockRate, 0));
|
||||
#endif
|
||||
#endif
|
||||
args1._Ad = Ad;
|
||||
args1._Bd = Bd;
|
||||
args1._Cd = C;
|
||||
args1._n = N;
|
||||
args1._n = N;
|
||||
args1.clockRate = clkRate;
|
||||
args2._Ad = NULL;
|
||||
args2._Bd = NULL;
|
||||
args2._Cd = NULL;
|
||||
args2._n = 0;
|
||||
args2._n = 0;
|
||||
args2.clockRate = clkRate;
|
||||
size1 = sizeof(args1);
|
||||
size2 = sizeof(args2);
|
||||
@@ -394,16 +387,14 @@ void ModuleLaunchKernel::AllocateMemory() {
|
||||
HIP_CHECK(hipEventCreate(&end_event1));
|
||||
HIP_CHECK(hipEventCreate(&start_event2));
|
||||
HIP_CHECK(hipEventCreate(&end_event2));
|
||||
HIP_CHECK(hipEventCreateWithFlags(&start_timingDisabled,
|
||||
hipEventDisableTiming));
|
||||
HIP_CHECK(hipEventCreateWithFlags(&end_timingDisabled,
|
||||
hipEventDisableTiming));
|
||||
HIP_CHECK(hipEventCreateWithFlags(&start_timingDisabled, hipEventDisableTiming));
|
||||
HIP_CHECK(hipEventCreateWithFlags(&end_timingDisabled, hipEventDisableTiming));
|
||||
}
|
||||
|
||||
void ModuleLaunchKernel::ModuleLoad() {
|
||||
constexpr auto matmulName = "matmul.code";
|
||||
constexpr auto matmulK = "matmulK";
|
||||
constexpr auto SixteenSec = "SixteenSecKernel";
|
||||
constexpr auto matmulK = "matmulK";
|
||||
constexpr auto SixteenSec = "SixteenSecKernel";
|
||||
constexpr auto KernelandExtra = "KernelandExtraParams";
|
||||
constexpr auto FourSec = "FourSecKernel";
|
||||
constexpr auto TwoSec = "TwoSecKernel";
|
||||
@@ -413,13 +404,11 @@ void ModuleLaunchKernel::ModuleLoad() {
|
||||
HIP_CHECK(hipModuleLoad(&Module, matmulName));
|
||||
HIP_CHECK(hipModuleGetFunction(&MultKernel, Module, matmulK));
|
||||
HIP_CHECK(hipModuleGetFunction(&SixteenSecKernel, Module, SixteenSec));
|
||||
HIP_CHECK(hipModuleGetFunction(&KernelandExtraParamKernel,
|
||||
Module, KernelandExtra));
|
||||
HIP_CHECK(hipModuleGetFunction(&KernelandExtraParamKernel, Module, KernelandExtra));
|
||||
HIP_CHECK(hipModuleGetFunction(&FourSecKernel, Module, FourSec));
|
||||
HIP_CHECK(hipModuleGetFunction(&TwoSecKernel, Module, TwoSec));
|
||||
HIP_CHECK(hipModuleGetFunction(&DummyKernel, Module, dummyKernel));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize,
|
||||
Module, globalDevVar));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, globalDevVar));
|
||||
}
|
||||
|
||||
void ModuleLaunchKernel::DeAllocateMemory() {
|
||||
@@ -452,15 +441,14 @@ bool ModuleLaunchKernel::ExtModule_KernelExecutionTime() {
|
||||
ModuleLoad();
|
||||
float time_4sec, time_2sec;
|
||||
|
||||
void *config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
|
||||
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(FourSecKernel, 1, 1, 1, 1, 1, 1, 0,
|
||||
stream1, NULL, reinterpret_cast<void**>(&config2),
|
||||
start_event1, end_event1, 0));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
|
||||
NULL, reinterpret_cast<void**>(&config2),
|
||||
start_event2, end_event2, 0));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(FourSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2), start_event1, end_event1,
|
||||
0));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2), start_event2, end_event2,
|
||||
0));
|
||||
HIP_CHECK(hipStreamSynchronize(stream1));
|
||||
HIP_CHECK(hipEventElapsedTime(&time_4sec, start_event1, end_event1));
|
||||
HIP_CHECK(hipEventElapsedTime(&time_2sec, start_event2, end_event2));
|
||||
@@ -484,12 +472,11 @@ bool ModuleLaunchKernel::ExtModule_Disabled_Timingflag() {
|
||||
ModuleLoad();
|
||||
hipError_t e;
|
||||
float time_2sec;
|
||||
void *config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
|
||||
NULL, reinterpret_cast<void**>(&config2),
|
||||
start_timingDisabled, end_timingDisabled, 0));
|
||||
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2), start_timingDisabled,
|
||||
end_timingDisabled, 0));
|
||||
HIP_CHECK(hipStreamSynchronize(stream1));
|
||||
e = hipEventElapsedTime(&time_2sec, start_timingDisabled, end_timingDisabled);
|
||||
if (e == hipErrorInvalidHandle) {
|
||||
@@ -516,18 +503,16 @@ bool ModuleLaunchKernel::ExtModule_ConcurencyCheck_GlobalVar(int conc_flag) {
|
||||
int deviceGlobal_h = 0;
|
||||
AllocateMemory();
|
||||
ModuleLoad();
|
||||
void *config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
|
||||
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(FourSecKernel, 1, 1, 1, 1, 1, 1, 0,
|
||||
stream1, NULL, reinterpret_cast<void**>(&config2),
|
||||
start_event1, end_event1, conc_flag));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
|
||||
NULL, reinterpret_cast<void**>(&config2),
|
||||
start_event2, end_event2, conc_flag));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(FourSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2), start_event1, end_event1,
|
||||
conc_flag));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(TwoSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2), start_event2, end_event2,
|
||||
conc_flag));
|
||||
HIP_CHECK(hipStreamSynchronize(stream1));
|
||||
HIP_CHECK(hipMemcpyDtoH(&deviceGlobal_h, hipDeviceptr_t(deviceGlobal),
|
||||
deviceGlobalSize));
|
||||
HIP_CHECK(hipMemcpyDtoH(&deviceGlobal_h, hipDeviceptr_t(deviceGlobal), deviceGlobalSize));
|
||||
if (conc_flag && deviceGlobal_h != 0x5555) {
|
||||
testStatus = true;
|
||||
} else if (!conc_flag && deviceGlobal_h == 0x5555) {
|
||||
@@ -550,45 +535,32 @@ bool ModuleLaunchKernel::ExtModule_ConcurrencyCheck_TimeVer() {
|
||||
AllocateMemory();
|
||||
ModuleLoad();
|
||||
int mismatch = 0;
|
||||
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
|
||||
void* config2[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args2, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size2,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
auto start = std::chrono::high_resolution_clock::now();
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(SixteenSecKernel, 1, 1, 1, 1, 1, 1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2),
|
||||
NULL, NULL, 0));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(MultKernel, N, N, 1, 32, 32 , 1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
NULL, NULL, 0));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(SixteenSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2), NULL, NULL, 0));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(MultKernel, N, N, 1, 32, 32, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), NULL, NULL, 0));
|
||||
HIP_CHECK(hipStreamSynchronize(stream1));
|
||||
auto stop = std::chrono::high_resolution_clock::now();
|
||||
auto duration1 = std::chrono::duration_cast<std::chrono::microseconds>
|
||||
(stop-start);
|
||||
auto duration1 = std::chrono::duration_cast<std::chrono::microseconds>(stop - start);
|
||||
start = std::chrono::high_resolution_clock::now();
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(SixteenSecKernel, 1, 1, 1, 1, 1, 1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2),
|
||||
NULL, NULL, 1));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(MultKernel, N, N, 1, 32, 32, 1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
NULL, NULL, 1));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(SixteenSecKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config2), NULL, NULL, 1));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(MultKernel, N, N, 1, 32, 32, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), NULL, NULL, 1));
|
||||
HIP_CHECK(hipStreamSynchronize(stream1));
|
||||
stop = std::chrono::high_resolution_clock::now();
|
||||
auto duration2 = std::chrono::duration_cast<std::chrono::microseconds>
|
||||
(stop-start);
|
||||
auto duration2 = std::chrono::duration_cast<std::chrono::microseconds>(stop - start);
|
||||
if (!(duration2.count() < duration1.count())) {
|
||||
testStatus = false;
|
||||
}
|
||||
for (int i = 0; i < N; i++) {
|
||||
for (int j = 0; j < N; j++) {
|
||||
if (C[i*N + j] != N)
|
||||
mismatch++;
|
||||
if (C[i * N + j] != N) mismatch++;
|
||||
}
|
||||
}
|
||||
if (mismatch) {
|
||||
@@ -603,84 +575,57 @@ bool ModuleLaunchKernel::ExtModule_Negative_tests() {
|
||||
hipError_t err;
|
||||
AllocateMemory();
|
||||
ModuleLoad();
|
||||
void *config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
void *params[] = {Ad};
|
||||
void* params[] = {Ad};
|
||||
// Passing nullptr to kernel function in hipExtModuleLaunchKernel API
|
||||
err = hipExtModuleLaunchKernel(nullptr, 1, 1, 1, 1, 1, 1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(nullptr, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed nullptr to kernel function");
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing Max int value to block dimensions
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, std::numeric_limits<uint32_t>::max(),
|
||||
std::numeric_limits<uint32_t>::max(),
|
||||
std::numeric_limits<uint32_t>::max(),
|
||||
std::numeric_limits<uint32_t>::max(), 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
std::numeric_limits<uint32_t>::max(), 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for max values to block dimension");
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing 0 as value for all dimensions
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 0, 0, 0,
|
||||
0,
|
||||
0,
|
||||
0, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 0, 0, 0, 0, 0, 0, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for 0 as value for all dimensions");
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing 0 as value for x dimension
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 0, 1, 1,
|
||||
0,
|
||||
1,
|
||||
1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 0, 1, 1, 0, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for 0 as value for x dimension");
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing 0 as value for y dimension
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 0, 1,
|
||||
1,
|
||||
0,
|
||||
1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 0, 1, 1, 0, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for 0 as value for y dimension");
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing 0 as value for z dimension
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 0,
|
||||
1,
|
||||
1,
|
||||
0, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 0, 1, 1, 0, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for 0 as value for z dimension");
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing both kernel and extra params
|
||||
err = hipExtModuleLaunchKernel(KernelandExtraParamKernel, 1, 1, 1, 1, 1, 1, 0,
|
||||
stream1, reinterpret_cast<void**>(¶ms),
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(KernelandExtraParamKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
|
||||
reinterpret_cast<void**>(¶ms),
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel fail when we pass both kernel,extra args");
|
||||
testStatus = false;
|
||||
@@ -688,58 +633,44 @@ bool ModuleLaunchKernel::ExtModule_Negative_tests() {
|
||||
// Passing more than maxthreadsperblock to block dimensions
|
||||
hipDeviceProp_t deviceProp;
|
||||
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
deviceProp.maxThreadsPerBlock+1,
|
||||
deviceProp.maxThreadsPerBlock+1,
|
||||
deviceProp.maxThreadsPerBlock+1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, deviceProp.maxThreadsPerBlock + 1,
|
||||
deviceProp.maxThreadsPerBlock + 1,
|
||||
deviceProp.maxThreadsPerBlock + 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for max group size");
|
||||
testStatus = false;
|
||||
}
|
||||
// Block dimension X = Max Allowed + 1
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
deviceProp.maxThreadsDim[0]+1,
|
||||
1,
|
||||
1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, deviceProp.maxThreadsDim[0] + 1, 1, 1, 0,
|
||||
stream1, NULL, reinterpret_cast<void**>(&config1), nullptr,
|
||||
nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for (MaxBlockDimX + 1)");
|
||||
testStatus = false;
|
||||
}
|
||||
// Block dimension Y = Max Allowed + 1
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
1,
|
||||
deviceProp.maxThreadsDim[1]+1,
|
||||
1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, 1, deviceProp.maxThreadsDim[1] + 1, 1, 0,
|
||||
stream1, NULL, reinterpret_cast<void**>(&config1), nullptr,
|
||||
nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for (MaxBlockDimY + 1)");
|
||||
testStatus = false;
|
||||
}
|
||||
// Block dimension Z = Max Allowed + 1
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
1,
|
||||
1,
|
||||
deviceProp.maxThreadsDim[2]+1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, 1, 1, deviceProp.maxThreadsDim[2] + 1, 0,
|
||||
stream1, NULL, reinterpret_cast<void**>(&config1), nullptr,
|
||||
nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for (MaxBlockDimZ + 1)");
|
||||
testStatus = false;
|
||||
}
|
||||
|
||||
// Passing invalid config data in extra params
|
||||
void *config3[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
void* config3[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
err = hipExtModuleLaunchKernel(MultKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config3),
|
||||
nullptr, nullptr, 0);
|
||||
reinterpret_cast<void**>(&config3), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
INFO("hipExtModuleLaunchKernel failed for invalid conf");
|
||||
testStatus = false;
|
||||
@@ -754,8 +685,7 @@ bool ModuleLaunchKernel::ExtModule_Corner_tests() {
|
||||
hipError_t err;
|
||||
AllocateMemory();
|
||||
ModuleLoad();
|
||||
void *config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args3,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size3,
|
||||
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args3, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size3,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
hipDeviceProp_t deviceProp;
|
||||
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
|
||||
@@ -765,25 +695,14 @@ bool ModuleLaunchKernel::ExtModule_Corner_tests() {
|
||||
unsigned int maxgridX = deviceProp.maxGridSize[0];
|
||||
unsigned int maxgridY = deviceProp.maxGridSize[1];
|
||||
unsigned int maxgridZ = deviceProp.maxGridSize[2];
|
||||
struct gridblockDim test[6] = {{1, 1, 1, maxblockX, 1, 1},
|
||||
{1, 1, 1, 1, maxblockY, 1},
|
||||
{1, 1, 1, 1, 1, maxblockZ},
|
||||
{maxgridX, 1, 1, 1, 1, 1},
|
||||
{1, maxgridY, 1, 1, 1, 1},
|
||||
{1, 1, maxgridZ, 1, 1, 1}};
|
||||
struct gridblockDim test[6] = {{1, 1, 1, maxblockX, 1, 1}, {1, 1, 1, 1, maxblockY, 1},
|
||||
{1, 1, 1, 1, 1, maxblockZ}, {maxgridX, 1, 1, 1, 1, 1},
|
||||
{1, maxgridY, 1, 1, 1, 1}, {1, 1, maxgridZ, 1, 1, 1}};
|
||||
|
||||
for (int i = 0; i < 6; i++) {
|
||||
err = hipExtModuleLaunchKernel(DummyKernel,
|
||||
test[i].gridX,
|
||||
test[i].gridY,
|
||||
test[i].gridZ,
|
||||
test[i].blockX,
|
||||
test[i].blockY,
|
||||
test[i].blockZ,
|
||||
0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(DummyKernel, test[i].gridX, test[i].gridY, test[i].gridZ,
|
||||
test[i].blockX, test[i].blockY, test[i].blockZ, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err != hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
@@ -798,34 +717,26 @@ bool ModuleLaunchKernel::Module_WorkGroup_Test() {
|
||||
hipError_t err;
|
||||
AllocateMemory();
|
||||
ModuleLoad();
|
||||
void *config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args3,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size3,
|
||||
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args3, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size3,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
hipDeviceProp_t deviceProp;
|
||||
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
|
||||
double cuberootVal =
|
||||
cbrt(static_cast<double>(deviceProp.maxThreadsPerBlock));
|
||||
double cuberootVal = cbrt(static_cast<double>(deviceProp.maxThreadsPerBlock));
|
||||
uint32_t cuberoot_floor = floor(cuberootVal);
|
||||
uint32_t cuberoot_ceil = ceil(cuberootVal);
|
||||
// Scenario: (block.x * block.y * block.z) <= Work Group Size where
|
||||
// block.x < MaxBlockDimX , block.y < MaxBlockDimY and block.z < MaxBlockDimZ
|
||||
err = hipExtModuleLaunchKernel(DummyKernel,
|
||||
1, 1, 1,
|
||||
cuberoot_floor, cuberoot_floor, cuberoot_floor,
|
||||
0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(DummyKernel, 1, 1, 1, cuberoot_floor, cuberoot_floor,
|
||||
cuberoot_floor, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err != hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Scenario: (block.x * block.y * block.z) > Work Group Size where
|
||||
// block.x < MaxBlockDimX , block.y < MaxBlockDimY and block.z < MaxBlockDimZ
|
||||
err = hipExtModuleLaunchKernel(DummyKernel,
|
||||
1, 1, 1,
|
||||
cuberoot_ceil, cuberoot_ceil, cuberoot_ceil + 1,
|
||||
0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1),
|
||||
nullptr, nullptr, 0);
|
||||
err = hipExtModuleLaunchKernel(DummyKernel, 1, 1, 1, cuberoot_ceil, cuberoot_ceil,
|
||||
cuberoot_ceil + 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1), nullptr, nullptr, 0);
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
@@ -862,6 +773,6 @@ TEST_CASE("Unit_hipExtModuleLaunchKernel_Functional") {
|
||||
}
|
||||
}
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -20,12 +20,12 @@ THE SOFTWARE.
|
||||
#include <hip_test_defgroups.hh>
|
||||
|
||||
/**
|
||||
* @addtogroup hipFuncGetAttributes
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipFuncGetAttributes(struct hipFuncAttributes* attr, const void* func)` -
|
||||
* Find out attributes for a given function
|
||||
*/
|
||||
* @addtogroup hipFuncGetAttributes
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipFuncGetAttributes(struct hipFuncAttributes* attr, const void* func)` -
|
||||
* Find out attributes for a given function
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
@@ -48,8 +48,7 @@ __global__ void getAttrFn(float* px, float* py) {
|
||||
TEST_CASE("Unit_hipFuncGetAttributes_basic") {
|
||||
hipFuncAttributes attr{};
|
||||
|
||||
auto r = hipFuncGetAttributes(&attr,
|
||||
reinterpret_cast<const void*>(&getAttrFn));
|
||||
auto r = hipFuncGetAttributes(&attr, reinterpret_cast<const void*>(&getAttrFn));
|
||||
REQUIRE(r == hipSuccess);
|
||||
REQUIRE(attr.maxThreadsPerBlock != 0);
|
||||
}
|
||||
|
||||
@@ -20,12 +20,12 @@ THE SOFTWARE.
|
||||
#include <hip_test_defgroups.hh>
|
||||
|
||||
/**
|
||||
* @addtogroup hipFuncSetAttribute
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipFuncSetAttribute(const void* func, hipFuncAttribute attr, int value)` -
|
||||
* Set attributes for a specific function
|
||||
*/
|
||||
* @addtogroup hipFuncSetAttribute
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipFuncSetAttribute(const void* func, hipFuncAttribute attr, int value)` -
|
||||
* Set attributes for a specific function
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
@@ -47,9 +47,7 @@ __global__ void fn(float* px, float* py) {
|
||||
|
||||
TEST_CASE("Unit_hipFuncSetAttribute_Basic") {
|
||||
HIP_CHECK(hipFuncSetAttribute(reinterpret_cast<const void*>(&fn),
|
||||
hipFuncAttributeMaxDynamicSharedMemorySize,
|
||||
0));
|
||||
hipFuncAttributeMaxDynamicSharedMemorySize, 0));
|
||||
HIP_CHECK(hipFuncSetAttribute(reinterpret_cast<const void*>(&fn),
|
||||
hipFuncAttributePreferredSharedMemoryCarveout,
|
||||
0));
|
||||
hipFuncAttributePreferredSharedMemoryCarveout, 0));
|
||||
}
|
||||
|
||||
@@ -19,7 +19,7 @@ THE SOFTWARE.
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip_test_defgroups.hh>
|
||||
|
||||
__global__ void ReverseSeq(int *A, int *B, int N) {
|
||||
__global__ void ReverseSeq(int* A, int* B, int N) {
|
||||
extern __shared__ int SMem[];
|
||||
int offset = threadIdx.x;
|
||||
int MirrorVal = N - offset - 1;
|
||||
@@ -28,12 +28,12 @@ __global__ void ReverseSeq(int *A, int *B, int N) {
|
||||
B[offset] = SMem[MirrorVal];
|
||||
}
|
||||
/**
|
||||
* @addtogroup hipFuncSetSharedMemConfig
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipFuncSetSharedMemConfig(const void* func, hipSharedMemConfig config)` -
|
||||
* Sets shared memory configuation for a specific function
|
||||
*/
|
||||
* @addtogroup hipFuncSetSharedMemConfig
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipFuncSetSharedMemConfig(const void* func, hipSharedMemConfig config)` -
|
||||
* Sets shared memory configuation for a specific function
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
@@ -63,8 +63,8 @@ TEST_CASE("Unit_hipFuncSetSharedMemConfig_functional") {
|
||||
|
||||
// Testing hipFuncSetSharedMemConfig() with hipSharedMemBankSizeDefault flag
|
||||
SECTION("Flag: hipSharedMemBankSizeDefault") {
|
||||
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>
|
||||
(&ReverseSeq), hipSharedMemBankSizeDefault));
|
||||
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>(&ReverseSeq),
|
||||
hipSharedMemBankSizeDefault));
|
||||
// Kernel Launch with shared mem size of = NELMTS * sizeof(int)
|
||||
ReverseSeq<<<1, NELMTS, NELMTS * sizeof(int)>>>(Ad, RAd, NELMTS);
|
||||
memset(Ah, 0, NELMTS * sizeof(int));
|
||||
@@ -77,8 +77,8 @@ TEST_CASE("Unit_hipFuncSetSharedMemConfig_functional") {
|
||||
|
||||
// Testing hipFuncSetSharedMemConfig() with hipSharedMemBankSizeFourBytes flag
|
||||
SECTION("Flag: hipSharedMemBankSizeFourBytes") {
|
||||
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>
|
||||
(&ReverseSeq), hipSharedMemBankSizeFourByte));
|
||||
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>(&ReverseSeq),
|
||||
hipSharedMemBankSizeFourByte));
|
||||
HIP_CHECK(hipMemset(RAd, 0, NELMTS * sizeof(int)));
|
||||
// Kernel Launch with shared mem size of = NELMTS * sizeof(int)
|
||||
ReverseSeq<<<1, NELMTS, NELMTS * sizeof(int)>>>(Ad, RAd, NELMTS);
|
||||
@@ -91,8 +91,8 @@ TEST_CASE("Unit_hipFuncSetSharedMemConfig_functional") {
|
||||
}
|
||||
// Testing hipFuncSetSharedMemConfig() with hipSharedMemBankSizeEightBytes flg
|
||||
SECTION("Flag: hipSharedMemBankSizeEightByte") {
|
||||
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>
|
||||
(&ReverseSeq), hipSharedMemBankSizeEightByte));
|
||||
HIP_CHECK(hipFuncSetSharedMemConfig(reinterpret_cast<const void*>(&ReverseSeq),
|
||||
hipSharedMemBankSizeEightByte));
|
||||
HIP_CHECK(hipMemset(RAd, 0, NELMTS * sizeof(int)));
|
||||
// Kernel Launch with shared mem size of = NELMTS * sizeof(int)
|
||||
ReverseSeq<<<1, NELMTS, NELMTS * sizeof(int)>>>(Ad, RAd, NELMTS);
|
||||
|
||||
@@ -35,17 +35,16 @@ THE SOFTWARE.
|
||||
#define LEN 64
|
||||
#define SIZE LEN * sizeof(float)
|
||||
|
||||
#define ARR_SIZE (32*32)
|
||||
#define SIZE_BYTES (ARR_SIZE*sizeof(int))
|
||||
#define ARR_SIZE (32 * 32)
|
||||
#define SIZE_BYTES (ARR_SIZE * sizeof(int))
|
||||
|
||||
extern "C" __global__ void bit_extract_kernel(uint32_t* C_d, const uint32_t*
|
||||
A_d, size_t N) {
|
||||
extern "C" __global__ void bit_extract_kernel(uint32_t* C_d, const uint32_t* A_d, size_t N) {
|
||||
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
|
||||
size_t stride = blockDim.x * gridDim.x;
|
||||
for (size_t i = offset; i < N; i += stride) {
|
||||
#if HT_AMD
|
||||
C_d[i] = __bitextract_u32(A_d[i], 8, 4);
|
||||
#else /* defined __HIP_PLATFORM_NVIDIA__ or other path */
|
||||
#else /* defined __HIP_PLATFORM_NVIDIA__ or other path */
|
||||
C_d[i] = ((A_d[i] & 0xf00) >> 8);
|
||||
#endif
|
||||
}
|
||||
@@ -54,18 +53,16 @@ extern "C" __global__ void bit_extract_kernel(uint32_t* C_d, const uint32_t*
|
||||
/**
|
||||
* Host Function to check for negative case.
|
||||
*/
|
||||
__host__ void hostFunction() {
|
||||
printf("hostFunction\n");
|
||||
}
|
||||
__host__ void hostFunction() { printf("hostFunction\n"); }
|
||||
|
||||
/**
|
||||
* Sample Kernel to be used for functional test cases
|
||||
*/
|
||||
__global__ void hipKernel(int *a) {
|
||||
__global__ void hipKernel(int* a) {
|
||||
int offset = blockDim.x * blockIdx.x + threadIdx.x;
|
||||
int stride = blockDim.x * gridDim.x;
|
||||
|
||||
for (int i = offset; i < ARR_SIZE; i+= stride) {
|
||||
for (int i = offset; i < ARR_SIZE; i += stride) {
|
||||
a[i] += a[i];
|
||||
}
|
||||
}
|
||||
@@ -73,7 +70,7 @@ __global__ void hipKernel(int *a) {
|
||||
/**
|
||||
* Local Function to validate the result
|
||||
*/
|
||||
bool verifyResult(int *a, int *output_ref, int arrSize) {
|
||||
bool verifyResult(int* a, int* output_ref, int arrSize) {
|
||||
for (int i = 0; i < arrSize; i++) {
|
||||
if (a[i] != output_ref[i]) {
|
||||
return false;
|
||||
@@ -126,20 +123,19 @@ TEST_CASE("Unit_hipGetFuncBySymbol_PositiveTest") {
|
||||
void* _Ad;
|
||||
size_t _N;
|
||||
} args;
|
||||
args._Cd = reinterpret_cast<void**> (C_d);
|
||||
args._Ad = reinterpret_cast<void**> (A_d);
|
||||
args._N = static_cast<size_t> (N);
|
||||
args._Cd = reinterpret_cast<void**>(C_d);
|
||||
args._Ad = reinterpret_cast<void**>(A_d);
|
||||
args._N = static_cast<size_t>(N);
|
||||
size_t size = sizeof(args);
|
||||
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
|
||||
hipFunction_t Function;
|
||||
HIPCHECK(hipGetFuncBySymbol(&Function,
|
||||
reinterpret_cast<void*>(bit_extract_kernel)));
|
||||
HIPCHECK(hipGetFuncBySymbol(&Function, reinterpret_cast<void*>(bit_extract_kernel)));
|
||||
|
||||
HIPCHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0, 0, NULL,
|
||||
reinterpret_cast<void**>(&config)));
|
||||
reinterpret_cast<void**>(&config)));
|
||||
|
||||
HIPCHECK(hipMemcpyDtoH(C_h, (hipDeviceptr_t)(C_d), Nbytes));
|
||||
|
||||
@@ -177,8 +173,7 @@ TEST_CASE("Unit_hipGetFuncBySymbol_NegativeTests") {
|
||||
REQUIRE(hipGetFuncBySymbol(&funcPointer, NULL) != hipSuccess);
|
||||
|
||||
// Passing hostFunction as second parameter
|
||||
REQUIRE(hipGetFuncBySymbol(&funcPointer,
|
||||
reinterpret_cast<const void*>(hostFunction)));
|
||||
REQUIRE(hipGetFuncBySymbol(&funcPointer, reinterpret_cast<const void*>(hostFunction)));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -225,12 +220,12 @@ TEST_CASE("Unit_hipGetFuncBySymbol_MultiDev") {
|
||||
for (int deviceId = 0; deviceId < deviceCount; deviceId++) {
|
||||
HIP_CHECK(hipSetDevice(deviceId));
|
||||
|
||||
REQUIRE(hipGetFuncBySymbol(&funcPointer,
|
||||
reinterpret_cast<const void*>(hipKernel))== hipSuccess);
|
||||
REQUIRE(hipGetFuncBySymbol(&funcPointer, reinterpret_cast<const void*>(hipKernel)) ==
|
||||
hipSuccess);
|
||||
|
||||
int *h_a = reinterpret_cast<int *>(malloc(SIZE_BYTES));
|
||||
int* h_a = reinterpret_cast<int*>(malloc(SIZE_BYTES));
|
||||
REQUIRE(h_a != nullptr);
|
||||
int *output_ref = reinterpret_cast<int *>(malloc(SIZE_BYTES));
|
||||
int* output_ref = reinterpret_cast<int*>(malloc(SIZE_BYTES));
|
||||
REQUIRE(output_ref != nullptr);
|
||||
|
||||
for (int i = 0; i < ARR_SIZE; i++) {
|
||||
@@ -238,7 +233,7 @@ TEST_CASE("Unit_hipGetFuncBySymbol_MultiDev") {
|
||||
output_ref[i] = 4;
|
||||
}
|
||||
|
||||
int *d_a = nullptr;
|
||||
int* d_a = nullptr;
|
||||
HIP_CHECK(hipMalloc(&d_a, SIZE_BYTES));
|
||||
REQUIRE(d_a != nullptr);
|
||||
HIP_CHECK(hipMemcpy(d_a, h_a, SIZE_BYTES, hipMemcpyHostToDevice));
|
||||
@@ -249,13 +244,11 @@ TEST_CASE("Unit_hipGetFuncBySymbol_MultiDev") {
|
||||
void* kernelParam[] = {d_a};
|
||||
auto size = sizeof(kernelParam);
|
||||
void* kernel_parameter[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
|
||||
|
||||
REQUIRE(hipModuleLaunchKernel(funcPointer,
|
||||
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
|
||||
0, 0, nullptr, kernel_parameter) == hipSuccess);
|
||||
REQUIRE(hipModuleLaunchKernel(funcPointer, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z, 0, 0,
|
||||
nullptr, kernel_parameter) == hipSuccess);
|
||||
|
||||
HIP_CHECK(hipMemcpy(h_a, d_a, SIZE_BYTES, hipMemcpyDeviceToHost));
|
||||
|
||||
@@ -273,9 +266,9 @@ TEST_CASE("Unit_hipGetFuncBySymbol_MultiDev") {
|
||||
void MultiThreadMultiDevFunc(int DevId) {
|
||||
HIP_CHECK(hipSetDevice(DevId));
|
||||
|
||||
int *h_a = reinterpret_cast<int *>(malloc(SIZE_BYTES));
|
||||
int* h_a = reinterpret_cast<int*>(malloc(SIZE_BYTES));
|
||||
REQUIRE(h_a != nullptr);
|
||||
int *output_ref = reinterpret_cast<int *>(malloc(SIZE_BYTES));
|
||||
int* output_ref = reinterpret_cast<int*>(malloc(SIZE_BYTES));
|
||||
REQUIRE(output_ref != nullptr);
|
||||
|
||||
for (int i = 0; i < ARR_SIZE; i++) {
|
||||
@@ -287,32 +280,27 @@ void MultiThreadMultiDevFunc(int DevId) {
|
||||
HIP_CHECK(hipSetDevice(DevId));
|
||||
HIP_CHECK(hipStreamCreate(&stream));
|
||||
|
||||
int *d_a = nullptr;
|
||||
int* d_a = nullptr;
|
||||
HIP_CHECK(hipMalloc(&d_a, SIZE_BYTES));
|
||||
REQUIRE(d_a != nullptr);
|
||||
HIP_CHECK(hipMemcpyAsync(d_a, h_a, SIZE_BYTES,
|
||||
hipMemcpyHostToDevice, stream));
|
||||
HIP_CHECK(hipMemcpyAsync(d_a, h_a, SIZE_BYTES, hipMemcpyHostToDevice, stream));
|
||||
|
||||
dim3 blocksPerGrid(1, 1, 1);
|
||||
dim3 threadsPerBlock(1, 1, 64);
|
||||
|
||||
hipFunction_t funcPointer;
|
||||
REQUIRE(hipGetFuncBySymbol(&funcPointer,
|
||||
reinterpret_cast<const void*>(hipKernel))== hipSuccess);
|
||||
REQUIRE(hipGetFuncBySymbol(&funcPointer, reinterpret_cast<const void*>(hipKernel)) == hipSuccess);
|
||||
|
||||
void* kernelParam[] = {d_a};
|
||||
auto size = sizeof(kernelParam);
|
||||
void* kernel_parameter[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
|
||||
|
||||
REQUIRE(hipModuleLaunchKernel(funcPointer,
|
||||
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
|
||||
0, stream, nullptr, kernel_parameter) == hipSuccess);
|
||||
REQUIRE(hipModuleLaunchKernel(funcPointer, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z, 0, stream,
|
||||
nullptr, kernel_parameter) == hipSuccess);
|
||||
|
||||
HIP_CHECK(hipMemcpyAsync(h_a, d_a, SIZE_BYTES,
|
||||
hipMemcpyDeviceToHost, stream));
|
||||
HIP_CHECK(hipMemcpyAsync(h_a, d_a, SIZE_BYTES, hipMemcpyDeviceToHost, stream));
|
||||
|
||||
REQUIRE(verifyResult(h_a, output_ref, ARR_SIZE) == true);
|
||||
|
||||
|
||||
@@ -19,29 +19,27 @@ THE SOFTWARE.
|
||||
|
||||
#include "hip/hip_runtime.h"
|
||||
|
||||
#define ARR_SIZE (32*32)
|
||||
#define SIZE (ARR_SIZE*sizeof(int))
|
||||
#define ARR_SIZE (32 * 32)
|
||||
#define SIZE (ARR_SIZE * sizeof(int))
|
||||
|
||||
#define HIP_CHECK(error) \
|
||||
{ \
|
||||
hipError_t localError = error; \
|
||||
if ((localError != hipSuccess) && \
|
||||
(localError != hipErrorPeerAccessAlreadyEnabled)) { \
|
||||
printf("error: '%s'(%d) from %s at %s:%d\n", \
|
||||
hipGetErrorString(localError), \
|
||||
localError, #error, __FUNCTION__, __LINE__);\
|
||||
exit(0); \
|
||||
} \
|
||||
}
|
||||
#define HIP_CHECK(error) \
|
||||
{ \
|
||||
hipError_t localError = error; \
|
||||
if ((localError != hipSuccess) && (localError != hipErrorPeerAccessAlreadyEnabled)) { \
|
||||
printf("error: '%s'(%d) from %s at %s:%d\n", hipGetErrorString(localError), localError, \
|
||||
#error, __FUNCTION__, __LINE__); \
|
||||
exit(0); \
|
||||
} \
|
||||
}
|
||||
|
||||
/**
|
||||
* Sample Kernel to be used for functional test cases
|
||||
*/
|
||||
__global__ void hipKernel(int *a) {
|
||||
__global__ void hipKernel(int* a) {
|
||||
int offset = blockDim.x * blockIdx.x + threadIdx.x;
|
||||
int stride = blockDim.x * gridDim.x;
|
||||
|
||||
for (int i = offset; i < ARR_SIZE; i+= stride) {
|
||||
for (int i = offset; i < ARR_SIZE; i += stride) {
|
||||
a[i] += a[i];
|
||||
}
|
||||
}
|
||||
@@ -53,17 +51,16 @@ __global__ void hipKernel(int *a) {
|
||||
int main() {
|
||||
hipFunction_t funcPointer;
|
||||
|
||||
if (hipGetFuncBySymbol(&funcPointer,
|
||||
reinterpret_cast<const void*>(hipKernel)) != hipSuccess) {
|
||||
return -1;
|
||||
if (hipGetFuncBySymbol(&funcPointer, reinterpret_cast<const void*>(hipKernel)) != hipSuccess) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
int *h_a = reinterpret_cast<int *>(malloc(SIZE));
|
||||
int* h_a = reinterpret_cast<int*>(malloc(SIZE));
|
||||
if (h_a == nullptr) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
int *output_ref = reinterpret_cast<int *>(malloc(SIZE));
|
||||
int* output_ref = reinterpret_cast<int*>(malloc(SIZE));
|
||||
if (output_ref == nullptr) {
|
||||
return -1;
|
||||
}
|
||||
@@ -73,7 +70,7 @@ int main() {
|
||||
output_ref[i] = 4;
|
||||
}
|
||||
|
||||
int *d_a = nullptr;
|
||||
int* d_a = nullptr;
|
||||
HIP_CHECK(hipMalloc(&d_a, SIZE));
|
||||
if (d_a == nullptr) {
|
||||
return -1;
|
||||
@@ -86,14 +83,12 @@ int main() {
|
||||
void* kernelParam[] = {d_a};
|
||||
auto size = sizeof(kernelParam);
|
||||
void* kernel_parameter[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
|
||||
|
||||
if (hipModuleLaunchKernel(funcPointer,
|
||||
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
|
||||
0, 0, nullptr, kernel_parameter) != hipSuccess) {
|
||||
return -1;
|
||||
if (hipModuleLaunchKernel(funcPointer, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z, 0, 0, nullptr,
|
||||
kernel_parameter) != hipSuccess) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemcpy(h_a, d_a, SIZE, hipMemcpyDeviceToHost));
|
||||
|
||||
@@ -53,103 +53,65 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
|
||||
int currentHipVersion = 0;
|
||||
HIP_CHECK(hipRuntimeGetVersion(¤tHipVersion));
|
||||
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLoad",
|
||||
&hipModuleLoad_ptr,
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLoad", &hipModuleLoad_ptr, currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(
|
||||
hipGetProcAddress("hipModuleUnload", &hipModuleUnload_ptr, currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleGetFunction", &hipModuleGetFunction_ptr, currentHipVersion,
|
||||
0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLaunchKernel", &hipModuleLaunchKernel_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleUnload",
|
||||
&hipModuleUnload_ptr,
|
||||
HIP_CHECK(hipGetProcAddress("hipGetFuncBySymbol", &hipGetFuncBySymbol_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipFuncGetAttributes", &hipFuncGetAttributes_ptr, currentHipVersion,
|
||||
0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipFuncGetAttribute", &hipFuncGetAttribute_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleGetGlobal", &hipModuleGetGlobal_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipExtModuleLaunchKernel", &hipExtModuleLaunchKernel_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleGetFunction",
|
||||
&hipModuleGetFunction_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLaunchKernel",
|
||||
&hipModuleLaunchKernel_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipGetFuncBySymbol",
|
||||
&hipGetFuncBySymbol_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipFuncGetAttributes",
|
||||
&hipFuncGetAttributes_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipFuncGetAttribute",
|
||||
&hipFuncGetAttribute_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleGetGlobal",
|
||||
&hipModuleGetGlobal_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipExtModuleLaunchKernel",
|
||||
&hipExtModuleLaunchKernel_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipHccModuleLaunchKernel",
|
||||
&hipHccModuleLaunchKernel_ptr,
|
||||
HIP_CHECK(hipGetProcAddress("hipHccModuleLaunchKernel", &hipHccModuleLaunchKernel_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
|
||||
hipError_t (*dyn_hipModuleLoad_ptr)(hipModule_t *, const char *) =
|
||||
reinterpret_cast<hipError_t (*)(hipModule_t *, const char *)>
|
||||
(hipModuleLoad_ptr);
|
||||
hipError_t (*dyn_hipModuleLoad_ptr)(hipModule_t*, const char*) =
|
||||
reinterpret_cast<hipError_t (*)(hipModule_t*, const char*)>(hipModuleLoad_ptr);
|
||||
hipError_t (*dyn_hipModuleUnload_ptr)(hipModule_t) =
|
||||
reinterpret_cast<hipError_t (*)(hipModule_t)>
|
||||
(hipModuleUnload_ptr);
|
||||
hipError_t (*dyn_hipModuleGetFunction_ptr)(
|
||||
hipFunction_t *, hipModule_t, const char *) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t *,
|
||||
hipModule_t, const char *)>
|
||||
(hipModuleGetFunction_ptr);
|
||||
reinterpret_cast<hipError_t (*)(hipModule_t)>(hipModuleUnload_ptr);
|
||||
hipError_t (*dyn_hipModuleGetFunction_ptr)(hipFunction_t*, hipModule_t, const char*) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t*, hipModule_t, const char*)>(
|
||||
hipModuleGetFunction_ptr);
|
||||
hipError_t (*dyn_hipModuleLaunchKernel_ptr)(
|
||||
hipFunction_t,
|
||||
unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, hipStream_t,
|
||||
void **, void **) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t,
|
||||
unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, hipStream_t,
|
||||
void **, void **) > (hipModuleLaunchKernel_ptr);
|
||||
hipError_t (*dyn_hipGetFuncBySymbol_ptr)(hipFunction_t *, const void *) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t *, const void *)>
|
||||
(hipGetFuncBySymbol_ptr);
|
||||
hipError_t (*dyn_hipFuncGetAttributes_ptr)(
|
||||
struct hipFuncAttributes *, const void *) =
|
||||
reinterpret_cast<hipError_t (*)(struct hipFuncAttributes *, const void *)>
|
||||
(hipFuncGetAttributes_ptr);
|
||||
hipError_t (*dyn_hipFuncGetAttribute_ptr)(
|
||||
int *, hipFunction_attribute, hipFunction_t) =
|
||||
reinterpret_cast<hipError_t (*)(int *, hipFunction_attribute,
|
||||
hipFunction_t)>(hipFuncGetAttribute_ptr);
|
||||
hipError_t (*dyn_hipModuleGetGlobal_ptr)(
|
||||
hipDeviceptr_t *, size_t *, hipModule_t, const char *) =
|
||||
reinterpret_cast<hipError_t (*)(hipDeviceptr_t *, size_t *,
|
||||
hipModule_t, const char *)>
|
||||
(hipModuleGetGlobal_ptr);
|
||||
hipFunction_t, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, unsigned int, hipStream_t, void**, void**) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t, unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, unsigned int, unsigned int, unsigned int,
|
||||
hipStream_t, void**, void**)>(hipModuleLaunchKernel_ptr);
|
||||
hipError_t (*dyn_hipGetFuncBySymbol_ptr)(hipFunction_t*, const void*) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t*, const void*)>(hipGetFuncBySymbol_ptr);
|
||||
hipError_t (*dyn_hipFuncGetAttributes_ptr)(struct hipFuncAttributes*, const void*) =
|
||||
reinterpret_cast<hipError_t (*)(struct hipFuncAttributes*, const void*)>(
|
||||
hipFuncGetAttributes_ptr);
|
||||
hipError_t (*dyn_hipFuncGetAttribute_ptr)(int*, hipFunction_attribute, hipFunction_t) =
|
||||
reinterpret_cast<hipError_t (*)(int*, hipFunction_attribute, hipFunction_t)>(
|
||||
hipFuncGetAttribute_ptr);
|
||||
hipError_t (*dyn_hipModuleGetGlobal_ptr)(hipDeviceptr_t*, size_t*, hipModule_t, const char*) =
|
||||
reinterpret_cast<hipError_t (*)(hipDeviceptr_t*, size_t*, hipModule_t, const char*)>(
|
||||
hipModuleGetGlobal_ptr);
|
||||
|
||||
hipError_t (*dyn_hipExtModuleLaunchKernel_ptr)(hipFunction_t,
|
||||
uint32_t, uint32_t, uint32_t,
|
||||
uint32_t, uint32_t, uint32_t,
|
||||
size_t, hipStream_t,
|
||||
void **, void **,
|
||||
hipEvent_t, hipEvent_t, uint32_t) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t,
|
||||
uint32_t, uint32_t, uint32_t,
|
||||
uint32_t, uint32_t, uint32_t,
|
||||
size_t, hipStream_t,
|
||||
void **, void **,
|
||||
hipEvent_t, hipEvent_t, uint32_t)>
|
||||
(hipExtModuleLaunchKernel_ptr);
|
||||
hipError_t (*dyn_hipExtModuleLaunchKernel_ptr)(hipFunction_t, uint32_t, uint32_t, uint32_t,
|
||||
uint32_t, uint32_t, uint32_t, size_t, hipStream_t,
|
||||
void**, void**, hipEvent_t, hipEvent_t, uint32_t) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t, uint32_t, uint32_t, uint32_t, uint32_t,
|
||||
uint32_t, uint32_t, size_t, hipStream_t, void**, void**,
|
||||
hipEvent_t, hipEvent_t, uint32_t)>(
|
||||
hipExtModuleLaunchKernel_ptr);
|
||||
|
||||
hipError_t (*dyn_hipHccModuleLaunchKernel_ptr)(hipFunction_t,
|
||||
uint32_t, uint32_t, uint32_t,
|
||||
uint32_t, uint32_t, uint32_t,
|
||||
size_t, hipStream_t,
|
||||
void **, void **,
|
||||
hipEvent_t, hipEvent_t) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t,
|
||||
uint32_t, uint32_t, uint32_t,
|
||||
uint32_t, uint32_t, uint32_t,
|
||||
size_t, hipStream_t,
|
||||
void **, void **,
|
||||
hipEvent_t, hipEvent_t)>
|
||||
(hipHccModuleLaunchKernel_ptr);
|
||||
hipError_t (*dyn_hipHccModuleLaunchKernel_ptr)(hipFunction_t, uint32_t, uint32_t, uint32_t,
|
||||
uint32_t, uint32_t, uint32_t, size_t, hipStream_t,
|
||||
void**, void**, hipEvent_t, hipEvent_t) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t, uint32_t, uint32_t, uint32_t, uint32_t,
|
||||
uint32_t, uint32_t, size_t, hipStream_t, void**, void**,
|
||||
hipEvent_t, hipEvent_t)>(hipHccModuleLaunchKernel_ptr);
|
||||
|
||||
// Validating hipModuleLoad API
|
||||
hipModule_t module;
|
||||
@@ -165,11 +127,11 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
|
||||
const int N = 10;
|
||||
const int Nbytes = 10 * sizeof(int);
|
||||
|
||||
int *hostArr = reinterpret_cast<int *>(malloc(Nbytes));
|
||||
int* hostArr = reinterpret_cast<int*>(malloc(Nbytes));
|
||||
REQUIRE(hostArr != nullptr);
|
||||
fillHostArray(hostArr, N, 10);
|
||||
|
||||
int *devArr = nullptr;
|
||||
int* devArr = nullptr;
|
||||
HIP_CHECK(hipMalloc(&devArr, Nbytes));
|
||||
REQUIRE(devArr != nullptr);
|
||||
HIP_CHECK(hipMemcpy(devArr, hostArr, Nbytes, hipMemcpyHostToDevice));
|
||||
@@ -178,7 +140,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
|
||||
dim3 threadsPerBlock(1, 1, N);
|
||||
|
||||
struct kernelParameters {
|
||||
void *arr;
|
||||
void* arr;
|
||||
int size;
|
||||
};
|
||||
kernelParameters kernelParam{};
|
||||
@@ -186,48 +148,39 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
|
||||
kernelParam.size = N;
|
||||
|
||||
auto size = sizeof(kernelParam);
|
||||
void* kernel_parameter[] = { HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END };
|
||||
void* kernel_parameter[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &kernelParam,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size, HIP_LAUNCH_PARAM_END};
|
||||
|
||||
HIP_CHECK(dyn_hipModuleLaunchKernel_ptr(function,
|
||||
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
|
||||
0, 0, nullptr, kernel_parameter));
|
||||
HIP_CHECK(dyn_hipModuleLaunchKernel_ptr(function, blocksPerGrid.x, blocksPerGrid.y,
|
||||
blocksPerGrid.z, threadsPerBlock.x, threadsPerBlock.y,
|
||||
threadsPerBlock.z, 0, 0, nullptr, kernel_parameter));
|
||||
|
||||
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
|
||||
REQUIRE(validateHostArray(hostArr, N, 12) == true);
|
||||
|
||||
// Validating hipExtModuleLaunchKernel API
|
||||
HIP_CHECK(dyn_hipExtModuleLaunchKernel_ptr(function,
|
||||
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
|
||||
0, 0,
|
||||
nullptr, kernel_parameter,
|
||||
nullptr, nullptr, 0));
|
||||
HIP_CHECK(dyn_hipExtModuleLaunchKernel_ptr(
|
||||
function, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z, threadsPerBlock.x,
|
||||
threadsPerBlock.y, threadsPerBlock.z, 0, 0, nullptr, kernel_parameter, nullptr, nullptr, 0));
|
||||
|
||||
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
|
||||
REQUIRE(validateHostArray(hostArr, N, 14) == true);
|
||||
|
||||
// Validating hipHccModuleLaunchKernel API
|
||||
HIP_CHECK(dyn_hipHccModuleLaunchKernel_ptr(function,
|
||||
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
|
||||
0, 0,
|
||||
nullptr, kernel_parameter,
|
||||
nullptr, nullptr));
|
||||
HIP_CHECK(dyn_hipHccModuleLaunchKernel_ptr(
|
||||
function, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z, threadsPerBlock.x,
|
||||
threadsPerBlock.y, threadsPerBlock.z, 0, 0, nullptr, kernel_parameter, nullptr, nullptr));
|
||||
|
||||
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
|
||||
REQUIRE(validateHostArray(hostArr, N, 16) == true);
|
||||
|
||||
// Validating hipGetFuncBySymbol API
|
||||
hipFunction_t functionWithOrgApi, functionWithFuncPtr;
|
||||
HIP_CHECK(hipGetFuncBySymbol(&functionWithOrgApi,
|
||||
reinterpret_cast<const void*>(addOneKernel)));
|
||||
HIP_CHECK(hipGetFuncBySymbol(&functionWithOrgApi, reinterpret_cast<const void*>(addOneKernel)));
|
||||
REQUIRE(functionWithOrgApi != nullptr);
|
||||
|
||||
HIP_CHECK(dyn_hipGetFuncBySymbol_ptr(&functionWithFuncPtr,
|
||||
reinterpret_cast<const void*>(addOneKernel)));
|
||||
reinterpret_cast<const void*>(addOneKernel)));
|
||||
REQUIRE(functionWithFuncPtr != nullptr);
|
||||
|
||||
REQUIRE(functionWithFuncPtr == functionWithOrgApi);
|
||||
@@ -235,44 +188,38 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
|
||||
// Validating hipFuncGetAttributes API
|
||||
struct hipFuncAttributes attrWithOrgApi, attrWithFuncPtr;
|
||||
|
||||
HIP_CHECK(hipFuncGetAttributes(&attrWithOrgApi,
|
||||
reinterpret_cast<const void*>(addOneKernel)));
|
||||
HIP_CHECK(dyn_hipFuncGetAttributes_ptr(&attrWithFuncPtr,
|
||||
reinterpret_cast<const void*>(addOneKernel)));
|
||||
HIP_CHECK(hipFuncGetAttributes(&attrWithOrgApi, reinterpret_cast<const void*>(addOneKernel)));
|
||||
HIP_CHECK(
|
||||
dyn_hipFuncGetAttributes_ptr(&attrWithFuncPtr, reinterpret_cast<const void*>(addOneKernel)));
|
||||
|
||||
REQUIRE(attrWithFuncPtr.binaryVersion == attrWithOrgApi.binaryVersion);
|
||||
REQUIRE(attrWithFuncPtr.cacheModeCA == attrWithOrgApi.cacheModeCA);
|
||||
REQUIRE(attrWithFuncPtr.constSizeBytes == attrWithOrgApi.constSizeBytes);
|
||||
REQUIRE(attrWithFuncPtr.localSizeBytes == attrWithOrgApi.localSizeBytes);
|
||||
REQUIRE(attrWithFuncPtr.maxDynamicSharedSizeBytes ==
|
||||
attrWithOrgApi.maxDynamicSharedSizeBytes);
|
||||
REQUIRE(attrWithFuncPtr.maxThreadsPerBlock ==
|
||||
attrWithOrgApi.maxThreadsPerBlock);
|
||||
REQUIRE(attrWithFuncPtr.maxDynamicSharedSizeBytes == attrWithOrgApi.maxDynamicSharedSizeBytes);
|
||||
REQUIRE(attrWithFuncPtr.maxThreadsPerBlock == attrWithOrgApi.maxThreadsPerBlock);
|
||||
REQUIRE(attrWithFuncPtr.numRegs == attrWithOrgApi.numRegs);
|
||||
REQUIRE(attrWithFuncPtr.preferredShmemCarveout ==
|
||||
attrWithOrgApi.preferredShmemCarveout);
|
||||
REQUIRE(attrWithFuncPtr.preferredShmemCarveout == attrWithOrgApi.preferredShmemCarveout);
|
||||
REQUIRE(attrWithFuncPtr.ptxVersion == attrWithOrgApi.ptxVersion);
|
||||
REQUIRE(attrWithFuncPtr.sharedSizeBytes == attrWithOrgApi.sharedSizeBytes);
|
||||
|
||||
// Validating hipFuncGetAttribute API
|
||||
hipFunction_attribute attributes[] = {
|
||||
HIP_FUNC_ATTRIBUTE_MAX_THREADS_PER_BLOCK,
|
||||
HIP_FUNC_ATTRIBUTE_SHARED_SIZE_BYTES,
|
||||
HIP_FUNC_ATTRIBUTE_CONST_SIZE_BYTES,
|
||||
HIP_FUNC_ATTRIBUTE_LOCAL_SIZE_BYTES,
|
||||
HIP_FUNC_ATTRIBUTE_NUM_REGS,
|
||||
HIP_FUNC_ATTRIBUTE_PTX_VERSION,
|
||||
HIP_FUNC_ATTRIBUTE_BINARY_VERSION,
|
||||
HIP_FUNC_ATTRIBUTE_CACHE_MODE_CA,
|
||||
HIP_FUNC_ATTRIBUTE_MAX_DYNAMIC_SHARED_SIZE_BYTES,
|
||||
HIP_FUNC_ATTRIBUTE_PREFERRED_SHARED_MEMORY_CARVEOUT};
|
||||
hipFunction_attribute attributes[] = {HIP_FUNC_ATTRIBUTE_MAX_THREADS_PER_BLOCK,
|
||||
HIP_FUNC_ATTRIBUTE_SHARED_SIZE_BYTES,
|
||||
HIP_FUNC_ATTRIBUTE_CONST_SIZE_BYTES,
|
||||
HIP_FUNC_ATTRIBUTE_LOCAL_SIZE_BYTES,
|
||||
HIP_FUNC_ATTRIBUTE_NUM_REGS,
|
||||
HIP_FUNC_ATTRIBUTE_PTX_VERSION,
|
||||
HIP_FUNC_ATTRIBUTE_BINARY_VERSION,
|
||||
HIP_FUNC_ATTRIBUTE_CACHE_MODE_CA,
|
||||
HIP_FUNC_ATTRIBUTE_MAX_DYNAMIC_SHARED_SIZE_BYTES,
|
||||
HIP_FUNC_ATTRIBUTE_PREFERRED_SHARED_MEMORY_CARVEOUT};
|
||||
|
||||
for ( auto attribute : attributes ) {
|
||||
for (auto attribute : attributes) {
|
||||
int valuewithOrgAPI = 0, valueWithFuncPointer = 0;
|
||||
|
||||
HIP_CHECK(hipFuncGetAttribute(&valuewithOrgAPI, attribute, function));
|
||||
HIP_CHECK(dyn_hipFuncGetAttribute_ptr(&valueWithFuncPointer, attribute,
|
||||
function));
|
||||
HIP_CHECK(dyn_hipFuncGetAttribute_ptr(&valueWithFuncPointer, attribute, function));
|
||||
|
||||
REQUIRE(valueWithFuncPointer == valuewithOrgAPI);
|
||||
}
|
||||
@@ -280,14 +227,13 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApis") {
|
||||
// Validating hipModuleGetGlobal API
|
||||
hipDeviceptr_t dptrWithOrgApi = nullptr;
|
||||
size_t bytesWithOrgApi = 0;
|
||||
HIP_CHECK(hipModuleGetGlobal(&dptrWithOrgApi, &bytesWithOrgApi,
|
||||
module, "globalDevData"));
|
||||
HIP_CHECK(hipModuleGetGlobal(&dptrWithOrgApi, &bytesWithOrgApi, module, "globalDevData"));
|
||||
REQUIRE(dptrWithOrgApi != nullptr);
|
||||
|
||||
hipDeviceptr_t dptrWithFuncPtr = nullptr;
|
||||
size_t bytesWithFuncPtr = 0;
|
||||
HIP_CHECK(dyn_hipModuleGetGlobal_ptr(&dptrWithFuncPtr, &bytesWithFuncPtr,
|
||||
module, "globalDevData") );
|
||||
HIP_CHECK(
|
||||
dyn_hipModuleGetGlobal_ptr(&dptrWithFuncPtr, &bytesWithFuncPtr, module, "globalDevData"));
|
||||
REQUIRE(dptrWithFuncPtr != nullptr);
|
||||
REQUIRE(bytesWithFuncPtr == 4);
|
||||
|
||||
@@ -323,24 +269,19 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisLoadData") {
|
||||
int currentHipVersion = 0;
|
||||
HIP_CHECK(hipRuntimeGetVersion(¤tHipVersion));
|
||||
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLoadData",
|
||||
&hipModuleLoadData_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLoadDataEx",
|
||||
&hipModuleLoadDataEx_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLoadData", &hipModuleLoadData_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLoadDataEx", &hipModuleLoadDataEx_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
|
||||
hipError_t (*dyn_hipModuleLoadData_ptr)(hipModule_t *, const void *) =
|
||||
reinterpret_cast<hipError_t (*)(hipModule_t *, const void *)>
|
||||
(hipModuleLoadData_ptr);
|
||||
hipError_t (*dyn_hipModuleLoadDataEx_ptr)(hipModule_t *, const void *,
|
||||
unsigned int, hipJitOption *, void **) =
|
||||
reinterpret_cast<hipError_t (*)(hipModule_t *, const void *,
|
||||
unsigned int, hipJitOption *, void **)>
|
||||
(hipModuleLoadDataEx_ptr);
|
||||
hipError_t (*dyn_hipModuleLoadData_ptr)(hipModule_t*, const void*) =
|
||||
reinterpret_cast<hipError_t (*)(hipModule_t*, const void*)>(hipModuleLoadData_ptr);
|
||||
hipError_t (*dyn_hipModuleLoadDataEx_ptr)(hipModule_t*, const void*, unsigned int, hipJitOption*,
|
||||
void**) =
|
||||
reinterpret_cast<hipError_t (*)(hipModule_t*, const void*, unsigned int, hipJitOption*,
|
||||
void**)>(hipModuleLoadDataEx_ptr);
|
||||
|
||||
const auto rtc = CreateRTCCharArray(
|
||||
R"(extern "C" __global__ void simpleKernel() {})");
|
||||
const auto rtc = CreateRTCCharArray(R"(extern "C" __global__ void simpleKernel() {})");
|
||||
|
||||
// Validating hipModuleLoadData API
|
||||
{
|
||||
@@ -352,9 +293,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisLoadData") {
|
||||
hipFunction_t function;
|
||||
HIP_CHECK(hipModuleGetFunction(&function, module, "simpleKernel"));
|
||||
REQUIRE(function != nullptr);
|
||||
HIP_CHECK(hipModuleLaunchKernel(function,
|
||||
1, 1, 1, 1, 1, 1,
|
||||
0, 0, nullptr, nullptr));
|
||||
HIP_CHECK(hipModuleLaunchKernel(function, 1, 1, 1, 1, 1, 1, 0, 0, nullptr, nullptr));
|
||||
|
||||
HIP_CHECK(hipModuleUnload(module));
|
||||
}
|
||||
@@ -363,22 +302,19 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisLoadData") {
|
||||
{
|
||||
hipModule_t module = nullptr;
|
||||
|
||||
HIP_CHECK(dyn_hipModuleLoadDataEx_ptr(&module, rtc.data(),
|
||||
0, nullptr, nullptr));
|
||||
HIP_CHECK(dyn_hipModuleLoadDataEx_ptr(&module, rtc.data(), 0, nullptr, nullptr));
|
||||
REQUIRE(module != nullptr);
|
||||
|
||||
hipFunction_t function;
|
||||
HIP_CHECK(hipModuleGetFunction(&function, module, "simpleKernel"));
|
||||
REQUIRE(function != nullptr);
|
||||
HIP_CHECK(hipModuleLaunchKernel(function,
|
||||
1, 1, 1, 1, 1, 1,
|
||||
0, 0, nullptr, nullptr));
|
||||
HIP_CHECK(hipModuleLaunchKernel(function, 1, 1, 1, 1, 1, 1, 0, 0, nullptr, nullptr));
|
||||
|
||||
HIP_CHECK(hipModuleUnload(module));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - This test will get the function pointer of different module management
|
||||
@@ -398,77 +334,63 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
return;
|
||||
}
|
||||
|
||||
void *hipModuleLaunchCooperativeKernel_ptr = nullptr;
|
||||
void *hipModuleLaunchCooperativeKernelMultiDevice_ptr = nullptr;
|
||||
void *hipLaunchCooperativeKernel_ptr = nullptr;
|
||||
void *hipLaunchCooperativeKernelMultiDevice_ptr = nullptr;
|
||||
void *hipExtLaunchMultiKernelMultiDevice_ptr = nullptr;
|
||||
void* hipModuleLaunchCooperativeKernel_ptr = nullptr;
|
||||
void* hipModuleLaunchCooperativeKernelMultiDevice_ptr = nullptr;
|
||||
void* hipLaunchCooperativeKernel_ptr = nullptr;
|
||||
void* hipLaunchCooperativeKernelMultiDevice_ptr = nullptr;
|
||||
void* hipExtLaunchMultiKernelMultiDevice_ptr = nullptr;
|
||||
|
||||
int currentHipVersion = 0;
|
||||
HIP_CHECK(hipRuntimeGetVersion(¤tHipVersion));
|
||||
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipModuleLaunchCooperativeKernel",
|
||||
&hipModuleLaunchCooperativeKernel_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipModuleLaunchCooperativeKernelMultiDevice",
|
||||
&hipModuleLaunchCooperativeKernelMultiDevice_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipLaunchCooperativeKernel",
|
||||
&hipLaunchCooperativeKernel_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipLaunchCooperativeKernelMultiDevice",
|
||||
&hipLaunchCooperativeKernelMultiDevice_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipExtLaunchMultiKernelMultiDevice",
|
||||
&hipExtLaunchMultiKernelMultiDevice_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLaunchCooperativeKernel",
|
||||
&hipModuleLaunchCooperativeKernel_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleLaunchCooperativeKernelMultiDevice",
|
||||
&hipModuleLaunchCooperativeKernelMultiDevice_ptr, currentHipVersion,
|
||||
0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipLaunchCooperativeKernel", &hipLaunchCooperativeKernel_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipLaunchCooperativeKernelMultiDevice",
|
||||
&hipLaunchCooperativeKernelMultiDevice_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipExtLaunchMultiKernelMultiDevice",
|
||||
&hipExtLaunchMultiKernelMultiDevice_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
|
||||
hipError_t (*dyn_hipModuleLaunchCooperativeKernel_ptr)(
|
||||
hipFunction_t,
|
||||
unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, hipStream_t, void **) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t,
|
||||
unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, hipStream_t, void **)>
|
||||
(hipModuleLaunchCooperativeKernel_ptr);
|
||||
hipFunction_t, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, unsigned int, hipStream_t, void**) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunction_t, unsigned int, unsigned int, unsigned int,
|
||||
unsigned int, unsigned int, unsigned int, unsigned int,
|
||||
hipStream_t, void**)>(hipModuleLaunchCooperativeKernel_ptr);
|
||||
|
||||
hipError_t (*dyn_hipModuleLaunchCooperativeKernelMultiDevice_ptr)(
|
||||
hipFunctionLaunchParams *, unsigned int, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunctionLaunchParams *,
|
||||
unsigned int, unsigned int)>
|
||||
(hipModuleLaunchCooperativeKernelMultiDevice_ptr);
|
||||
hipError_t (*dyn_hipModuleLaunchCooperativeKernelMultiDevice_ptr)(hipFunctionLaunchParams*,
|
||||
unsigned int, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(hipFunctionLaunchParams*, unsigned int, unsigned int)>(
|
||||
hipModuleLaunchCooperativeKernelMultiDevice_ptr);
|
||||
|
||||
hipError_t (*dyn_hipLaunchCooperativeKernel_ptr)(
|
||||
const void *, dim3, dim3, void **, unsigned int, hipStream_t) =
|
||||
reinterpret_cast<hipError_t (*)(const void *, dim3, dim3, void **,
|
||||
unsigned int, hipStream_t)>
|
||||
(hipLaunchCooperativeKernel_ptr);
|
||||
hipError_t (*dyn_hipLaunchCooperativeKernel_ptr)(const void*, dim3, dim3, void**, unsigned int,
|
||||
hipStream_t) =
|
||||
reinterpret_cast<hipError_t (*)(const void*, dim3, dim3, void**, unsigned int, hipStream_t)>(
|
||||
hipLaunchCooperativeKernel_ptr);
|
||||
|
||||
hipError_t (*dyn_hipLaunchCooperativeKernelMultiDevice_ptr)(
|
||||
hipLaunchParams *, int, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(hipLaunchParams *, int, unsigned int)>
|
||||
(hipLaunchCooperativeKernelMultiDevice_ptr);
|
||||
hipError_t (*dyn_hipLaunchCooperativeKernelMultiDevice_ptr)(hipLaunchParams*, int, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(hipLaunchParams*, int, unsigned int)>(
|
||||
hipLaunchCooperativeKernelMultiDevice_ptr);
|
||||
|
||||
hipError_t (*dyn_hipExtLaunchMultiKernelMultiDevice_ptr)(
|
||||
hipLaunchParams *, int, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(hipLaunchParams *, int, unsigned int)>
|
||||
(hipExtLaunchMultiKernelMultiDevice_ptr);
|
||||
hipError_t (*dyn_hipExtLaunchMultiKernelMultiDevice_ptr)(hipLaunchParams*, int, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(hipLaunchParams*, int, unsigned int)>(
|
||||
hipExtLaunchMultiKernelMultiDevice_ptr);
|
||||
|
||||
const int N = 10;
|
||||
const int Nbytes = 10 * sizeof(int);
|
||||
|
||||
int *hostArr = reinterpret_cast<int *>(malloc(Nbytes));
|
||||
int* hostArr = reinterpret_cast<int*>(malloc(Nbytes));
|
||||
REQUIRE(hostArr != nullptr);
|
||||
fillHostArray(hostArr, N, 10);
|
||||
|
||||
int *devArr = nullptr;
|
||||
int* devArr = nullptr;
|
||||
HIP_CHECK(hipMalloc(&devArr, Nbytes));
|
||||
REQUIRE(devArr != nullptr);
|
||||
HIP_CHECK(hipMemcpy(devArr, hostArr, Nbytes, hipMemcpyHostToDevice));
|
||||
@@ -477,13 +399,13 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
dim3 threadsPerBlock(1, 1, N);
|
||||
|
||||
struct kernelParameters {
|
||||
void *arr;
|
||||
void* arr;
|
||||
int size;
|
||||
};
|
||||
kernelParameters kernelParam;
|
||||
kernelParam.arr = devArr;
|
||||
kernelParam.size = N;
|
||||
void *kernel_parameter[] = {&kernelParam.arr, &kernelParam.size};
|
||||
void* kernel_parameter[] = {&kernelParam.arr, &kernelParam.size};
|
||||
|
||||
// Validating hipModuleLaunchCooperativeKernel API
|
||||
{
|
||||
@@ -495,10 +417,9 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
HIP_CHECK(hipModuleGetFunction(&function, module, "addKernel"));
|
||||
REQUIRE(function != nullptr);
|
||||
|
||||
HIP_CHECK(dyn_hipModuleLaunchCooperativeKernel_ptr(function,
|
||||
blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z,
|
||||
threadsPerBlock.x, threadsPerBlock.y, threadsPerBlock.z,
|
||||
0, 0, kernel_parameter));
|
||||
HIP_CHECK(dyn_hipModuleLaunchCooperativeKernel_ptr(
|
||||
function, blocksPerGrid.x, blocksPerGrid.y, blocksPerGrid.z, threadsPerBlock.x,
|
||||
threadsPerBlock.y, threadsPerBlock.z, 0, 0, kernel_parameter));
|
||||
|
||||
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
|
||||
REQUIRE(validateHostArray(hostArr, N, 12) == true);
|
||||
@@ -510,9 +431,9 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
int deviceCount = 0;
|
||||
HIP_CHECK(hipGetDeviceCount(&deviceCount));
|
||||
|
||||
hipModule_t *module = new hipModule_t[deviceCount];
|
||||
hipFunction_t *function = new hipFunction_t[deviceCount];
|
||||
hipStream_t *streamArr = new hipStream_t[deviceCount];
|
||||
hipModule_t* module = new hipModule_t[deviceCount];
|
||||
hipFunction_t* function = new hipFunction_t[deviceCount];
|
||||
hipStream_t* streamArr = new hipStream_t[deviceCount];
|
||||
|
||||
for (int i = 0; i < deviceCount; ++i) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
@@ -521,8 +442,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
HIP_CHECK(hipModuleLoad(&module[i], "addKernel.code"));
|
||||
REQUIRE(module[i] != nullptr);
|
||||
|
||||
HIP_CHECK(hipModuleGetFunction(&function[i], module[i],
|
||||
"sampleModuleKernel"));
|
||||
HIP_CHECK(hipModuleGetFunction(&function[i], module[i], "sampleModuleKernel"));
|
||||
REQUIRE(function[i] != nullptr);
|
||||
}
|
||||
|
||||
@@ -543,8 +463,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
params[i].hStream = streamArr[i];
|
||||
}
|
||||
|
||||
HIP_CHECK(dyn_hipModuleLaunchCooperativeKernelMultiDevice_ptr(
|
||||
params.data(), deviceCount, 0));
|
||||
HIP_CHECK(dyn_hipModuleLaunchCooperativeKernelMultiDevice_ptr(params.data(), deviceCount, 0));
|
||||
|
||||
for (int i = 0; i < deviceCount; ++i) {
|
||||
HIP_CHECK(hipStreamSynchronize(params[i].hStream));
|
||||
@@ -558,10 +477,9 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
|
||||
// Validating hipLaunchCooperativeKernel API
|
||||
{
|
||||
HIP_CHECK(dyn_hipLaunchCooperativeKernel_ptr(
|
||||
reinterpret_cast<void *>(addOneKernel),
|
||||
dim3(1, 1, 1), dim3(1, 1, 1),
|
||||
kernel_parameter, 0, 0));
|
||||
HIP_CHECK(dyn_hipLaunchCooperativeKernel_ptr(reinterpret_cast<void*>(addOneKernel),
|
||||
dim3(1, 1, 1), dim3(1, 1, 1), kernel_parameter, 0,
|
||||
0));
|
||||
HIP_CHECK(hipMemcpy(hostArr, devArr, Nbytes, hipMemcpyDeviceToHost));
|
||||
REQUIRE(validateHostArray(hostArr, N, 13) == true);
|
||||
}
|
||||
@@ -571,7 +489,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
int deviceCount = 0;
|
||||
HIP_CHECK(hipGetDeviceCount(&deviceCount));
|
||||
|
||||
hipStream_t *streamArr = new hipStream_t[deviceCount];
|
||||
hipStream_t* streamArr = new hipStream_t[deviceCount];
|
||||
|
||||
for (int i = 0; i < deviceCount; ++i) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
@@ -581,7 +499,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
std::vector<hipLaunchParams> params(deviceCount);
|
||||
|
||||
for (int i = 0; i < deviceCount; ++i) {
|
||||
params[i].func = reinterpret_cast<void *>(simpleKernel);
|
||||
params[i].func = reinterpret_cast<void*>(simpleKernel);
|
||||
params[i].gridDim = {1, 1, 1};
|
||||
params[i].blockDim = {1, 1, 1};
|
||||
params[i].args = nullptr;
|
||||
@@ -589,8 +507,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
params[i].stream = streamArr[i];
|
||||
}
|
||||
|
||||
HIP_CHECK(dyn_hipLaunchCooperativeKernelMultiDevice_ptr(
|
||||
params.data(), deviceCount, 0));
|
||||
HIP_CHECK(dyn_hipLaunchCooperativeKernelMultiDevice_ptr(params.data(), deviceCount, 0));
|
||||
|
||||
for (int i = 0; i < deviceCount; ++i) {
|
||||
HIP_CHECK(hipStreamSynchronize(params[i].stream));
|
||||
@@ -606,7 +523,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
int deviceCount = 0;
|
||||
HIP_CHECK(hipGetDeviceCount(&deviceCount));
|
||||
|
||||
hipStream_t *streamArr = new hipStream_t[deviceCount];
|
||||
hipStream_t* streamArr = new hipStream_t[deviceCount];
|
||||
|
||||
for (int i = 0; i < deviceCount; ++i) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
@@ -616,7 +533,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
std::vector<hipLaunchParams> params(deviceCount);
|
||||
|
||||
for (int i = 0; i < deviceCount; ++i) {
|
||||
params[i].func = reinterpret_cast<void *>(simpleKernel);
|
||||
params[i].func = reinterpret_cast<void*>(simpleKernel);
|
||||
params[i].gridDim = {1, 1, 1};
|
||||
params[i].blockDim = {1, 1, 1};
|
||||
params[i].args = nullptr;
|
||||
@@ -624,8 +541,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisCooperativeKernels") {
|
||||
params[i].stream = streamArr[i];
|
||||
}
|
||||
|
||||
HIP_CHECK(dyn_hipExtLaunchMultiKernelMultiDevice_ptr(
|
||||
params.data(), deviceCount, 0));
|
||||
HIP_CHECK(dyn_hipExtLaunchMultiKernelMultiDevice_ptr(params.data(), deviceCount, 0));
|
||||
|
||||
for (int i = 0; i < deviceCount; ++i) {
|
||||
HIP_CHECK(hipStreamSynchronize(params[i].stream));
|
||||
@@ -658,8 +574,7 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
|
||||
void* hipModuleOccupancyMaxPotentialBlockSize_ptr = nullptr;
|
||||
void* hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr = nullptr;
|
||||
void* hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr = nullptr;
|
||||
void* hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr =
|
||||
nullptr;
|
||||
void* hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr = nullptr;
|
||||
void* hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr = nullptr;
|
||||
void* hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr = nullptr;
|
||||
void* hipOccupancyMaxPotentialBlockSize_ptr = nullptr;
|
||||
@@ -667,73 +582,61 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
|
||||
int currentHipVersion = 0;
|
||||
HIP_CHECK(hipRuntimeGetVersion(¤tHipVersion));
|
||||
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipModuleOccupancyMaxPotentialBlockSize",
|
||||
&hipModuleOccupancyMaxPotentialBlockSize_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipModuleOccupancyMaxPotentialBlockSizeWithFlags",
|
||||
&hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipModuleOccupancyMaxActiveBlocksPerMultiprocessor",
|
||||
&hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags",
|
||||
&hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipOccupancyMaxActiveBlocksPerMultiprocessor",
|
||||
&hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags",
|
||||
&hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress(
|
||||
"hipOccupancyMaxPotentialBlockSize",
|
||||
&hipOccupancyMaxPotentialBlockSize_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleOccupancyMaxPotentialBlockSize",
|
||||
&hipModuleOccupancyMaxPotentialBlockSize_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleOccupancyMaxPotentialBlockSizeWithFlags",
|
||||
&hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleOccupancyMaxActiveBlocksPerMultiprocessor",
|
||||
&hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags",
|
||||
&hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipOccupancyMaxActiveBlocksPerMultiprocessor",
|
||||
&hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr, currentHipVersion,
|
||||
0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags",
|
||||
&hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr,
|
||||
currentHipVersion, 0, nullptr));
|
||||
HIP_CHECK(hipGetProcAddress("hipOccupancyMaxPotentialBlockSize",
|
||||
&hipOccupancyMaxPotentialBlockSize_ptr, currentHipVersion, 0,
|
||||
nullptr));
|
||||
|
||||
hipError_t(*dyn_hipModuleOccupancyMaxPotentialBlockSize_ptr)(
|
||||
int *, int *, hipFunction_t, size_t, int) =
|
||||
reinterpret_cast<hipError_t (*)(int *, int *, hipFunction_t, size_t, int)>
|
||||
(hipModuleOccupancyMaxPotentialBlockSize_ptr);
|
||||
hipError_t (*dyn_hipModuleOccupancyMaxPotentialBlockSize_ptr)(int*, int*, hipFunction_t, size_t,
|
||||
int) =
|
||||
reinterpret_cast<hipError_t (*)(int*, int*, hipFunction_t, size_t, int)>(
|
||||
hipModuleOccupancyMaxPotentialBlockSize_ptr);
|
||||
|
||||
hipError_t(*dyn_hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr)(
|
||||
int *, int *, hipFunction_t, size_t, int, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(int *, int *, hipFunction_t,
|
||||
size_t, int, unsigned int)>
|
||||
(hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr);
|
||||
hipError_t (*dyn_hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr)(
|
||||
int*, int*, hipFunction_t, size_t, int, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(int*, int*, hipFunction_t, size_t, int, unsigned int)>(
|
||||
hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr);
|
||||
|
||||
hipError_t(*dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr)(
|
||||
int *, hipFunction_t, int, size_t) =
|
||||
reinterpret_cast<hipError_t (*)(int *, hipFunction_t, int, size_t)>
|
||||
(hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr);
|
||||
hipError_t (*dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr)(int*, hipFunction_t, int,
|
||||
size_t) =
|
||||
reinterpret_cast<hipError_t (*)(int*, hipFunction_t, int, size_t)>(
|
||||
hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr);
|
||||
|
||||
hipError_t(
|
||||
*dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr)(
|
||||
int *, hipFunction_t, int, size_t, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(int *, hipFunction_t, int,
|
||||
size_t, unsigned int)>
|
||||
(hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr);
|
||||
hipError_t (*dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr)(
|
||||
int*, hipFunction_t, int, size_t, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(int*, hipFunction_t, int, size_t, unsigned int)>(
|
||||
hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr);
|
||||
|
||||
hipError_t(*dyn_hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr)(
|
||||
int *, const void *, int, size_t) =
|
||||
reinterpret_cast<hipError_t (*)(int *, const void *, int, size_t)>
|
||||
(hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr);
|
||||
hipError_t (*dyn_hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr)(int*, const void*, int,
|
||||
size_t) =
|
||||
reinterpret_cast<hipError_t (*)(int*, const void*, int, size_t)>(
|
||||
hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr);
|
||||
|
||||
hipError_t(*dyn_hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr)(
|
||||
int *, const void *, int, size_t, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(int *, const void *,
|
||||
int, size_t, unsigned int)>
|
||||
(hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr);
|
||||
hipError_t (*dyn_hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr)(
|
||||
int*, const void*, int, size_t, unsigned int) =
|
||||
reinterpret_cast<hipError_t (*)(int*, const void*, int, size_t, unsigned int)>(
|
||||
hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr);
|
||||
|
||||
hipError_t(*dyn_hipOccupancyMaxPotentialBlockSize_ptr)(
|
||||
int *, int *, const void *, size_t, int) =
|
||||
reinterpret_cast<hipError_t (*)(int *, int *, const void *, size_t, int)>
|
||||
(hipOccupancyMaxPotentialBlockSize_ptr);
|
||||
hipError_t (*dyn_hipOccupancyMaxPotentialBlockSize_ptr)(int*, int*, const void*, size_t, int) =
|
||||
reinterpret_cast<hipError_t (*)(int*, int*, const void*, size_t, int)>(
|
||||
hipOccupancyMaxPotentialBlockSize_ptr);
|
||||
|
||||
hipModule_t module;
|
||||
HIP_CHECK(hipModuleLoad(&module, "addKernel.code"));
|
||||
@@ -747,10 +650,9 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
|
||||
|
||||
// Validating hipModuleOccupancyMaxPotentialBlockSize API
|
||||
{
|
||||
HIP_CHECK(hipModuleOccupancyMaxPotentialBlockSize(&gridSize, &blockSize,
|
||||
function, 0, 0));
|
||||
HIP_CHECK(hipModuleOccupancyMaxPotentialBlockSize(&gridSize, &blockSize, function, 0, 0));
|
||||
HIP_CHECK(dyn_hipModuleOccupancyMaxPotentialBlockSize_ptr(
|
||||
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr, function, 0, 0));
|
||||
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr, function, 0, 0));
|
||||
|
||||
REQUIRE(gridSizeWithFuncPtr == gridSize);
|
||||
REQUIRE(blockSizeWithFuncPtr == blockSize);
|
||||
@@ -758,12 +660,14 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
|
||||
|
||||
// Validating hipModuleOccupancyMaxPotentialBlockSizeWithFlags API
|
||||
{
|
||||
gridSize = 0; blockSize = 0;
|
||||
gridSizeWithFuncPtr = 0; blockSizeWithFuncPtr = 0;
|
||||
HIP_CHECK(hipModuleOccupancyMaxPotentialBlockSizeWithFlags(
|
||||
&gridSize, &blockSize, function, 0, 0, 0));
|
||||
gridSize = 0;
|
||||
blockSize = 0;
|
||||
gridSizeWithFuncPtr = 0;
|
||||
blockSizeWithFuncPtr = 0;
|
||||
HIP_CHECK(
|
||||
hipModuleOccupancyMaxPotentialBlockSizeWithFlags(&gridSize, &blockSize, function, 0, 0, 0));
|
||||
HIP_CHECK(dyn_hipModuleOccupancyMaxPotentialBlockSizeWithFlags_ptr(
|
||||
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr, function, 0, 0, 0));
|
||||
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr, function, 0, 0, 0));
|
||||
|
||||
REQUIRE(gridSizeWithFuncPtr == gridSize);
|
||||
REQUIRE(blockSizeWithFuncPtr == blockSize);
|
||||
@@ -772,63 +676,61 @@ TEST_CASE("Unit_hipGetProcAddress_ModuleApisOccupancy") {
|
||||
int numBlocks = 0, numBlocksWithFuncPtr = 0;
|
||||
// Validating hipModuleOccupancyMaxActiveBlocksPerMultiprocessor API
|
||||
{
|
||||
HIP_CHECK(hipModuleOccupancyMaxActiveBlocksPerMultiprocessor(
|
||||
&numBlocks, function, blockSize, 0));
|
||||
HIP_CHECK(dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr(
|
||||
&numBlocksWithFuncPtr, function, blockSize, 0));
|
||||
HIP_CHECK(
|
||||
hipModuleOccupancyMaxActiveBlocksPerMultiprocessor(&numBlocks, function, blockSize, 0));
|
||||
HIP_CHECK(dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessor_ptr(&numBlocksWithFuncPtr,
|
||||
function, blockSize, 0));
|
||||
|
||||
REQUIRE(numBlocksWithFuncPtr == numBlocks);
|
||||
}
|
||||
|
||||
// Validating hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags API
|
||||
{
|
||||
numBlocks = 0; numBlocksWithFuncPtr = 0;
|
||||
HIP_CHECK(hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(
|
||||
&numBlocks, function, blockSize, 0, 0));
|
||||
HIP_CHECK(
|
||||
dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr(
|
||||
&numBlocksWithFuncPtr, function, blockSize, 0, 0));
|
||||
numBlocks = 0;
|
||||
numBlocksWithFuncPtr = 0;
|
||||
HIP_CHECK(hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(&numBlocks, function,
|
||||
blockSize, 0, 0));
|
||||
HIP_CHECK(dyn_hipModuleOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr(
|
||||
&numBlocksWithFuncPtr, function, blockSize, 0, 0));
|
||||
|
||||
REQUIRE(numBlocksWithFuncPtr == numBlocks);
|
||||
}
|
||||
|
||||
// Validating hipOccupancyMaxActiveBlocksPerMultiprocessor API
|
||||
{
|
||||
numBlocks = 0; numBlocksWithFuncPtr = 0;
|
||||
numBlocks = 0;
|
||||
numBlocksWithFuncPtr = 0;
|
||||
HIP_CHECK(hipOccupancyMaxActiveBlocksPerMultiprocessor(
|
||||
&numBlocks, reinterpret_cast<const void *>(addOneKernel),
|
||||
blockSize, 0));
|
||||
&numBlocks, reinterpret_cast<const void*>(addOneKernel), blockSize, 0));
|
||||
HIP_CHECK(dyn_hipOccupancyMaxActiveBlocksPerMultiprocessor_ptr(
|
||||
&numBlocksWithFuncPtr,
|
||||
reinterpret_cast<const void *>(addOneKernel), blockSize, 0));
|
||||
&numBlocksWithFuncPtr, reinterpret_cast<const void*>(addOneKernel), blockSize, 0));
|
||||
|
||||
REQUIRE(numBlocksWithFuncPtr == numBlocks);
|
||||
}
|
||||
|
||||
// Validating hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags API
|
||||
{
|
||||
numBlocks = 0; numBlocksWithFuncPtr = 0;
|
||||
numBlocks = 0;
|
||||
numBlocksWithFuncPtr = 0;
|
||||
HIP_CHECK(hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(
|
||||
&numBlocks, reinterpret_cast<const void *>(addOneKernel),
|
||||
blockSize, 0, 0));
|
||||
&numBlocks, reinterpret_cast<const void*>(addOneKernel), blockSize, 0, 0));
|
||||
HIP_CHECK(dyn_hipOccupancyMaxActiveBlocksPerMultiprocessorWithFlags_ptr(
|
||||
&numBlocksWithFuncPtr,
|
||||
reinterpret_cast<const void *>(addOneKernel),
|
||||
blockSize, 0, 0));
|
||||
&numBlocksWithFuncPtr, reinterpret_cast<const void*>(addOneKernel), blockSize, 0, 0));
|
||||
|
||||
REQUIRE(numBlocksWithFuncPtr == numBlocks);
|
||||
}
|
||||
|
||||
// Validating hipOccupancyMaxPotentialBlockSize API
|
||||
{
|
||||
gridSize = 0; blockSize = 0;
|
||||
gridSizeWithFuncPtr = 0; blockSizeWithFuncPtr = 0;
|
||||
HIP_CHECK(hipOccupancyMaxPotentialBlockSize(
|
||||
&gridSize, &blockSize,
|
||||
reinterpret_cast<const void *>(addOneKernel), 0, 0));
|
||||
HIP_CHECK(dyn_hipOccupancyMaxPotentialBlockSize_ptr(
|
||||
&gridSizeWithFuncPtr, &blockSizeWithFuncPtr,
|
||||
reinterpret_cast<const void *>(addOneKernel), 0, 0));
|
||||
gridSize = 0;
|
||||
blockSize = 0;
|
||||
gridSizeWithFuncPtr = 0;
|
||||
blockSizeWithFuncPtr = 0;
|
||||
HIP_CHECK(hipOccupancyMaxPotentialBlockSize(&gridSize, &blockSize,
|
||||
reinterpret_cast<const void*>(addOneKernel), 0, 0));
|
||||
HIP_CHECK(dyn_hipOccupancyMaxPotentialBlockSize_ptr(&gridSizeWithFuncPtr, &blockSizeWithFuncPtr,
|
||||
reinterpret_cast<const void*>(addOneKernel),
|
||||
0, 0));
|
||||
|
||||
REQUIRE(gridSizeWithFuncPtr == gridSize);
|
||||
REQUIRE(blockSizeWithFuncPtr == blockSize);
|
||||
|
||||
@@ -61,22 +61,22 @@ TEST_CASE("Unit_hipHccModuleLaunchKernel_basic") {
|
||||
size_t width = GENERATE(3, 4, 100);
|
||||
size_t widthInBytes = width * sizeof(int);
|
||||
int *A_d, *B_d;
|
||||
int *A_h = reinterpret_cast<int*>(malloc(widthInBytes));
|
||||
int *B_h = reinterpret_cast<int*>(malloc(widthInBytes));
|
||||
int* A_h = reinterpret_cast<int*>(malloc(widthInBytes));
|
||||
int* B_h = reinterpret_cast<int*>(malloc(widthInBytes));
|
||||
for (int i = 0; i < width; i++) {
|
||||
A_h[i] = i;
|
||||
}
|
||||
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&A_d), widthInBytes));
|
||||
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&B_d), widthInBytes));
|
||||
void *kernelArgs[3] = {&A_d, &B_d, &widthInBytes};
|
||||
void* kernelArgs[3] = {&A_d, &B_d, &widthInBytes};
|
||||
HIP_CHECK(hipMemcpyHtoD((hipDeviceptr_t)A_d, A_h, widthInBytes));
|
||||
hipModule_t module;
|
||||
HIP_CHECK(hipModuleLoad(&module, fileName));
|
||||
hipFunction_t kernelFunc;
|
||||
HIP_CHECK(hipModuleGetFunction(&kernelFunc, module, kernel_name));
|
||||
|
||||
HIP_CHECK(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, 1, 0, 0,
|
||||
kernelArgs, nullptr, nullptr, nullptr));
|
||||
HIP_CHECK(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, 1, 0, 0, kernelArgs,
|
||||
nullptr, nullptr, nullptr));
|
||||
HIP_CHECK(hipMemcpyDtoH(B_h, (hipDeviceptr_t)B_d, widthInBytes));
|
||||
for (int i = 0; i < width; i++) {
|
||||
REQUIRE(A_h[i] == B_h[i]);
|
||||
@@ -105,43 +105,42 @@ TEST_CASE("Unit_hipHccModuleLaunchKernel_NegTst") {
|
||||
int *A_d, *B_d;
|
||||
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&A_d), widthInBytes));
|
||||
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&B_d), widthInBytes));
|
||||
void *kernelArgs[3] = {&A_d, &B_d, &widthInBytes};
|
||||
void* kernelArgs[3] = {&A_d, &B_d, &widthInBytes};
|
||||
hipModule_t module;
|
||||
HIP_CHECK(hipModuleLoad(&module, fileName));
|
||||
hipFunction_t kernelFunc;
|
||||
HIP_CHECK(hipModuleGetFunction(&kernelFunc, module, kernel_name));
|
||||
SECTION("nullptr to f(first argument)") {
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(nullptr, width, 1, 1, width, 1, 1,
|
||||
0, 0, kernelArgs, nullptr, nullptr, nullptr),
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(nullptr, width, 1, 1, width, 1, 1, 0, 0, kernelArgs,
|
||||
nullptr, nullptr, nullptr),
|
||||
hipErrorInvalidHandle);
|
||||
}
|
||||
SECTION("-1 to localWorkSizeX(fifth argument)") {
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, -1, 1, 1,
|
||||
0, 0, kernelArgs, nullptr, nullptr, nullptr),
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, -1, 1, 1, 0, 0, kernelArgs,
|
||||
nullptr, nullptr, nullptr),
|
||||
hipErrorInvalidConfiguration);
|
||||
}
|
||||
SECTION("-1 to localWorkSizeY(sixth argument)") {
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width,
|
||||
-1, 1, 0, 0, kernelArgs, nullptr, nullptr, nullptr),
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, -1, 1, 0, 0,
|
||||
kernelArgs, nullptr, nullptr, nullptr),
|
||||
hipErrorInvalidConfiguration);
|
||||
}
|
||||
SECTION("-1 to localWorkSizeZ(seventh argument)") {
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1,
|
||||
-1, 0, 0, kernelArgs, nullptr, nullptr, nullptr),
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, -1, 0, 0,
|
||||
kernelArgs, nullptr, nullptr, nullptr),
|
||||
hipErrorInvalidConfiguration);
|
||||
}
|
||||
SECTION("-1 to sharedMemBytes(eighth argument)") {
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1,
|
||||
1, -1, 0, kernelArgs, nullptr, nullptr, nullptr),
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, 1, -1, 0,
|
||||
kernelArgs, nullptr, nullptr, nullptr),
|
||||
hipErrorInvalidValue);
|
||||
}
|
||||
SECTION("nullptr to kernelParams(10th argument)") {
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1,
|
||||
1, 0, 0, nullptr, nullptr, nullptr, nullptr),
|
||||
HIP_CHECK_ERROR(hipHccModuleLaunchKernel(kernelFunc, width, 1, 1, width, 1, 1, 0, 0, nullptr,
|
||||
nullptr, nullptr, nullptr),
|
||||
hipErrorInvalidValue);
|
||||
}
|
||||
HIP_CHECK(hipModuleUnload(module));
|
||||
HIP_CHECK(hipFree(A_d));
|
||||
HIP_CHECK(hipFree(B_d));
|
||||
}
|
||||
|
||||
|
||||
@@ -29,25 +29,21 @@ static constexpr auto fileName3 = "copiousArgKernel3.code";
|
||||
static constexpr auto fileName16 = "copiousArgKernel16.code";
|
||||
static constexpr auto fileName17 = "copiousArgKernel17.code";
|
||||
|
||||
static constexpr int coeff[12] =
|
||||
{2, 3, 5, 7, 11, 13, 17, 19, 23, 29, 31, 37};
|
||||
static constexpr int coeff[12] = {2, 3, 5, 7, 11, 13, 17, 19, 23, 29, 31, 37};
|
||||
|
||||
static void fillDataTransfer2Dev(int *hostBuf, int *devBuf, size_t len) {
|
||||
static void fillDataTransfer2Dev(int* hostBuf, int* devBuf, size_t len) {
|
||||
unsigned int seed = time(nullptr);
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
hostBuf[i] = (HipTest::RAND_R(&seed) & 0xFF);
|
||||
}
|
||||
HIP_CHECK(hipMemcpy(devBuf, hostBuf, len*sizeof(int),
|
||||
hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(devBuf, hostBuf, len * sizeof(int), hipMemcpyHostToDevice));
|
||||
}
|
||||
|
||||
static void verifyDevResult(int *hostBuf, int *devBuf, int coef1, int coef2,
|
||||
size_t len) {
|
||||
int *buf = new int[len];
|
||||
HIP_CHECK(hipMemcpy(buf, devBuf, len*sizeof(int),
|
||||
hipMemcpyDeviceToHost));
|
||||
static void verifyDevResult(int* hostBuf, int* devBuf, int coef1, int coef2, size_t len) {
|
||||
int* buf = new int[len];
|
||||
HIP_CHECK(hipMemcpy(buf, devBuf, len * sizeof(int), hipMemcpyDeviceToHost));
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
REQUIRE(buf[i] == (coef1*hostBuf[i] + coef2));
|
||||
REQUIRE(buf[i] == (coef1 * hostBuf[i] + coef2));
|
||||
}
|
||||
delete[] buf;
|
||||
}
|
||||
@@ -57,17 +53,17 @@ TEST_CASE("Unit_KerArgOptimization_Saxpy") {
|
||||
constexpr size_t arraylenBytes = arraylen * sizeof(int);
|
||||
constexpr auto blocksize = 256;
|
||||
// Allocate host resources
|
||||
int *x1_h = new int[arraylen];
|
||||
int* x1_h = new int[arraylen];
|
||||
REQUIRE(x1_h != nullptr);
|
||||
int *x2_h = new int[arraylen];
|
||||
int* x2_h = new int[arraylen];
|
||||
REQUIRE(x2_h != nullptr);
|
||||
int *x3_h = new int[arraylen];
|
||||
int* x3_h = new int[arraylen];
|
||||
REQUIRE(x3_h != nullptr);
|
||||
int *x4_h = new int[arraylen];
|
||||
int* x4_h = new int[arraylen];
|
||||
REQUIRE(x4_h != nullptr);
|
||||
int *x5_h = new int[arraylen];
|
||||
int* x5_h = new int[arraylen];
|
||||
REQUIRE(x5_h != nullptr);
|
||||
int *x6_h = new int[arraylen];
|
||||
int* x6_h = new int[arraylen];
|
||||
REQUIRE(x6_h != nullptr);
|
||||
// Allocate device resources
|
||||
int *x1_d, *x2_d, *x3_d, *x4_d, *x5_d, *x6_d;
|
||||
@@ -89,22 +85,22 @@ TEST_CASE("Unit_KerArgOptimization_Saxpy") {
|
||||
struct {
|
||||
int a1;
|
||||
int a2;
|
||||
void *x1;
|
||||
void* x1;
|
||||
int b1;
|
||||
int b2;
|
||||
void *x2;
|
||||
void* x2;
|
||||
int c1;
|
||||
int c2;
|
||||
void *x3;
|
||||
void* x3;
|
||||
int d1;
|
||||
int d2;
|
||||
void *x4;
|
||||
void* x4;
|
||||
int e1;
|
||||
int e2;
|
||||
void *x5;
|
||||
void* x5;
|
||||
int f1;
|
||||
int f2;
|
||||
void *x6;
|
||||
void* x6;
|
||||
} args;
|
||||
args.a1 = coeff[0];
|
||||
args.a2 = coeff[1];
|
||||
@@ -125,8 +121,7 @@ TEST_CASE("Unit_KerArgOptimization_Saxpy") {
|
||||
args.f2 = coeff[11];
|
||||
args.x6 = x6_d;
|
||||
size_t size = sizeof(args);
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
|
||||
// Get module and function from module
|
||||
@@ -160,9 +155,8 @@ TEST_CASE("Unit_KerArgOptimization_Saxpy") {
|
||||
HIP_CHECK(hipModuleLoad(&Module, fileName17));
|
||||
HIP_CHECK(hipModuleGetFunction(&Function, Module, kernel_name));
|
||||
}
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(Function, arraylen, 1, 1,
|
||||
blocksize, 1, 1, 0, 0, NULL,
|
||||
reinterpret_cast<void**>(&config), 0));
|
||||
HIP_CHECK(hipExtModuleLaunchKernel(Function, arraylen, 1, 1, blocksize, 1, 1, 0, 0, NULL,
|
||||
reinterpret_cast<void**>(&config), 0));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
// Verify results
|
||||
verifyDevResult(x1_h, x1_d, coeff[0], coeff[1], arraylen);
|
||||
|
||||
@@ -20,15 +20,15 @@ THE SOFTWARE.
|
||||
#include <hip_test_defgroups.hh>
|
||||
|
||||
constexpr int MANAGED_VAR_INIT_VALUE = 10;
|
||||
constexpr auto fileName = "managed_kernel.code";
|
||||
constexpr auto fileName = "managed_kernel.code";
|
||||
|
||||
/**
|
||||
* @addtogroup hipModuleGetGlobal
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleGetGlobal(hipDeviceptr_t* dptr, size_t* bytes, hipModule_t hmod, const char* name)` -
|
||||
* Returns a global pointer from a module
|
||||
*/
|
||||
* @addtogroup hipModuleGetGlobal
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleGetGlobal(hipDeviceptr_t* dptr, size_t* bytes, hipModule_t hmod, const char*
|
||||
* name)` - Returns a global pointer from a module
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
@@ -52,9 +52,7 @@ TEST_CASE("Unit_hipModuleGetGlobal_Functional") {
|
||||
HIP_CHECK(hipGetDeviceCount(&numDevices));
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
int managed_memory = 0;
|
||||
HIPCHECK(hipDeviceGetAttribute(&managed_memory,
|
||||
hipDeviceAttributeManagedMemory,
|
||||
i));
|
||||
HIPCHECK(hipDeviceGetAttribute(&managed_memory, hipDeviceAttributeManagedMemory, i));
|
||||
if (!managed_memory) {
|
||||
HipTest::HIP_SKIP_TEST("managed memory access not supported on device");
|
||||
return;
|
||||
@@ -70,11 +68,9 @@ TEST_CASE("Unit_hipModuleGetGlobal_Functional") {
|
||||
HIP_CHECK(hipModuleLoad(&Module, fileName));
|
||||
hipFunction_t Function;
|
||||
HIP_CHECK(hipModuleGetFunction(&Function, Module, "GPU_func"));
|
||||
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, 1, 1, 1, 0, 0,
|
||||
NULL, NULL));
|
||||
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, 1, 1, 1, 0, 0, NULL, NULL));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
HIP_CHECK(hipModuleGetGlobal(reinterpret_cast<hipDeviceptr_t*>(&x),
|
||||
&xSize, Module, "x"));
|
||||
HIP_CHECK(hipModuleGetGlobal(reinterpret_cast<hipDeviceptr_t*>(&x), &xSize, Module, "x"));
|
||||
HIP_CHECK(hipMemcpyDtoH(&data, hipDeviceptr_t(x), xSize));
|
||||
if (data != (1 + MANAGED_VAR_INIT_VALUE)) {
|
||||
HIP_CHECK(hipModuleUnload(Module));
|
||||
|
||||
@@ -33,18 +33,19 @@ constexpr auto CODE_OBJ_MULTIARCH = "vcpy_kernel_multarch.code";
|
||||
#endif
|
||||
|
||||
/**
|
||||
* @addtogroup hipModuleLoad
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
|
||||
* Loads code object from file into a module
|
||||
*/
|
||||
* @addtogroup hipModuleLoad
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
|
||||
* Loads code object from file into a module
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Test case to load and execute a code object file for the current GPU architecture.
|
||||
* - Test case to load and execute a code object file for the multiple GPU architectures including the current
|
||||
* - Test case to load and execute a code object file for the multiple GPU architectures including
|
||||
the current
|
||||
|
||||
* Test source
|
||||
* ------------------------
|
||||
@@ -54,7 +55,7 @@ constexpr auto CODE_OBJ_MULTIARCH = "vcpy_kernel_multarch.code";
|
||||
* - HIP_VERSION >= 5.6
|
||||
*/
|
||||
|
||||
bool testCodeObjFile(const char *codeObjFile) {
|
||||
bool testCodeObjFile(const char* codeObjFile) {
|
||||
float *A, *B, *Ad, *Bd;
|
||||
A = new float[LEN];
|
||||
B = new float[LEN];
|
||||
@@ -85,12 +86,10 @@ bool testCodeObjFile(const char *codeObjFile) {
|
||||
args._Bd = reinterpret_cast<void*>(Bd);
|
||||
size_t size = sizeof(args);
|
||||
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0,
|
||||
stream, NULL,
|
||||
reinterpret_cast<void**>(&config)));
|
||||
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0, stream, NULL,
|
||||
reinterpret_cast<void**>(&config)));
|
||||
|
||||
HIP_CHECK(hipStreamDestroy(stream));
|
||||
|
||||
@@ -114,8 +113,8 @@ bool testCodeObjFile(const char *codeObjFile) {
|
||||
#ifdef __linux__
|
||||
// Check if environment variable $ROCM_PATH is defined
|
||||
bool isRocmPathSet() {
|
||||
FILE *fpipe;
|
||||
char const *command = "echo $ROCM_PATH";
|
||||
FILE* fpipe;
|
||||
char const* command = "echo $ROCM_PATH";
|
||||
fpipe = popen(command, "r");
|
||||
|
||||
if (fpipe == nullptr) {
|
||||
@@ -143,14 +142,12 @@ bool testMultiTargArchCodeObj() {
|
||||
HIP_CHECK(hipGetDeviceProperties(&props, 0));
|
||||
// Hardcoding the codeobject lines in multiple string to avoid cpplint warning
|
||||
std::string CodeObjL1 = "#include \"hip/hip_runtime.h\"\n";
|
||||
std::string CodeObjL2 =
|
||||
"extern \"C\" __global__ void hello_world(float* a, float* b) {\n";
|
||||
std::string CodeObjL2 = "extern \"C\" __global__ void hello_world(float* a, float* b) {\n";
|
||||
std::string CodeObjL3 = " int tx = threadIdx.x;\n";
|
||||
std::string CodeObjL4 = " b[tx] = a[tx];\n";
|
||||
std::string CodeObjL5 = "}";
|
||||
// Creating the full code object string
|
||||
static std::string CodeObj = CodeObjL1 + CodeObjL2 + CodeObjL3 +
|
||||
CodeObjL4 + CodeObjL5;
|
||||
static std::string CodeObj = CodeObjL1 + CodeObjL2 + CodeObjL3 + CodeObjL4 + CodeObjL5;
|
||||
std::ofstream ofs("/tmp/vcpy_kernel.cpp", std::ofstream::out);
|
||||
ofs << CodeObj;
|
||||
ofs.close();
|
||||
@@ -172,15 +169,12 @@ bool testMultiTargArchCodeObj() {
|
||||
const char* genco_option = "--offload-arch";
|
||||
const char* input_codeobj = "/tmp/vcpy_kernel.cpp";
|
||||
const char* rocm_enumerator = "${ROCM_PATH}/bin/rocm_agent_enumerator";
|
||||
snprintf(command, COMMAND_LEN,
|
||||
rocm_enumerator,
|
||||
hipcc_path, genco_option, props.gcnArchName, input_codeobj,
|
||||
CODE_OBJ_MULTIARCH);
|
||||
snprintf(command, COMMAND_LEN, rocm_enumerator, hipcc_path, genco_option, props.gcnArchName,
|
||||
input_codeobj, CODE_OBJ_MULTIARCH);
|
||||
|
||||
system((const char*)command);
|
||||
// Check if the code object file is created
|
||||
snprintf(command, COMMAND_LEN, "./%s",
|
||||
CODE_OBJ_MULTIARCH);
|
||||
snprintf(command, COMMAND_LEN, "./%s", CODE_OBJ_MULTIARCH);
|
||||
|
||||
if (access(command, F_OK) == -1) {
|
||||
INFO("Code Object File not found \n");
|
||||
|
||||
@@ -211,6 +211,6 @@ TEST_CASE("Unit_hipModuleLaunchCooperativeKernel_Negative_Parameters") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group ModuleTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group ModuleTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -101,13 +101,13 @@ bool Module_Negative_tests() {
|
||||
args1._Ad = nullptr;
|
||||
args1._Bd = nullptr;
|
||||
args1._Cd = nullptr;
|
||||
args1._n = 0;
|
||||
args1._n = 0;
|
||||
hipFunction_t MultKernel, KernelandExtraParamKernel;
|
||||
size_t size1;
|
||||
size1 = sizeof(args1);
|
||||
hipModule_t Module;
|
||||
hipStream_t stream1;
|
||||
hipDeviceptr_t *Ad = nullptr;
|
||||
hipDeviceptr_t* Ad = nullptr;
|
||||
#ifdef HT_NVIDIA
|
||||
HIP_CHECK(hipInit(0));
|
||||
hipCtx_t context;
|
||||
@@ -116,122 +116,88 @@ bool Module_Negative_tests() {
|
||||
|
||||
HIP_CHECK(hipModuleLoad(&Module, fileName));
|
||||
HIP_CHECK(hipModuleGetFunction(&MultKernel, Module, matmulK));
|
||||
HIP_CHECK(hipModuleGetFunction(&KernelandExtraParamKernel,
|
||||
Module, KernelandExtra));
|
||||
void *config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
void *params[] = {Ad};
|
||||
HIP_CHECK(hipModuleGetFunction(&KernelandExtraParamKernel, Module, KernelandExtra));
|
||||
void* config1[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args1, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
void* params[] = {Ad};
|
||||
HIP_CHECK(hipStreamCreate(&stream1));
|
||||
// Passing nullptr to kernel function
|
||||
err = hipModuleLaunchKernel(nullptr, 1, 1, 1, 1, 1, 1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(nullptr, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing Max int value to block dimensions
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, std::numeric_limits<uint32_t>::max(),
|
||||
std::numeric_limits<uint32_t>::max(),
|
||||
std::numeric_limits<uint32_t>::max(),
|
||||
std::numeric_limits<uint32_t>::max(),
|
||||
0, stream1, NULL,
|
||||
std::numeric_limits<uint32_t>::max(), 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing 0 as value for all dimensions
|
||||
err = hipModuleLaunchKernel(MultKernel, 0, 0, 0,
|
||||
0,
|
||||
0,
|
||||
0, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(MultKernel, 0, 0, 0, 0, 0, 0, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing 0 as value for x dimension
|
||||
err = hipModuleLaunchKernel(MultKernel, 0, 1, 1,
|
||||
0,
|
||||
1,
|
||||
1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(MultKernel, 0, 1, 1, 0, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing 0 as value for y dimension
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 0, 1,
|
||||
1,
|
||||
0,
|
||||
1, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 0, 1, 1, 0, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing 0 as value for z dimension
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 0,
|
||||
1,
|
||||
1,
|
||||
0, 0,
|
||||
stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 0, 1, 1, 0, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing both kernel and extra params
|
||||
err = hipModuleLaunchKernel(KernelandExtraParamKernel, 1, 1, 1, 1,
|
||||
1, 1, 0, stream1,
|
||||
reinterpret_cast<void**>(¶ms),
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err =
|
||||
hipModuleLaunchKernel(KernelandExtraParamKernel, 1, 1, 1, 1, 1, 1, 0, stream1,
|
||||
reinterpret_cast<void**>(¶ms), reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing more than maxthreadsperblock to block dimensions
|
||||
hipDeviceProp_t deviceProp;
|
||||
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
deviceProp.maxThreadsPerBlock+1,
|
||||
deviceProp.maxThreadsPerBlock+1,
|
||||
deviceProp.maxThreadsPerBlock+1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, deviceProp.maxThreadsPerBlock + 1,
|
||||
deviceProp.maxThreadsPerBlock + 1, deviceProp.maxThreadsPerBlock + 1,
|
||||
0, stream1, NULL, reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Block dimension X = Max Allowed + 1
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
deviceProp.maxThreadsDim[0]+1,
|
||||
1,
|
||||
1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, deviceProp.maxThreadsDim[0] + 1, 1, 1, 0,
|
||||
stream1, NULL, reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Block dimension Y = Max Allowed + 1
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
1,
|
||||
deviceProp.maxThreadsDim[1]+1,
|
||||
1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, 1, deviceProp.maxThreadsDim[1] + 1, 1, 0,
|
||||
stream1, NULL, reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Block dimension Z = Max Allowed + 1
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1,
|
||||
1,
|
||||
1,
|
||||
deviceProp.maxThreadsDim[2]+1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config1));
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, 1, 1, deviceProp.maxThreadsDim[2] + 1, 0,
|
||||
stream1, NULL, reinterpret_cast<void**>(&config1));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Passing invalid config data to extra params
|
||||
void *config3[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
void* config3[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size1,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
err = hipModuleLaunchKernel(MultKernel, 1, 1, 1, 1, 1, 1, 0, stream1, NULL,
|
||||
reinterpret_cast<void**>(&config3));
|
||||
reinterpret_cast<void**>(&config3));
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
@@ -275,22 +241,13 @@ bool Module_GridBlock_Corner_Tests() {
|
||||
unsigned int maxgridY = deviceProp.maxGridSize[1];
|
||||
unsigned int maxgridZ = deviceProp.maxGridSize[2];
|
||||
#endif
|
||||
struct gridblockDim test[6] = {{1, 1, 1, maxblockX, 1, 1},
|
||||
{1, 1, 1, 1, maxblockY, 1},
|
||||
{1, 1, 1, 1, 1, maxblockZ},
|
||||
{maxgridX, 1, 1, 1, 1, 1},
|
||||
{1, maxgridY, 1, 1, 1, 1},
|
||||
{1, 1, maxgridZ, 1, 1, 1}};
|
||||
struct gridblockDim test[6] = {{1, 1, 1, maxblockX, 1, 1}, {1, 1, 1, 1, maxblockY, 1},
|
||||
{1, 1, 1, 1, 1, maxblockZ}, {maxgridX, 1, 1, 1, 1, 1},
|
||||
{1, maxgridY, 1, 1, 1, 1}, {1, 1, maxgridZ, 1, 1, 1}};
|
||||
for (int i = 0; i < 6; i++) {
|
||||
err = hipModuleLaunchKernel(DummyKernel,
|
||||
test[i].gridX,
|
||||
test[i].gridY,
|
||||
test[i].gridZ,
|
||||
test[i].blockX,
|
||||
test[i].blockY,
|
||||
test[i].blockZ,
|
||||
0,
|
||||
stream1, NULL, NULL);
|
||||
err = hipModuleLaunchKernel(DummyKernel, test[i].gridX, test[i].gridY, test[i].gridZ,
|
||||
test[i].blockX, test[i].blockY, test[i].blockZ, 0, stream1, NULL,
|
||||
NULL);
|
||||
if (err != hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
@@ -321,25 +278,20 @@ bool Module_WorkGroup_Test() {
|
||||
// Passing Max int value to block dimensions
|
||||
hipDeviceProp_t deviceProp;
|
||||
HIP_CHECK(hipGetDeviceProperties(&deviceProp, 0));
|
||||
double cuberootVal =
|
||||
cbrt(static_cast<double>(deviceProp.maxThreadsPerBlock));
|
||||
double cuberootVal = cbrt(static_cast<double>(deviceProp.maxThreadsPerBlock));
|
||||
uint32_t cuberoot_floor = floor(cuberootVal);
|
||||
uint32_t cuberoot_ceil = ceil(cuberootVal);
|
||||
// Scenario: (block.x * block.y * block.z) <= Work Group Size where
|
||||
// block.x < MaxBlockDimX , block.y < MaxBlockDimY and block.z < MaxBlockDimZ
|
||||
err = hipModuleLaunchKernel(DummyKernel,
|
||||
1, 1, 1,
|
||||
cuberoot_floor, cuberoot_floor, cuberoot_floor,
|
||||
0, stream1, NULL, NULL);
|
||||
err = hipModuleLaunchKernel(DummyKernel, 1, 1, 1, cuberoot_floor, cuberoot_floor, cuberoot_floor,
|
||||
0, stream1, NULL, NULL);
|
||||
if (err != hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
// Scenario: (block.x * block.y * block.z) > Work Group Size where
|
||||
// block.x < MaxBlockDimX , block.y < MaxBlockDimY and block.z < MaxBlockDimZ
|
||||
err = hipModuleLaunchKernel(DummyKernel,
|
||||
1, 1, 1,
|
||||
cuberoot_ceil, cuberoot_ceil, cuberoot_ceil + 1,
|
||||
0, stream1, NULL, NULL);
|
||||
err = hipModuleLaunchKernel(DummyKernel, 1, 1, 1, cuberoot_ceil, cuberoot_ceil, cuberoot_ceil + 1,
|
||||
0, stream1, NULL, NULL);
|
||||
if (err == hipSuccess) {
|
||||
testStatus = false;
|
||||
}
|
||||
|
||||
@@ -107,14 +107,14 @@ TEST_CASE("Unit_hipModuleLoadData_Negative_Image_Is_An_Empty_String") {
|
||||
}
|
||||
|
||||
/**
|
||||
* @addtogroup hipModuleLoad hipModuleGetFunction
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
|
||||
* Loads code object from file into a module
|
||||
* `hipError_t hipModuleGetFunction(hipFunction_t* function, hipModule_t module, const char* kname)` -
|
||||
* Function with kname will be extracted if present in module
|
||||
*/
|
||||
* @addtogroup hipModuleLoad hipModuleGetFunction
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
|
||||
* Loads code object from file into a module
|
||||
* `hipError_t hipModuleGetFunction(hipFunction_t* function, hipModule_t module, const char* kname)`
|
||||
* - Function with kname will be extracted if present in module
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
@@ -172,11 +172,10 @@ TEST_CASE("Unit_hipModuleLoadData_Functional") {
|
||||
args._Bd = reinterpret_cast<void*>(Bd);
|
||||
size_t size = sizeof(args);
|
||||
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0,
|
||||
stream, NULL, reinterpret_cast<void**>(&config)));
|
||||
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, LEN, 1, 1, 0, stream, NULL,
|
||||
reinterpret_cast<void**>(&config)));
|
||||
|
||||
HIP_CHECK(hipStreamDestroy(stream));
|
||||
|
||||
@@ -185,8 +184,8 @@ TEST_CASE("Unit_hipModuleLoadData_Functional") {
|
||||
for (uint32_t i = 0; i < LEN; i++) {
|
||||
REQUIRE(A[i] == B[i]);
|
||||
}
|
||||
delete [] A;
|
||||
delete [] B;
|
||||
delete[] A;
|
||||
delete[] B;
|
||||
HIP_CHECK(hipModuleUnload(Module));
|
||||
HIP_CHECK(hipFree(Ad));
|
||||
HIP_CHECK(hipFree(Bd));
|
||||
|
||||
@@ -20,18 +20,18 @@ THE SOFTWARE.
|
||||
#include <hip_test_defgroups.hh>
|
||||
#include <hip_test_process.hh>
|
||||
/**
|
||||
* @addtogroup hipModuleLoad hipModuleLoadData hipModuleLoadDataEx
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
|
||||
* Loads code object from file into a module
|
||||
* `hipError_t hipModuleLoadData (hipModule_t *module, const void *image)` -
|
||||
* Builds module from code object which resides in host memory. Image is pointer to that location.
|
||||
* `hipError_t hipModuleLoadDataEx (hipModule_t *module, const void *image,
|
||||
* unsigned int numOptions, hipJitOption *options, void **optionValues)` -
|
||||
* Builds module from code object which resides in host memory. Image is pointer to that
|
||||
* location. Options are not used.
|
||||
*/
|
||||
* @addtogroup hipModuleLoad hipModuleLoadData hipModuleLoadDataEx
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleLoad(hipModule_t* module, const char* fname)` -
|
||||
* Loads code object from file into a module
|
||||
* `hipError_t hipModuleLoadData (hipModule_t *module, const void *image)` -
|
||||
* Builds module from code object which resides in host memory. Image is pointer to that location.
|
||||
* `hipError_t hipModuleLoadDataEx (hipModule_t *module, const void *image,
|
||||
* unsigned int numOptions, hipJitOption *options, void **optionValues)` -
|
||||
* Builds module from code object which resides in host memory. Image is pointer to that
|
||||
* location. Options are not used.
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
|
||||
@@ -35,12 +35,12 @@ TEST_CASE("Unit_hipModuleUnload_Negative_Double_Unload") {
|
||||
HIP_CHECK_ERROR(hipModuleUnload(module), hipErrorNotFound);
|
||||
}
|
||||
/**
|
||||
* @addtogroup hipModuleUnload
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleUnload(hipModule_t module)` -
|
||||
* Frees the module
|
||||
*/
|
||||
* @addtogroup hipModuleUnload
|
||||
* @{
|
||||
* @ingroup ModuleTest
|
||||
* `hipError_t hipModuleUnload(hipModule_t module)` -
|
||||
* Frees the module
|
||||
*/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
@@ -52,11 +52,11 @@ TEST_CASE("Unit_hipModuleUnload_Negative_Double_Unload") {
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.6
|
||||
*/
|
||||
*/
|
||||
TEST_CASE("Unit_hipModuleLoad_basic") {
|
||||
constexpr auto fileName = "vcpy_kernel.code";
|
||||
hipModule_t module;
|
||||
HIP_CHECK(hipModuleLoad(&module, fileName));
|
||||
REQUIRE(module != nullptr);
|
||||
HIP_CHECK(hipModuleUnload(module));
|
||||
constexpr auto fileName = "vcpy_kernel.code";
|
||||
hipModule_t module;
|
||||
HIP_CHECK(hipModuleLoad(&module, fileName));
|
||||
REQUIRE(module != nullptr);
|
||||
HIP_CHECK(hipModuleUnload(module));
|
||||
}
|
||||
|
||||
@@ -20,18 +20,17 @@ THE SOFTWARE.
|
||||
constexpr int GLOBAL_BUF_SIZE = 2048;
|
||||
|
||||
__device__ float deviceGlobalFloat;
|
||||
__device__ int deviceGlobalInt1;
|
||||
__device__ int deviceGlobalInt2;
|
||||
__device__ short deviceGlobalShort; //NOLINT
|
||||
__device__ char deviceGlobalChar;
|
||||
__device__ int deviceGlobalInt1;
|
||||
__device__ int deviceGlobalInt2;
|
||||
__device__ short deviceGlobalShort; // NOLINT
|
||||
__device__ char deviceGlobalChar;
|
||||
|
||||
__device__ int getSquareOfGlobalFloat() {
|
||||
return static_cast<int>(deviceGlobalFloat*deviceGlobalFloat);
|
||||
return static_cast<int>(deviceGlobalFloat * deviceGlobalFloat);
|
||||
}
|
||||
|
||||
extern "C" __global__ void testWeightedCopy(int* a, int* b) {
|
||||
int tx = threadIdx.x;
|
||||
b[tx] = deviceGlobalInt1 * a[tx] + deviceGlobalInt2 +
|
||||
static_cast<int>(deviceGlobalShort) + static_cast<int>(deviceGlobalChar)
|
||||
+ getSquareOfGlobalFloat();
|
||||
b[tx] = deviceGlobalInt1 * a[tx] + deviceGlobalInt2 + static_cast<int>(deviceGlobalShort) +
|
||||
static_cast<int>(deviceGlobalChar) + getSquareOfGlobalFloat();
|
||||
}
|
||||
|
||||
@@ -19,6 +19,4 @@ THE SOFTWARE.
|
||||
#include "hip/hip_runtime.h"
|
||||
__managed__ int x = 10;
|
||||
|
||||
extern "C" __global__ void GPU_func() {
|
||||
x++;
|
||||
}
|
||||
extern "C" __global__ void GPU_func() { x++; }
|
||||
|
||||
@@ -16,11 +16,10 @@ LIABILITY, WHETHER INN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
#include"hip/hip_runtime.h"
|
||||
#include "hip/hip_runtime.h"
|
||||
__device__ int deviceGlobal = 1;
|
||||
|
||||
extern "C" __global__ void matmulK(int clockrate, int* A, int* B, int* C,
|
||||
int N) {
|
||||
extern "C" __global__ void matmulK(int clockrate, int* A, int* B, int* C, int N) {
|
||||
int ROW = blockIdx.y * blockDim.y + threadIdx.y;
|
||||
int COL = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int tmpSum = 0;
|
||||
@@ -33,8 +32,7 @@ extern "C" __global__ void matmulK(int clockrate, int* A, int* B, int* C,
|
||||
}
|
||||
}
|
||||
|
||||
extern "C" __global__ void KernelandExtraParams(int* A, int* B, int* C,
|
||||
int *D, int N) {
|
||||
extern "C" __global__ void KernelandExtraParams(int* A, int* B, int* C, int* D, int N) {
|
||||
int ROW = blockIdx.y * blockDim.y + threadIdx.y;
|
||||
int COL = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int tmpSum = 0;
|
||||
@@ -50,31 +48,24 @@ extern "C" __global__ void KernelandExtraParams(int* A, int* B, int* C,
|
||||
|
||||
__device__ void Delay(uint32_t interval, const uint32_t ticks_per_ms) {
|
||||
while (interval--) {
|
||||
#if HT_AMD
|
||||
#if HT_AMD
|
||||
uint64_t start = wall_clock64();
|
||||
while (wall_clock64() - start < ticks_per_ms) {
|
||||
__builtin_amdgcn_s_sleep(10);
|
||||
}
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
#endif
|
||||
#if HT_NVIDIA
|
||||
uint64_t start = clock64();
|
||||
while (clock64() - start < ticks_per_ms) {
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
extern "C" __global__ void SixteenSecKernel(int clockrate) {
|
||||
Delay(16000, clockrate);
|
||||
}
|
||||
extern "C" __global__ void SixteenSecKernel(int clockrate) { Delay(16000, clockrate); }
|
||||
|
||||
extern "C" __global__ void TwoSecKernel(int clockrate) {
|
||||
Delay(2000, clockrate);
|
||||
}
|
||||
extern "C" __global__ void TwoSecKernel(int clockrate) { Delay(2000, clockrate); }
|
||||
|
||||
extern "C" __global__ void FourSecKernel(int clockrate) {
|
||||
Delay(4000, clockrate);
|
||||
}
|
||||
extern "C" __global__ void FourSecKernel(int clockrate) { Delay(4000, clockrate); }
|
||||
|
||||
extern "C" __global__ void dummyKernel() {
|
||||
}
|
||||
extern "C" __global__ void dummyKernel() {}
|
||||
|
||||
@@ -20,24 +20,21 @@ THE SOFTWARE.
|
||||
#include <fstream>
|
||||
#include <cstddef>
|
||||
#include <vector>
|
||||
#include<iostream>
|
||||
#define HIP_CHECK(error)\
|
||||
{\
|
||||
hipError_t localError = error;\
|
||||
if ((localError != hipSuccess) && \
|
||||
(localError != hipErrorPeerAccessAlreadyEnabled)) {\
|
||||
printf("error: '%s'(%d) from %s at %s:%d\n", \
|
||||
hipGetErrorString(localError), \
|
||||
localError, #error, __FUNCTION__, __LINE__);\
|
||||
exit(0);\
|
||||
}\
|
||||
}
|
||||
#include <iostream>
|
||||
#define HIP_CHECK(error) \
|
||||
{ \
|
||||
hipError_t localError = error; \
|
||||
if ((localError != hipSuccess) && (localError != hipErrorPeerAccessAlreadyEnabled)) { \
|
||||
printf("error: '%s'(%d) from %s at %s:%d\n", hipGetErrorString(localError), localError, \
|
||||
#error, __FUNCTION__, __LINE__); \
|
||||
exit(0); \
|
||||
} \
|
||||
}
|
||||
constexpr auto CODEOBJ_FILE = "kernel_composite_test.code";
|
||||
|
||||
bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
|
||||
char* globTestID) {
|
||||
bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer, char* globTestID) {
|
||||
constexpr auto CODEOBJ_GLOB_KERNEL1 = "testWeightedCopy";
|
||||
size_t N = 16*16;
|
||||
size_t N = 16 * 16;
|
||||
size_t Nbytes = N * sizeof(int);
|
||||
int *A_d, *B_d;
|
||||
int *A_h, *B_h;
|
||||
@@ -47,8 +44,8 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
|
||||
HIP_CHECK(hipMalloc(&A_d, Nbytes));
|
||||
HIP_CHECK(hipMalloc(&B_d, Nbytes));
|
||||
|
||||
A_h = reinterpret_cast<int *>(malloc(Nbytes));
|
||||
B_h = reinterpret_cast<int *>(malloc(Nbytes));
|
||||
A_h = reinterpret_cast<int*>(malloc(Nbytes));
|
||||
B_h = reinterpret_cast<int*>(malloc(Nbytes));
|
||||
// set host buffers
|
||||
for (size_t idx = 0; idx < N; idx++) {
|
||||
A_h[idx] = deviceid;
|
||||
@@ -58,56 +55,37 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
|
||||
hipModule_t Module;
|
||||
hipFunction_t Function;
|
||||
int check = atoi(globTestID);
|
||||
/**
|
||||
* Validates hipModuleLoadUnload if globTestID = 1
|
||||
* Validates hipModuleLoadDataUnload if globTestID = 2
|
||||
* Validates hipModuleLoadDataExUnload if globTestID = 3
|
||||
*/
|
||||
/**
|
||||
* Validates hipModuleLoadUnload if globTestID = 1
|
||||
* Validates hipModuleLoadDataUnload if globTestID = 2
|
||||
* Validates hipModuleLoadDataExUnload if globTestID = 3
|
||||
*/
|
||||
switch (check) {
|
||||
case 1:
|
||||
HIP_CHECK(hipModuleLoad(&Module, CODEOBJ_FILE));
|
||||
case 2:
|
||||
HIP_CHECK(hipModuleLoadData(&Module, &buffer[0]));
|
||||
case 3:
|
||||
HIP_CHECK(hipModuleLoadDataEx(&Module,
|
||||
&buffer[0], 0, nullptr, nullptr));
|
||||
HIP_CHECK(hipModuleLoadDataEx(&Module, &buffer[0], 0, nullptr, nullptr));
|
||||
}
|
||||
HIP_CHECK(hipModuleGetFunction(&Function, Module,
|
||||
CODEOBJ_GLOB_KERNEL1));
|
||||
HIP_CHECK(hipModuleGetFunction(&Function, Module, CODEOBJ_GLOB_KERNEL1));
|
||||
float deviceGlobalFloatH = 3.14;
|
||||
int deviceGlobalInt1H = 100*deviceid;
|
||||
int deviceGlobalInt2H = 50*deviceid;
|
||||
uint32_t deviceGlobalShortH = 25*deviceid;
|
||||
char deviceGlobalCharH = 13*deviceid;
|
||||
int deviceGlobalInt1H = 100 * deviceid;
|
||||
int deviceGlobalInt2H = 50 * deviceid;
|
||||
uint32_t deviceGlobalShortH = 25 * deviceid;
|
||||
char deviceGlobalCharH = 13 * deviceid;
|
||||
hipDeviceptr_t deviceGlobal;
|
||||
size_t deviceGlobalSize;
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
|
||||
&deviceGlobalSize,
|
||||
Module, "deviceGlobalFloat"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
|
||||
&deviceGlobalFloatH,
|
||||
deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
|
||||
&deviceGlobalSize,
|
||||
Module, "deviceGlobalInt1"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
|
||||
&deviceGlobalInt1H,
|
||||
deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
|
||||
&deviceGlobalSize,
|
||||
Module,
|
||||
"deviceGlobalInt2"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
|
||||
&deviceGlobalInt2H, deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
|
||||
&deviceGlobalSize,
|
||||
Module, "deviceGlobalShort"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
|
||||
&deviceGlobalShortH, deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal,
|
||||
&deviceGlobalSize, Module, "deviceGlobalChar"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal),
|
||||
&deviceGlobalCharH, deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalFloat"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalFloatH, deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalInt1"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalInt1H, deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalInt2"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalInt2H, deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalShort"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalShortH, deviceGlobalSize));
|
||||
HIP_CHECK(hipModuleGetGlobal(&deviceGlobal, &deviceGlobalSize, Module, "deviceGlobalChar"));
|
||||
HIP_CHECK(hipMemcpyHtoD(hipDeviceptr_t(deviceGlobal), &deviceGlobalCharH, deviceGlobalSize));
|
||||
// Launch Function kernel function
|
||||
|
||||
hipStream_t stream;
|
||||
@@ -121,12 +99,10 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
|
||||
args._Bd = reinterpret_cast<void*>(B_d);
|
||||
size_t size = sizeof(args);
|
||||
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args,
|
||||
HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
void* config[] = {HIP_LAUNCH_PARAM_BUFFER_POINTER, &args, HIP_LAUNCH_PARAM_BUFFER_SIZE, &size,
|
||||
HIP_LAUNCH_PARAM_END};
|
||||
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1,
|
||||
N, 1, 1, 0, stream, NULL,
|
||||
reinterpret_cast<void**>(&config)));
|
||||
HIP_CHECK(hipModuleLaunchKernel(Function, 1, 1, 1, N, 1, 1, 0, stream, NULL,
|
||||
reinterpret_cast<void**>(&config)));
|
||||
// Copy buffer from decice to host
|
||||
HIP_CHECK(hipMemcpyAsync(B_h, B_d, Nbytes, hipMemcpyDeviceToHost, stream));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
@@ -134,13 +110,12 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
|
||||
|
||||
// Check the results
|
||||
for (size_t idx = 0; idx < N; idx++) {
|
||||
if (B_h[idx] != (deviceGlobalInt1H*A_h[idx]
|
||||
+ deviceGlobalInt2H
|
||||
+ static_cast<int>(deviceGlobalShortH) +
|
||||
+ static_cast<int>(deviceGlobalCharH)
|
||||
+ static_cast<int>(deviceGlobalFloatH*deviceGlobalFloatH))) {
|
||||
// exit the current process with failure
|
||||
return false;
|
||||
if (B_h[idx] !=
|
||||
(deviceGlobalInt1H * A_h[idx] + deviceGlobalInt2H + static_cast<int>(deviceGlobalShortH) +
|
||||
+static_cast<int>(deviceGlobalCharH) +
|
||||
static_cast<int>(deviceGlobalFloatH * deviceGlobalFloatH))) {
|
||||
// exit the current process with failure
|
||||
return false;
|
||||
}
|
||||
}
|
||||
HIP_CHECK(hipModuleUnload(Module));
|
||||
@@ -153,10 +128,9 @@ bool testhipModuleLoadUnloadFunc(const std::vector<char>& buffer,
|
||||
return true;
|
||||
}
|
||||
int main(int argc, char* argv[]) {
|
||||
if(argc > 0) {
|
||||
if (argc > 0) {
|
||||
bool value = false;
|
||||
std::ifstream file(CODEOBJ_FILE,
|
||||
std::ios::binary | std::ios::ate);
|
||||
std::ifstream file(CODEOBJ_FILE, std::ios::binary | std::ios::ate);
|
||||
std::streamsize fsize = file.tellg();
|
||||
file.seekg(0, std::ios::beg);
|
||||
std::vector<char> buffer(fsize);
|
||||
|
||||
@@ -19,6 +19,6 @@ THE SOFTWARE.
|
||||
#include "hip/hip_runtime.h"
|
||||
|
||||
extern "C" __global__ void hello_world(float* a, float* b) {
|
||||
int tx = threadIdx.x;
|
||||
b[tx] = a[tx];
|
||||
int tx = threadIdx.x;
|
||||
b[tx] = a[tx];
|
||||
}
|
||||
|
||||
新しいイシューから参照
ユーザーをブロックする