SWDEV-470698 - fix formatting, add format check workflow (#657)
Bu işleme şunda yer alıyor:
işlemeyi yapan:
GitHub
ebeveyn
5840940caa
işleme
f7338717ae
@@ -51,7 +51,7 @@ static __global__ void vecSqrSingBlk(int* A_d, size_t NELEM) {
|
||||
* - HIP_VERSION >= 6.1
|
||||
*/
|
||||
TEST_CASE("Unit_kernel_Assign_threadIdx_to_auto") {
|
||||
int *A_d;
|
||||
int* A_d;
|
||||
const unsigned blocks = 256;
|
||||
const unsigned threadsPerBlock = 128;
|
||||
size_t N = (blocks * threadsPerBlock);
|
||||
@@ -66,12 +66,11 @@ TEST_CASE("Unit_kernel_Assign_threadIdx_to_auto") {
|
||||
// Transfer data and perform operations on GPU
|
||||
HIP_CHECK(hipMalloc(&A_d, Nbytes));
|
||||
HIP_CHECK(hipMemcpy(A_d, A_h.data(), Nbytes, hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(vecSqrSingBlk, dim3(blocks), dim3(threadsPerBlock),
|
||||
0, 0, A_d, N);
|
||||
hipLaunchKernelGGL(vecSqrSingBlk, dim3(blocks), dim3(threadsPerBlock), 0, 0, A_d, N);
|
||||
HIP_CHECK(hipMemcpy(C_h.data(), A_d, Nbytes, hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
for (size_t i = 0; i < N; i++) {
|
||||
REQUIRE(C_h[i] == (i*i));
|
||||
REQUIRE(C_h[i] == (i * i));
|
||||
}
|
||||
HIP_CHECK(hipFree(A_d));
|
||||
}
|
||||
|
||||
@@ -31,8 +31,8 @@ THE SOFTWARE.
|
||||
* - HIP_VERSION >= 6.0
|
||||
*/
|
||||
TEST_CASE("Unit_ConfigureCall") {
|
||||
struct dim3 grid_dim {};
|
||||
struct dim3 block_dim {};
|
||||
struct dim3 grid_dim{};
|
||||
struct dim3 block_dim{};
|
||||
size_t shared_memory_size = 1024;
|
||||
|
||||
HIP_CHECK(hipConfigureCall(grid_dim, block_dim, shared_memory_size));
|
||||
@@ -50,10 +50,10 @@ TEST_CASE("Unit_ConfigureCall") {
|
||||
* - HIP_VERSION >= 6.0
|
||||
*/
|
||||
TEST_CASE("Unit_ConfigureCall_CheckParams") {
|
||||
struct dim3 grid_dim { 16, 8, 1 };
|
||||
struct dim3 test_grid_dim {};
|
||||
struct dim3 block_dim { 16, 8, 1 };
|
||||
struct dim3 test_block_dim {};
|
||||
struct dim3 grid_dim{16, 8, 1};
|
||||
struct dim3 test_grid_dim{};
|
||||
struct dim3 block_dim{16, 8, 1};
|
||||
struct dim3 test_block_dim{};
|
||||
size_t shmem_size = 1024;
|
||||
size_t test_shmem_size = 0;
|
||||
hipStream_t test_stream;
|
||||
|
||||
@@ -20,7 +20,7 @@ THE SOFTWARE.
|
||||
#include <hip_test_kernels.hh>
|
||||
#include <hip_test_checkers.hh>
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
|
||||
#pragma clang diagnostic ignored "-Wunused-parameter"
|
||||
|
||||
@@ -29,32 +29,29 @@ unsigned threadsPerBlock = 256;
|
||||
template <unsigned batch, typename T>
|
||||
__device__ void sum(T* sdata, unsigned groupElements, unsigned tid) {
|
||||
T tmp;
|
||||
if (groupElements < batch)
|
||||
return;
|
||||
if (groupElements < batch) return;
|
||||
// sdata[tid] += sdata[tid - batch/2] does not work when block size is
|
||||
// greater than wave size because one wave may complete before another
|
||||
// wave.
|
||||
if (tid >= batch/2 && tid < groupElements)
|
||||
tmp = sdata[tid - batch/2];
|
||||
if (tid >= batch / 2 && tid < groupElements) tmp = sdata[tid - batch / 2];
|
||||
__syncthreads();
|
||||
if (tid >= batch/2 && tid < groupElements)
|
||||
sdata[tid] += tmp;
|
||||
if (tid >= batch / 2 && tid < groupElements) sdata[tid] += tmp;
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void testExternSharedKernel(const T* A_d, const T* B_d, T* C_d,
|
||||
size_t numElements, size_t groupElements) {
|
||||
__global__ void testExternSharedKernel(const T* A_d, const T* B_d, T* C_d, size_t numElements,
|
||||
size_t groupElements) {
|
||||
// declare dynamic shared memory
|
||||
extern __shared__ double sdata0[];
|
||||
T* sdata = reinterpret_cast<T *>(sdata0);
|
||||
T* sdata = reinterpret_cast<T*>(sdata0);
|
||||
|
||||
size_t gid = (blockIdx.x * blockDim.x + threadIdx.x);
|
||||
size_t tid = threadIdx.x;
|
||||
|
||||
// initialize dynamic shared memory
|
||||
if (tid < groupElements) {
|
||||
sdata[tid] = static_cast<T>(tid);
|
||||
sdata[tid] = static_cast<T>(tid);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
@@ -71,15 +68,14 @@ __global__ void testExternSharedKernel(const T* A_d, const T* B_d, T* C_d,
|
||||
C_d[gid] = A_d[gid] + B_d[gid] + sdata[tid % groupElements];
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void testExternShared(size_t N, unsigned groupElements) {
|
||||
template <typename T> void testExternShared(size_t N, unsigned groupElements) {
|
||||
size_t Nbytes = N * sizeof(T);
|
||||
|
||||
T *A_d, *B_d, *C_d;
|
||||
T *A_h, *B_h, *C_h;
|
||||
|
||||
HipTest::initArrays(&A_d, &B_d, &C_d, &A_h, &B_h, &C_h, N, false);
|
||||
unsigned blocks = N/threadsPerBlock;
|
||||
unsigned blocks = N / threadsPerBlock;
|
||||
assert(N == blocks * threadsPerBlock);
|
||||
|
||||
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
|
||||
@@ -90,8 +86,7 @@ void testExternShared(size_t N, unsigned groupElements) {
|
||||
|
||||
// launch kernel with dynamic shared memory
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(testExternSharedKernel<T>), dim3(blocks),
|
||||
dim3(threadsPerBlock), groupMemBytes, 0, A_d, B_d, C_d,
|
||||
N, groupElements);
|
||||
dim3(threadsPerBlock), groupMemBytes, 0, A_d, B_d, C_d, N, groupElements);
|
||||
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
|
||||
@@ -164,13 +159,12 @@ TEST_CASE("Unit_hipDynamicShared") {
|
||||
|
||||
SECTION("test case with float for max LDS size") {
|
||||
int maxLDS = 0;
|
||||
HIP_CHECK(hipDeviceGetAttribute(&maxLDS,
|
||||
hipDeviceAttributeMaxSharedMemoryPerBlock, 0));
|
||||
testExternShared<float>(1024, maxLDS/sizeof(float));
|
||||
HIP_CHECK(hipDeviceGetAttribute(&maxLDS, hipDeviceAttributeMaxSharedMemoryPerBlock, 0));
|
||||
testExternShared<float>(1024, maxLDS / sizeof(float));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -20,9 +20,9 @@ THE SOFTWARE.
|
||||
#include <hip_test_kernels.hh>
|
||||
#include <hip_test_checkers.hh>
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
#define LEN (16 * 1024)
|
||||
|
||||
#define LEN (16 * 1024)
|
||||
#define SIZE (LEN * sizeof(float))
|
||||
|
||||
__global__ void vectorAdd(float* Ad, float* Bd) {
|
||||
@@ -46,7 +46,7 @@ __global__ void vectorAdd(float* Ad, float* Bd) {
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Assign max dynamic shared memory to kernel function and
|
||||
* - Assign max dynamic shared memory to kernel function and
|
||||
* verify the results.
|
||||
|
||||
* Test source
|
||||
@@ -62,17 +62,16 @@ TEST_CASE("Unit_hipDynamicShared2") {
|
||||
A = new float[LEN];
|
||||
B = new float[LEN];
|
||||
for (int i = 0; i < LEN; i++) {
|
||||
A[i] = 1.0f;
|
||||
B[i] = 1.0f;
|
||||
A[i] = 1.0f;
|
||||
B[i] = 1.0f;
|
||||
}
|
||||
HIP_CHECK(hipMalloc(&Ad, SIZE));
|
||||
HIP_CHECK(hipMalloc(&Bd, SIZE));
|
||||
HIP_CHECK(hipMemcpy(Ad, A, SIZE, hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(Bd, B, SIZE, hipMemcpyHostToDevice));
|
||||
|
||||
hipError_t ret = hipFuncSetAttribute(
|
||||
reinterpret_cast<const void*>(&vectorAdd),
|
||||
hipFuncAttributeMaxDynamicSharedMemorySize, SIZE);
|
||||
hipError_t ret = hipFuncSetAttribute(reinterpret_cast<const void*>(&vectorAdd),
|
||||
hipFuncAttributeMaxDynamicSharedMemorySize, SIZE);
|
||||
|
||||
REQUIRE(ret == hipSuccess);
|
||||
hipLaunchKernelGGL(vectorAdd, dim3(1, 1, 1), dim3(64, 1, 1), SIZE, 0, Ad, Bd);
|
||||
@@ -89,6 +88,6 @@ TEST_CASE("Unit_hipDynamicShared2") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -20,7 +20,7 @@ THE SOFTWARE.
|
||||
#include <hip_test_kernels.hh>
|
||||
#include <hip_test_checkers.hh>
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
|
||||
#pragma clang diagnostic ignored "-Wunused-parameter"
|
||||
|
||||
@@ -49,11 +49,11 @@ __global__ void Empty(int param) {}
|
||||
*/
|
||||
|
||||
TEST_CASE("Unit_hipEmptyKernel") {
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(Empty), dim3(1), dim3(1), 0, 0, 0);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(Empty), dim3(1), dim3(1), 0, 0, 0);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -21,37 +21,35 @@ THE SOFTWARE.
|
||||
#include <hip_test_kernels.hh>
|
||||
#include <hip_test_checkers.hh>
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
#include "hip/hip_ext.h"
|
||||
|
||||
static unsigned threadsPerBlock = 256;
|
||||
static unsigned blocksPerCU = 6;
|
||||
|
||||
struct _t {
|
||||
double _a, _b, _c, _d, _e, _f, _g, _h, _i, _j;
|
||||
double _a, _b, _c, _d, _e, _f, _g, _h, _i, _j;
|
||||
};
|
||||
|
||||
typedef struct _t _T;
|
||||
|
||||
__global__ void sKernel(_T s, double *a) {
|
||||
*a = s._a + s._b + s._c + s._d + s._e + s._f + s._g + s._h + s._i + s._j;
|
||||
__global__ void sKernel(_T s, double* a) {
|
||||
*a = s._a + s._b + s._c + s._d + s._e + s._f + s._g + s._h + s._i + s._j;
|
||||
}
|
||||
|
||||
__global__ void mKernel(char f, int16_t a, int b, double c,
|
||||
int16_t d, int e, double* res) {
|
||||
*res = a + b + c + d + e + f;
|
||||
__global__ void mKernel(char f, int16_t a, int b, double c, int16_t d, int e, double* res) {
|
||||
*res = a + b + c + d + e + f;
|
||||
}
|
||||
|
||||
void testMixData() {
|
||||
double m = 0;
|
||||
double *d_m;
|
||||
double* d_m;
|
||||
HIP_CHECK(hipMalloc(&d_m, sizeof(double)));
|
||||
int a = 1, e = 10;
|
||||
int16_t b = 2, d = 4;
|
||||
double c = 3.0;
|
||||
char ff = 10;
|
||||
hipExtLaunchKernelGGL(mKernel, 1, 1, 0, 0, nullptr, nullptr, 0, ff,
|
||||
b, a, c, d, e, d_m);
|
||||
hipExtLaunchKernelGGL(mKernel, 1, 1, 0, 0, nullptr, nullptr, 0, ff, b, a, c, d, e, d_m);
|
||||
HIP_CHECK(hipMemcpy(&m, d_m, sizeof(double), hipMemcpyDeviceToHost));
|
||||
REQUIRE(m == 30.0);
|
||||
HIP_CHECK(hipFree(d_m));
|
||||
@@ -59,7 +57,7 @@ void testMixData() {
|
||||
|
||||
void testStruct() {
|
||||
double m = 0;
|
||||
double *d_m;
|
||||
double* d_m;
|
||||
HIP_CHECK(hipMalloc(&d_m, sizeof(double)));
|
||||
_T s{1, 2, 3, 4, 5, 6, 7, 8, 9, 10};
|
||||
hipExtLaunchKernelGGL(sKernel, 1, 1, 0, 0, nullptr, nullptr, 0, s, d_m);
|
||||
@@ -80,10 +78,9 @@ void test(size_t N) {
|
||||
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(B_d, B_h, Nbytes, hipMemcpyHostToDevice));
|
||||
|
||||
hipExtLaunchKernelGGL(HipTest::vectorADD, dim3(blocks),
|
||||
dim3(threadsPerBlock), 0, 0, nullptr, nullptr, 0,
|
||||
static_cast<const int*>(A_d),
|
||||
static_cast<const int*>(B_d), C_d, N);
|
||||
hipExtLaunchKernelGGL(HipTest::vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, 0, nullptr,
|
||||
nullptr, 0, static_cast<const int*>(A_d), static_cast<const int*>(B_d), C_d,
|
||||
N);
|
||||
|
||||
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
@@ -99,7 +96,8 @@ void test(size_t N) {
|
||||
std::uint32_t sharedMemBytes, hipStream_t stream,
|
||||
hipEvent_t startEvent, hipEvent_t stopEvent, std::uint32_t flags,
|
||||
Args... args)` -
|
||||
* Launches kernel with dimention parameters and shared memory on stream with templated kernel and arguments
|
||||
* Launches kernel with dimention parameters and shared memory on stream with templated kernel and
|
||||
arguments
|
||||
*/
|
||||
|
||||
/**
|
||||
@@ -125,15 +123,11 @@ TEST_CASE("Unit_hipExtLaunchKernelGGL") {
|
||||
size_t N = 4 * 1024 * 1024;
|
||||
test(N);
|
||||
}
|
||||
SECTION("testStruct run") {
|
||||
testStruct();
|
||||
}
|
||||
SECTION("testMixData run") {
|
||||
testMixData();
|
||||
}
|
||||
SECTION("testStruct run") { testStruct(); }
|
||||
SECTION("testMixData run") { testMixData(); }
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -21,7 +21,7 @@ THE SOFTWARE.
|
||||
#include <hip_test_kernels.hh>
|
||||
#include <hip_test_checkers.hh>
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
|
||||
static unsigned threadsPerBlock = 256;
|
||||
static unsigned blocksPerCU = 6;
|
||||
@@ -30,15 +30,14 @@ static unsigned blocksPerCU = 6;
|
||||
__device__ int foo(int i) { return i + 1; }
|
||||
|
||||
|
||||
template <typename T>
|
||||
__global__ void vectorADD2(T* A_d, T* B_d, T* C_d, size_t N) {
|
||||
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
|
||||
size_t stride = blockDim.x * gridDim.x;
|
||||
template <typename T> __global__ void vectorADD2(T* A_d, T* B_d, T* C_d, size_t N) {
|
||||
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
|
||||
size_t stride = blockDim.x * gridDim.x;
|
||||
|
||||
for (size_t i = offset; i < N; i += stride) {
|
||||
double foo = __hiloint2double(A_d[i], B_d[i]);
|
||||
C_d[i] = __double2loint(foo) + __double2hiint(foo);
|
||||
}
|
||||
for (size_t i = offset; i < N; i += stride) {
|
||||
double foo = __hiloint2double(A_d[i], B_d[i]);
|
||||
C_d[i] = __double2loint(foo) + __double2hiint(foo);
|
||||
}
|
||||
}
|
||||
|
||||
int test_gl2(size_t N) {
|
||||
@@ -52,8 +51,7 @@ int test_gl2(size_t N) {
|
||||
// Full vadd in one large chunk, to get things started:
|
||||
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(B_d, B_h, Nbytes, hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(vectorADD2, dim3(blocks), dim3(threadsPerBlock),
|
||||
0, 0, A_d, B_d, C_d, N);
|
||||
hipLaunchKernelGGL(vectorADD2, dim3(blocks), dim3(threadsPerBlock), 0, 0, A_d, B_d, C_d, N);
|
||||
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
// verify
|
||||
@@ -107,18 +105,14 @@ int test_triple_chevron(size_t N) {
|
||||
|
||||
TEST_CASE("Unit_hipGridLaunch") {
|
||||
size_t N = 4 * 1024 * 1024;
|
||||
SECTION("Test test_gl2") {
|
||||
test_gl2(N);
|
||||
}
|
||||
SECTION("Test test_gl2") { test_gl2(N); }
|
||||
|
||||
#if __HIP__
|
||||
SECTION("Test triple_chevron") {
|
||||
test_triple_chevron(N);
|
||||
}
|
||||
SECTION("Test triple_chevron") { test_triple_chevron(N); }
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -20,7 +20,7 @@ THE SOFTWARE.
|
||||
#include <hip_test_kernels.hh>
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip_test_checkers.hh>
|
||||
|
||||
|
||||
#include <hip/math_functions.h>
|
||||
|
||||
#pragma clang diagnostic ignored "-Wunused-variable"
|
||||
@@ -43,11 +43,10 @@ __device__ __forceinline__ int sum1_forceinline(int a) { return a + 1; }
|
||||
|
||||
__device__ __host__ float PlusOne(float x) { return x + 1.0; }
|
||||
|
||||
__global__ void MyKernel(const float* a, const float* b, float* c,
|
||||
unsigned N) {
|
||||
__global__ void MyKernel(const float* a, const float* b, float* c, unsigned N) {
|
||||
unsigned gid = threadIdx.x;
|
||||
if (gid < N) {
|
||||
c[gid] = a[gid] + PlusOne(b[gid]);
|
||||
c[gid] = a[gid] + PlusOne(b[gid]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -55,18 +54,16 @@ void callMyKernel() {
|
||||
float *a, *b, *c;
|
||||
const unsigned blockSize = 256;
|
||||
unsigned N = blockSize;
|
||||
hipLaunchKernelGGL(MyKernel, dim3(N / blockSize), dim3(blockSize),
|
||||
0, 0, a, b, c, N);
|
||||
hipLaunchKernelGGL(MyKernel, dim3(N / blockSize), dim3(blockSize), 0, 0, a, b, c, N);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void vectorADD(T __restrict__* A_d, T* B_d, T* C_d, size_t N) {
|
||||
template <typename T> __global__ void vectorADD(T __restrict__* A_d, T* B_d, T* C_d, size_t N) {
|
||||
#ifdef NOT_YET
|
||||
int a = __shfl_up(x, 1);
|
||||
#endif
|
||||
float x = 1.0;
|
||||
#ifdef NOT_YET
|
||||
float fastZ = __sin(x);
|
||||
float fastZ = __sin(x);
|
||||
#endif
|
||||
__syncthreads();
|
||||
|
||||
@@ -74,7 +71,7 @@ __global__ void vectorADD(T __restrict__* A_d, T* B_d, T* C_d, size_t N) {
|
||||
size_t stride = blockDim.x * gridDim.x;
|
||||
|
||||
for (size_t i = offset; i < N; i += stride) {
|
||||
C_d[i] = A_d[i] + B_d[i];
|
||||
C_d[i] = A_d[i] + B_d[i];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -101,11 +98,9 @@ __global__ void vectorADD(T __restrict__* A_d, T* B_d, T* C_d, size_t N) {
|
||||
* - HIP_VERSION >= 5.5
|
||||
*/
|
||||
|
||||
TEST_CASE("Unit_hipLanguageExtensions") {
|
||||
REQUIRE(true);
|
||||
}
|
||||
TEST_CASE("Unit_hipLanguageExtensions") { REQUIRE(true); }
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -37,23 +37,19 @@ Testcase Scenarios : hipLaunchBounds_With_maxThreadsPerBlock_blocksPerCU
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip_test_kernels.hh>
|
||||
|
||||
__global__ void
|
||||
__launch_bounds__(128, 2)
|
||||
MyKernel(int N, int *x, int val) {
|
||||
__global__ void __launch_bounds__(128, 2) MyKernel(int N, int* x, int val) {
|
||||
for (int i = 0; i < N; i++) {
|
||||
x[i] = val;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void
|
||||
__launch_bounds__(64)
|
||||
MyKernel_2(int N, int *x, int val) {
|
||||
__global__ void __launch_bounds__(64) MyKernel_2(int N, int* x, int val) {
|
||||
for (int i = 0; i < N; i++) {
|
||||
x[i] = val;
|
||||
}
|
||||
}
|
||||
|
||||
static bool verify(int N, int *x, int val) {
|
||||
static bool verify(int N, int* x, int val) {
|
||||
for (int i = 0; i < N; i++) {
|
||||
if (x[i] != val) {
|
||||
return false;
|
||||
@@ -65,9 +61,9 @@ static bool verify(int N, int *x, int val) {
|
||||
TEST_CASE("Unit_hipLaunchBounds_With_maxThreadsPerBlock_Check") {
|
||||
constexpr size_t N = 10000;
|
||||
hipError_t ret;
|
||||
int *x;
|
||||
int* x;
|
||||
|
||||
HIP_CHECK(hipMallocManaged(&x, N*sizeof(int)));
|
||||
HIP_CHECK(hipMallocManaged(&x, N * sizeof(int)));
|
||||
REQUIRE(x != nullptr);
|
||||
|
||||
SECTION("Passing threadsPerBlock same as kernel launch_bounds") {
|
||||
@@ -105,9 +101,9 @@ TEST_CASE("Unit_hipLaunchBounds_With_maxThreadsPerBlock_Check") {
|
||||
TEST_CASE("Unit_hipLaunchBounds_With_maxThreadsPerBlock_blocksPerCU_Check") {
|
||||
constexpr size_t N = 10000;
|
||||
hipError_t ret;
|
||||
int *x;
|
||||
int* x;
|
||||
|
||||
HIP_CHECK(hipMallocManaged(&x, N*sizeof(int)));
|
||||
HIP_CHECK(hipMallocManaged(&x, N * sizeof(int)));
|
||||
REQUIRE(x != nullptr);
|
||||
|
||||
SECTION("Passing threadsPerBlock same as kernel launch_bounds") {
|
||||
@@ -170,4 +166,3 @@ TEST_CASE("Unit_hipLaunchBounds_With_maxThreadsPerBlock_blocksPerCU_Check") {
|
||||
|
||||
HIP_CHECK(hipFree(x));
|
||||
}
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ static __device__ int devArr[N];
|
||||
//------------------------------------------------------------------------------
|
||||
// Kernel using hipLaunchKernelEx
|
||||
//------------------------------------------------------------------------------
|
||||
__global__ void cooperativeKernelEx(int *output, int totalThreads) {
|
||||
__global__ void cooperativeKernelEx(int* output, int totalThreads) {
|
||||
cooperative_groups::grid_group grid = cooperative_groups::this_grid();
|
||||
int tid = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if (tid < totalThreads) {
|
||||
@@ -41,7 +41,7 @@ __global__ void cooperativeKernelEx(int *output, int totalThreads) {
|
||||
//------------------------------------------------------------------------------
|
||||
// Kernel using hipLaunchKernelExC
|
||||
//------------------------------------------------------------------------------
|
||||
__global__ void cooperativeKernelExC(int *output, int totalThreads) {
|
||||
__global__ void cooperativeKernelExC(int* output, int totalThreads) {
|
||||
cooperative_groups::grid_group grid = cooperative_groups::this_grid();
|
||||
int tid = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if (tid < totalThreads) {
|
||||
@@ -61,7 +61,7 @@ static __global__ void emptyKernel() {}
|
||||
* Kernel which doesn't use cooperative groups and takes an argument
|
||||
* and updates the value with 100
|
||||
*/
|
||||
static __global__ void argKernel(int *val) { *val = 100; }
|
||||
static __global__ void argKernel(int* val) { *val = 100; }
|
||||
|
||||
/*
|
||||
* Kernel which uses cooperative groups and without any arguments
|
||||
@@ -79,7 +79,7 @@ static __global__ void coopEmptykernel() {
|
||||
* 2) Wait for all the blocks completes it's operations
|
||||
* 3) Fill each element in the output array with sum of elements in devArr
|
||||
*/
|
||||
static __global__ void coopFillArrayKernel(int *output) {
|
||||
static __global__ void coopFillArrayKernel(int* output) {
|
||||
cooperative_groups::grid_group grid = cooperative_groups::this_grid();
|
||||
|
||||
if (blockIdx.x == 0)
|
||||
@@ -110,7 +110,7 @@ static __global__ void coopFillArrayKernel(int *output) {
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void normalKernel(int *output, int totalThreads) {
|
||||
__global__ void normalKernel(int* output, int totalThreads) {
|
||||
int tid = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if (tid < totalThreads) {
|
||||
output[tid] = tid * 3;
|
||||
@@ -157,10 +157,10 @@ TEST_CASE("Unit_hipLaunchKernelExC_NegetiveTsts") {
|
||||
config.attrs = &attr;
|
||||
config.numAttrs = 1;
|
||||
|
||||
int *d_output = nullptr;
|
||||
int* d_output = nullptr;
|
||||
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
|
||||
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
|
||||
void *kernelArgs[] = {&d_output, (void *)&totalThreads};
|
||||
void* kernelArgs[] = {&d_output, (void*)&totalThreads};
|
||||
|
||||
SECTION("Kernel function as nullptr") {
|
||||
HIP_CHECK_ERROR(hipLaunchKernelExC(&config, nullptr, kernelArgs),
|
||||
@@ -168,15 +168,12 @@ TEST_CASE("Unit_hipLaunchKernelExC_NegetiveTsts") {
|
||||
}
|
||||
|
||||
SECTION("Kernel args as nullptr") {
|
||||
HIP_CHECK_ERROR(
|
||||
hipLaunchKernelExC(&config, (void *)cooperativeKernelExC, nullptr),
|
||||
hipErrorInvalidValue);
|
||||
HIP_CHECK_ERROR(hipLaunchKernelExC(&config, (void*)cooperativeKernelExC, nullptr),
|
||||
hipErrorInvalidValue);
|
||||
}
|
||||
|
||||
SECTION("Non Cooparative Kernel") {
|
||||
HIP_CHECK_ERROR(
|
||||
hipLaunchKernelExC(&config, (void *)normalKernel, kernelArgs),
|
||||
hipSuccess);
|
||||
HIP_CHECK_ERROR(hipLaunchKernelExC(&config, (void*)normalKernel, kernelArgs), hipSuccess);
|
||||
}
|
||||
|
||||
hipLaunchConfig_t invalidConfig = {};
|
||||
@@ -192,9 +189,7 @@ TEST_CASE("Unit_hipLaunchKernelExC_NegetiveTsts") {
|
||||
invalidConfig.numAttrs = 1;
|
||||
|
||||
SECTION("Invalid Kernel Config") {
|
||||
HIP_CHECK_ERROR(hipLaunchKernelExC(&invalidConfig,
|
||||
(void *)cooperativeKernelExC,
|
||||
kernelArgs),
|
||||
HIP_CHECK_ERROR(hipLaunchKernelExC(&invalidConfig, (void*)cooperativeKernelExC, kernelArgs),
|
||||
hipErrorInvalidConfiguration);
|
||||
}
|
||||
}
|
||||
@@ -231,7 +226,7 @@ TEST_CASE("Unit_hipLaunchKernelEx_NegetiveTsts") {
|
||||
config.attrs = &attr;
|
||||
config.numAttrs = 1;
|
||||
|
||||
int *d_output = nullptr;
|
||||
int* d_output = nullptr;
|
||||
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
|
||||
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
|
||||
|
||||
@@ -240,10 +235,9 @@ TEST_CASE("Unit_hipLaunchKernelEx_NegetiveTsts") {
|
||||
}
|
||||
|
||||
SECTION("Non Cooparative Kernel") {
|
||||
HIP_CHECK_ERROR(hipLaunchKernelEx(&config,
|
||||
(void (*)(int *, int))normalKernel,
|
||||
d_output, totalThreads),
|
||||
hipSuccess);
|
||||
HIP_CHECK_ERROR(
|
||||
hipLaunchKernelEx(&config, (void (*)(int*, int))normalKernel, d_output, totalThreads),
|
||||
hipSuccess);
|
||||
}
|
||||
|
||||
hipLaunchConfig_t invalidConfig = {};
|
||||
@@ -259,18 +253,17 @@ TEST_CASE("Unit_hipLaunchKernelEx_NegetiveTsts") {
|
||||
invalidConfig.numAttrs = 1;
|
||||
|
||||
SECTION("Invalid Kernel Config") {
|
||||
HIP_CHECK_ERROR(hipLaunchKernelEx(&invalidConfig,
|
||||
(void (*)(int *, int))cooperativeKernelEx,
|
||||
HIP_CHECK_ERROR(hipLaunchKernelEx(&invalidConfig, (void (*)(int*, int))cooperativeKernelEx,
|
||||
d_output, totalThreads),
|
||||
hipErrorInvalidConfiguration);
|
||||
}
|
||||
}
|
||||
|
||||
bool runTest(const char *testName, const void *kernelFunc, int totalThreads,
|
||||
int blockSize, int flagValue, bool useTemplate) {
|
||||
bool runTest(const char* testName, const void* kernelFunc, int totalThreads, int blockSize,
|
||||
int flagValue, bool useTemplate) {
|
||||
const int numBlocks = (totalThreads + blockSize - 1) / blockSize;
|
||||
|
||||
int *d_output = nullptr;
|
||||
int* d_output = nullptr;
|
||||
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
|
||||
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
|
||||
|
||||
@@ -288,33 +281,31 @@ bool runTest(const char *testName, const void *kernelFunc, int totalThreads,
|
||||
|
||||
// For a kernel parameter declared as "int* output", pass the address of the
|
||||
// device pointer.
|
||||
void *kernelArgs[] = {&d_output, (void *)&totalThreads};
|
||||
void* kernelArgs[] = {&d_output, (void*)&totalThreads};
|
||||
|
||||
if (useTemplate) {
|
||||
HIP_CHECK(hipLaunchKernelEx(&config, (void (*)(int *, int))kernelFunc,
|
||||
d_output, totalThreads));
|
||||
HIP_CHECK(hipLaunchKernelEx(&config, (void (*)(int*, int))kernelFunc, d_output, totalThreads));
|
||||
} else {
|
||||
HIP_CHECK(hipLaunchKernelExC(&config, kernelFunc, kernelArgs));
|
||||
}
|
||||
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
int *h_output = (int *)malloc(totalThreads * sizeof(int));
|
||||
HIP_CHECK(hipMemcpy(h_output, d_output, totalThreads * sizeof(int),
|
||||
hipMemcpyDeviceToHost));
|
||||
int* h_output = (int*)malloc(totalThreads * sizeof(int));
|
||||
HIP_CHECK(hipMemcpy(h_output, d_output, totalThreads * sizeof(int), hipMemcpyDeviceToHost));
|
||||
|
||||
// Verify results.
|
||||
bool success = true;
|
||||
if (h_output[0] != flagValue) {
|
||||
printf("%s test failed: Expected flag %d at index 0, got %d\n", testName,
|
||||
flagValue, h_output[0]);
|
||||
printf("%s test failed: Expected flag %d at index 0, got %d\n", testName, flagValue,
|
||||
h_output[0]);
|
||||
success = false;
|
||||
}
|
||||
for (int i = 1; i < totalThreads; i++) {
|
||||
int expectedValue = (flagValue == 1111) ? i : (i * 3);
|
||||
if (h_output[i] != expectedValue) {
|
||||
printf("%s test failed at index %d: Expected %d, got %d\n", testName, i,
|
||||
expectedValue, h_output[i]);
|
||||
printf("%s test failed at index %d: Expected %d, got %d\n", testName, i, expectedValue,
|
||||
h_output[i]);
|
||||
success = false;
|
||||
break;
|
||||
}
|
||||
@@ -344,12 +335,10 @@ TEST_CASE("Unit_hipLaunchKernelEx_Functional") {
|
||||
}
|
||||
std::string api_type = GENERATE("hipLaunchKernelEx", "hipLaunchKernelExC");
|
||||
if (api_type == "hipLaunchKernelEx") {
|
||||
REQUIRE(runTest(api_type.c_str(), (void *)cooperativeKernelEx, 64, 16, 2222,
|
||||
true) == true);
|
||||
REQUIRE(runTest(api_type.c_str(), (void*)cooperativeKernelEx, 64, 16, 2222, true) == true);
|
||||
}
|
||||
if (api_type == "hipLaunchKernelExC") {
|
||||
REQUIRE(runTest(api_type.c_str(), (void *)cooperativeKernelExC, 64, 16,
|
||||
1111, false) == true);
|
||||
REQUIRE(runTest(api_type.c_str(), (void*)cooperativeKernelExC, 64, 16, 1111, false) == true);
|
||||
}
|
||||
}
|
||||
/**
|
||||
@@ -391,13 +380,10 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_Different_Kernels") {
|
||||
config.numAttrs = 1;
|
||||
|
||||
SECTION("Normal kernel with no arguments") {
|
||||
SECTION("hipLaunchKernelEx") {
|
||||
HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel));
|
||||
}
|
||||
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel)); }
|
||||
|
||||
SECTION("hipLaunchKernelExC") {
|
||||
HIP_CHECK(hipLaunchKernelExC(
|
||||
&config, reinterpret_cast<void *>(emptyKernel), nullptr));
|
||||
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(emptyKernel), nullptr));
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
}
|
||||
@@ -406,17 +392,14 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_Different_Kernels") {
|
||||
config.gridDim = dim3{1, 1, 1};
|
||||
config.blockDim = dim3{1, 1, 1};
|
||||
|
||||
int *devMem = nullptr;
|
||||
int* devMem = nullptr;
|
||||
HIP_CHECK(hipMalloc(&devMem, sizeof(int)));
|
||||
|
||||
SECTION("hipLaunchKernelEx") {
|
||||
HIP_CHECK(hipLaunchKernelEx(&config, argKernel, devMem));
|
||||
}
|
||||
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, argKernel, devMem)); }
|
||||
|
||||
SECTION("hipLaunchKernelExC") {
|
||||
void *kernel_args[1] = {&devMem};
|
||||
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void *>(argKernel),
|
||||
kernel_args));
|
||||
void* kernel_args[1] = {&devMem};
|
||||
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(argKernel), kernel_args));
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
@@ -426,13 +409,10 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_Different_Kernels") {
|
||||
}
|
||||
|
||||
SECTION("Cooperative kernel with no arguments") {
|
||||
SECTION("hipLaunchKernelEx") {
|
||||
HIP_CHECK(hipLaunchKernelEx(&config, coopEmptykernel));
|
||||
}
|
||||
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, coopEmptykernel)); }
|
||||
|
||||
SECTION("hipLaunchKernelExC") {
|
||||
HIP_CHECK(hipLaunchKernelExC(
|
||||
&config, reinterpret_cast<void *>(coopEmptykernel), nullptr));
|
||||
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(coopEmptykernel), nullptr));
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
}
|
||||
@@ -475,7 +455,7 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_CooperativeKernelWithArgs") {
|
||||
hostMem[i] = 0;
|
||||
}
|
||||
|
||||
int *devMem = nullptr;
|
||||
int* devMem = nullptr;
|
||||
HIP_CHECK(hipMalloc(&devMem, N * sizeof(int)));
|
||||
HIP_CHECK(hipMemcpy(devMem, hostMem, N * sizeof(int), hipMemcpyDefault));
|
||||
|
||||
@@ -484,9 +464,9 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_CooperativeKernelWithArgs") {
|
||||
}
|
||||
|
||||
SECTION("hipLaunchKernelExC") {
|
||||
void *kernel_args[1] = {&devMem};
|
||||
HIP_CHECK(hipLaunchKernelExC(
|
||||
&config, reinterpret_cast<void *>(coopFillArrayKernel), kernel_args));
|
||||
void* kernel_args[1] = {&devMem};
|
||||
HIP_CHECK(
|
||||
hipLaunchKernelExC(&config, reinterpret_cast<void*>(coopFillArrayKernel), kernel_args));
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
HIP_CHECK(hipMemcpy(hostMem, devMem, N * sizeof(int), hipMemcpyDefault));
|
||||
@@ -534,47 +514,35 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_MaxBlockDims") {
|
||||
config.numAttrs = 1;
|
||||
|
||||
SECTION("blockDim.x == maxBlockDimX") {
|
||||
const unsigned int x =
|
||||
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimX, 0);
|
||||
const unsigned int x = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimX, 0);
|
||||
config.blockDim = dim3{x, 1, 1};
|
||||
|
||||
SECTION("hipLaunchKernelEx") {
|
||||
HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel));
|
||||
}
|
||||
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel)); }
|
||||
|
||||
SECTION("hipLaunchKernelExC") {
|
||||
HIP_CHECK(hipLaunchKernelExC(
|
||||
&config, reinterpret_cast<void *>(emptyKernel), nullptr));
|
||||
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(emptyKernel), nullptr));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("blockDim.y == maxBlockDimY") {
|
||||
const unsigned int y =
|
||||
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimY, 0);
|
||||
const unsigned int y = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimY, 0);
|
||||
config.blockDim = dim3{1, y, 1};
|
||||
|
||||
SECTION("hipLaunchKernelEx") {
|
||||
HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel));
|
||||
}
|
||||
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel)); }
|
||||
|
||||
SECTION("hipLaunchKernelExC") {
|
||||
HIP_CHECK(hipLaunchKernelExC(
|
||||
&config, reinterpret_cast<void *>(emptyKernel), nullptr));
|
||||
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(emptyKernel), nullptr));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("blockDim.z == maxBlockDimZ") {
|
||||
const unsigned int z =
|
||||
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimZ, 0);
|
||||
const unsigned int z = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimZ, 0);
|
||||
config.blockDim = dim3{1, 1, z};
|
||||
|
||||
SECTION("hipLaunchKernelEx") {
|
||||
HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel));
|
||||
}
|
||||
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel)); }
|
||||
|
||||
SECTION("hipLaunchKernelExC") {
|
||||
HIP_CHECK(hipLaunchKernelExC(
|
||||
&config, reinterpret_cast<void *>(emptyKernel), nullptr));
|
||||
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(emptyKernel), nullptr));
|
||||
}
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
@@ -54,7 +54,7 @@ THE SOFTWARE.
|
||||
// Bit fields are broken
|
||||
#define ENABLE_BIT_FIELDS 0
|
||||
|
||||
static const int BLOCK_DIM_SIZE = 512;
|
||||
static const int BLOCK_DIM_SIZE = 512;
|
||||
|
||||
// allocate memory on device and host for result validation
|
||||
static bool *result_d, *result_h;
|
||||
@@ -64,8 +64,7 @@ static hipError_t hipHostMallocError = hipErrorUnknown;
|
||||
static hipError_t hipMemsetError = hipErrorUnknown;
|
||||
|
||||
static void ResultValidation() {
|
||||
HIP_CHECK(hipMemcpy(result_h, result_d, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipMemcpy(result_h, result_d, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
REQUIRE(result_h[k] == true);
|
||||
@@ -86,8 +85,8 @@ static void ResetValidationMem() {
|
||||
// This test is to verify Struct with variables
|
||||
// support, read from device.
|
||||
typedef struct hipLaunchKernelStruct1 {
|
||||
int li; // local int
|
||||
float lf; // local float
|
||||
int li; // local int
|
||||
float lf; // local float
|
||||
bool result; // local bool
|
||||
} hipLaunchKernelStruct_t1;
|
||||
|
||||
@@ -126,14 +125,14 @@ typedef struct hipLaunchKernelStruct5 {
|
||||
typedef struct hipLaunchKernelStruct6 {
|
||||
char c1;
|
||||
int16_t si;
|
||||
} __attribute__((aligned(8))) hipLaunchKernelStruct_t6;
|
||||
} __attribute__((aligned(8))) hipLaunchKernelStruct_t6;
|
||||
|
||||
// This test is to verify struct with aligned(16),
|
||||
// right now it's brokenon hcc & hip-clang
|
||||
typedef struct hipLaunchKernelStruct7 {
|
||||
char c1;
|
||||
int16_t si;
|
||||
} __attribute__((aligned(16))) hipLaunchKernelStruct_t7;
|
||||
} __attribute__((aligned(16))) hipLaunchKernelStruct_t7;
|
||||
|
||||
// This test is to verify struct with packed & aligned,
|
||||
// size should be 4Bytes right now it's broken on hcc & hip-clang
|
||||
@@ -141,7 +140,7 @@ typedef struct hipLaunchKernelStruct8 {
|
||||
char c1;
|
||||
int16_t si;
|
||||
bool b;
|
||||
}__attribute__((packed, aligned(4))) hipLaunchKernelStruct_t8;
|
||||
} __attribute__((packed, aligned(4))) hipLaunchKernelStruct_t8;
|
||||
|
||||
// This test is to verify struct with packed, no alignment as Sam suggested
|
||||
// size should be 4Bytes, right now it's broken on hcc & hip-clang
|
||||
@@ -149,7 +148,7 @@ typedef struct hipLaunchKernelStruct8A {
|
||||
char c1;
|
||||
int16_t si;
|
||||
bool b;
|
||||
}__attribute__((packed)) hipLaunchKernelStruct_t8A;
|
||||
} __attribute__((packed)) hipLaunchKernelStruct_t8A;
|
||||
|
||||
// This test is to verify struct with alignment, no packing as Sam suggested
|
||||
// size should be 8Bytes as no packing, right now it's broken on hcc & hip-clang
|
||||
@@ -157,7 +156,7 @@ typedef struct hipLaunchKernelStruct8B {
|
||||
char c1;
|
||||
int16_t si;
|
||||
bool b;
|
||||
}__attribute__((aligned(8))) hipLaunchKernelStruct_t8B;
|
||||
} __attribute__((aligned(8))) hipLaunchKernelStruct_t8B;
|
||||
|
||||
// This test is to verify const struct object
|
||||
typedef struct hipLaunchKernelStruct9 {
|
||||
@@ -181,8 +180,8 @@ typedef struct hipLaunchKernelStruct11 {
|
||||
// This test is to verify struct with simple class object
|
||||
class base {
|
||||
public:
|
||||
int i = 0;
|
||||
base() {}
|
||||
int i = 0;
|
||||
base() {}
|
||||
};
|
||||
typedef struct hipLaunchKernelStruct12 {
|
||||
base b;
|
||||
@@ -210,14 +209,13 @@ typedef struct hipLaunchKernelStruct15 {
|
||||
} hipLaunchKernelStruct_t15;
|
||||
|
||||
// This test is to verify simple template struct
|
||||
template<typename T>
|
||||
struct hipLaunchKernelStruct_t16 {
|
||||
template <typename T> struct hipLaunchKernelStruct_t16 {
|
||||
T t1;
|
||||
};
|
||||
|
||||
// This test is to verify simple explicity template struct
|
||||
template<typename T> struct hipLaunchKernelStruct_t17 {};
|
||||
template<> // explicit template
|
||||
template <typename T> struct hipLaunchKernelStruct_t17 {};
|
||||
template <> // explicit template
|
||||
struct hipLaunchKernelStruct_t17<int> {
|
||||
int t1;
|
||||
};
|
||||
@@ -246,301 +244,257 @@ typedef struct hipLaunchKernelStruct21 {
|
||||
|
||||
// Passing struct to a hipLaunchKernelGGL(),
|
||||
// read and write into the same struct
|
||||
__global__ void hipLaunchKernelStructFunc1(
|
||||
hipLaunchKernelStruct_t1 hipLaunchKernelStruct_,
|
||||
bool* result_d1) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
__global__ void hipLaunchKernelStructFunc1(hipLaunchKernelStruct_t1 hipLaunchKernelStruct_,
|
||||
bool* result_d1) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
// set the result to true if the condition met
|
||||
result_d1[x] = ((hipLaunchKernelStruct_.li == 1)
|
||||
&& (hipLaunchKernelStruct_.lf == 1.0)
|
||||
&& (hipLaunchKernelStruct_.result == false));
|
||||
// set the result to true if the condition met
|
||||
result_d1[x] = ((hipLaunchKernelStruct_.li == 1) && (hipLaunchKernelStruct_.lf == 1.0) &&
|
||||
(hipLaunchKernelStruct_.result == false));
|
||||
}
|
||||
|
||||
// Passing struct to a hipLaunchKernelGGL(), checks padding,
|
||||
// read and write into the same struct
|
||||
__global__ void hipLaunchKernelStructFunc2(
|
||||
hipLaunchKernelStruct_t2 hipLaunchKernelStruct_,
|
||||
bool* result_d2) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
__global__ void hipLaunchKernelStructFunc2(hipLaunchKernelStruct_t2 hipLaunchKernelStruct_,
|
||||
bool* result_d2) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
// set the result to true if the condition met
|
||||
result_d2[x] = ((hipLaunchKernelStruct_.c1 == 'a')
|
||||
&& (hipLaunchKernelStruct_.l1 == 1.0)
|
||||
&& (hipLaunchKernelStruct_.c2 == 'b')
|
||||
&& (hipLaunchKernelStruct_.l2 == 2.0) );
|
||||
// set the result to true if the condition met
|
||||
result_d2[x] = ((hipLaunchKernelStruct_.c1 == 'a') && (hipLaunchKernelStruct_.l1 == 1.0) &&
|
||||
(hipLaunchKernelStruct_.c2 == 'b') && (hipLaunchKernelStruct_.l2 == 2.0));
|
||||
}
|
||||
|
||||
// Passing struct to a hipLaunchKernelGGL(), checks padding,
|
||||
// read and write into the same struct
|
||||
__global__ void hipLaunchKernelStructFunc3(
|
||||
hipLaunchKernelStruct_t3 hipLaunchKernelStruct_,
|
||||
bool* result_d3) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
__global__ void hipLaunchKernelStructFunc3(hipLaunchKernelStruct_t3 hipLaunchKernelStruct_,
|
||||
bool* result_d3) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
// set the result to true if the condition met
|
||||
result_d3[x] = ((hipLaunchKernelStruct_.bf1 == 1)
|
||||
&& (hipLaunchKernelStruct_.bf2 == 1)
|
||||
&& (hipLaunchKernelStruct_.l1 == 1.0)
|
||||
&& (hipLaunchKernelStruct_.bf3 == 1) );
|
||||
// set the result to true if the condition met
|
||||
result_d3[x] = ((hipLaunchKernelStruct_.bf1 == 1) && (hipLaunchKernelStruct_.bf2 == 1) &&
|
||||
(hipLaunchKernelStruct_.l1 == 1.0) && (hipLaunchKernelStruct_.bf3 == 1));
|
||||
}
|
||||
|
||||
// Passing empty struct to a hipLaunchKernelGGL(),
|
||||
// check the size of 1Byte, set result_d4 to true if condition met
|
||||
__global__ void hipLaunchKernelStructFunc4(
|
||||
hipLaunchKernelStruct_t4 hipLaunchKernelStruct_,
|
||||
bool* result_d4) {
|
||||
__global__ void hipLaunchKernelStructFunc4(hipLaunchKernelStruct_t4 hipLaunchKernelStruct_,
|
||||
bool* result_d4) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
// set the result to true if the condition met
|
||||
result_d4[x] = (sizeof(hipLaunchKernelStruct_) == 1);
|
||||
result_d4[x] = (sizeof(hipLaunchKernelStruct_) == 1);
|
||||
}
|
||||
|
||||
// Passing struct with pointer object to a hipLaunchKernelGGL()
|
||||
__global__ void hipLaunchKernelStructFunc5(
|
||||
hipLaunchKernelStruct_t5 hipLaunchKernelStruct_,
|
||||
bool* result_d5) {
|
||||
__global__ void hipLaunchKernelStructFunc5(hipLaunchKernelStruct_t5 hipLaunchKernelStruct_,
|
||||
bool* result_d5) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
// set the result to true if the condition met
|
||||
result_d5[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (*hipLaunchKernelStruct_.cp == 'p'));
|
||||
result_d5[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (*hipLaunchKernelStruct_.cp == 'p'));
|
||||
}
|
||||
|
||||
// Passing struct which is aligned to 8Byte to a hipLaunchKernelGGL(),
|
||||
// set the result_d6 to true if condition met
|
||||
__global__ void hipLaunchKernelStructFunc6(
|
||||
hipLaunchKernelStruct_t6 hipLaunchKernelStruct_,
|
||||
bool* result_d6) {
|
||||
__global__ void hipLaunchKernelStructFunc6(hipLaunchKernelStruct_t6 hipLaunchKernelStruct_,
|
||||
bool* result_d6) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
// set the result to true if the condition met
|
||||
// get the address of the struct
|
||||
// size_t(p)%8 will be 0 if aligned to 8Byte address space
|
||||
int *p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
|
||||
result_d6[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (hipLaunchKernelStruct_.si == 1)
|
||||
&& ((size_t(p))%8 ==0));
|
||||
int* p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
|
||||
result_d6[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
|
||||
((size_t(p)) % 8 == 0));
|
||||
}
|
||||
|
||||
// Passing struct which is aligned to 16Byte,
|
||||
// set the result_d7 to true if condition met
|
||||
__global__ void hipLaunchKernelStructFunc7(
|
||||
hipLaunchKernelStruct_t7 hipLaunchKernelStruct_,
|
||||
bool* result_d7) {
|
||||
__global__ void hipLaunchKernelStructFunc7(hipLaunchKernelStruct_t7 hipLaunchKernelStruct_,
|
||||
bool* result_d7) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
// set the result to true if the condition met
|
||||
// get the address of the struct
|
||||
// size_t(p)%16 will be 0 if aligned to 16Byte address space
|
||||
int *p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
|
||||
result_d7[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (hipLaunchKernelStruct_.si == 1)
|
||||
&& ((size_t(p))%16 ==0) );
|
||||
int* p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
|
||||
result_d7[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
|
||||
((size_t(p)) % 16 == 0));
|
||||
}
|
||||
|
||||
// Passing struct which is packed & aligned to 4Byte,
|
||||
// set the result_d8 to true if condition met
|
||||
__global__ void hipLaunchKernelStructFunc8(
|
||||
hipLaunchKernelStruct_t8 hipLaunchKernelStruct_,
|
||||
bool* result_d8) {
|
||||
__global__ void hipLaunchKernelStructFunc8(hipLaunchKernelStruct_t8 hipLaunchKernelStruct_,
|
||||
bool* result_d8) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
// get the address of the xth element, struct[x],
|
||||
// size_t(p)%4 will be 0 if aligned to 4Byte address space
|
||||
int *p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
|
||||
result_d8[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (hipLaunchKernelStruct_.si == 1)
|
||||
&& ((size_t(p))%4 ==0)
|
||||
&& (sizeof(hipLaunchKernelStruct_) == 4));
|
||||
int* p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
|
||||
result_d8[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
|
||||
((size_t(p)) % 4 == 0) && (sizeof(hipLaunchKernelStruct_) == 4));
|
||||
}
|
||||
|
||||
// Passing struct which is packed only, as Sam suggested, should be 4Bytes
|
||||
// set the result_d8A to true if condition met
|
||||
__global__ void hipLaunchKernelStructFunc8A(
|
||||
hipLaunchKernelStruct_t8A hipLaunchKernelStruct_,
|
||||
bool* result_d8A) {
|
||||
__global__ void hipLaunchKernelStructFunc8A(hipLaunchKernelStruct_t8A hipLaunchKernelStruct_,
|
||||
bool* result_d8A) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
// this is packed struct
|
||||
// the address will not be aglined in this case hence condition removed
|
||||
// only sizeof(hipLaunchKernelStruct_) will be valided
|
||||
result_d8A[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (hipLaunchKernelStruct_.si == 1)
|
||||
&& (sizeof(hipLaunchKernelStruct_) == 4));
|
||||
result_d8A[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
|
||||
(sizeof(hipLaunchKernelStruct_) == 4));
|
||||
}
|
||||
|
||||
// Passing struct which is aligned(4) only, as Sam suggested
|
||||
// , size should be 8Bytes, set the result_d8B to true if condition met
|
||||
__global__ void hipLaunchKernelStructFunc8B(
|
||||
hipLaunchKernelStruct_t8B hipLaunchKernelStruct_,
|
||||
bool* result_d8B) {
|
||||
__global__ void hipLaunchKernelStructFunc8B(hipLaunchKernelStruct_t8B hipLaunchKernelStruct_,
|
||||
bool* result_d8B) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
// get the address of the xth element, struct[x],
|
||||
// size_t(p)%4 will be 0 if aligned to 4Byte address space
|
||||
int *p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
|
||||
result_d8B[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (hipLaunchKernelStruct_.si == 1)
|
||||
&& ((size_t(p))%8 == 0)
|
||||
&& (sizeof(hipLaunchKernelStruct_) == 8));
|
||||
int* p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
|
||||
result_d8B[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
|
||||
((size_t(p)) % 8 == 0) && (sizeof(hipLaunchKernelStruct_) == 8));
|
||||
}
|
||||
|
||||
// Passing struct with uint pointer object to a hipLaunchKernelGGL()
|
||||
__global__ void hipLaunchKernelStructFunc9(
|
||||
const hipLaunchKernelStruct_t9 hipLaunchKernelStruct_,
|
||||
bool* result_d9) {
|
||||
__global__ void hipLaunchKernelStructFunc9(const hipLaunchKernelStruct_t9 hipLaunchKernelStruct_,
|
||||
bool* result_d9) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
// set the result to true if the condition met
|
||||
result_d9[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (*hipLaunchKernelStruct_.ip == 1));
|
||||
result_d9[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (*hipLaunchKernelStruct_.ip == 1));
|
||||
}
|
||||
|
||||
// Passing struct with stdint types object, uintN_t, to a hipLaunchKernelGGL()
|
||||
__global__ void hipLaunchKernelStructFunc10(
|
||||
hipLaunchKernelStruct_t10 hipLaunchKernelStruct_,
|
||||
bool* result_d10) {
|
||||
__global__ void hipLaunchKernelStructFunc10(hipLaunchKernelStruct_t10 hipLaunchKernelStruct_,
|
||||
bool* result_d10) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
result_d10[x] = ((hipLaunchKernelStruct_.u64 == UINT64_MAX)
|
||||
&& (hipLaunchKernelStruct_.u32 == 1)
|
||||
&& (hipLaunchKernelStruct_.u8 == UINT8_MAX));
|
||||
result_d10[x] = ((hipLaunchKernelStruct_.u64 == UINT64_MAX) &&
|
||||
(hipLaunchKernelStruct_.u32 == 1) && (hipLaunchKernelStruct_.u8 == UINT8_MAX));
|
||||
}
|
||||
|
||||
// Passing struct with volatile member, to a hipLaunchKernelGGL()
|
||||
__global__ void hipLaunchKernelStructFunc11(
|
||||
hipLaunchKernelStruct_t11 hipLaunchKernelStruct_,
|
||||
bool* result_d11) {
|
||||
__global__ void hipLaunchKernelStructFunc11(hipLaunchKernelStruct_t11 hipLaunchKernelStruct_,
|
||||
bool* result_d11) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
result_d11[x] = ((hipLaunchKernelStruct_.i1 == 1)
|
||||
&& (hipLaunchKernelStruct_.vint == 0));
|
||||
result_d11[x] = ((hipLaunchKernelStruct_.i1 == 1) && (hipLaunchKernelStruct_.vint == 0));
|
||||
}
|
||||
|
||||
// Passing struct with simple class obj, to a hipLaunchKernelGGL()
|
||||
__global__ void hipLaunchKernelStructFunc12(
|
||||
hipLaunchKernelStruct_t12 hipLaunchKernelStruct_,
|
||||
bool* result_d12) {
|
||||
__global__ void hipLaunchKernelStructFunc12(hipLaunchKernelStruct_t12 hipLaunchKernelStruct_,
|
||||
bool* result_d12) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
result_d12[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (hipLaunchKernelStruct_.b.i == 0));
|
||||
result_d12[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.b.i == 0));
|
||||
}
|
||||
|
||||
// Passing struct with simple __device__ func(), to a hipLaunchKernelGGL()
|
||||
__global__ void hipLaunchKernelStructFunc13(
|
||||
hipLaunchKernelStruct_t13 hipLaunchKernelStruct_,
|
||||
bool* result_d13) {
|
||||
__global__ void hipLaunchKernelStructFunc13(hipLaunchKernelStruct_t13 hipLaunchKernelStruct_,
|
||||
bool* result_d13) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
result_d13[x] = ((hipLaunchKernelStruct_.i1 == 1)
|
||||
&& (hipLaunchKernelStruct_.getvalue() == 1));
|
||||
result_d13[x] = ((hipLaunchKernelStruct_.i1 == 1) && (hipLaunchKernelStruct_.getvalue() == 1));
|
||||
}
|
||||
|
||||
// Passing struct with array variable, write to from device
|
||||
__global__ void hipLaunchKernelStructFunc14(
|
||||
hipLaunchKernelStruct_t14 hipLaunchKernelStruct_,
|
||||
bool* result_d14) {
|
||||
__global__ void hipLaunchKernelStructFunc14(hipLaunchKernelStruct_t14 hipLaunchKernelStruct_,
|
||||
bool* result_d14) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
hipLaunchKernelStruct_.writeint[x] = 1;
|
||||
// set the result to true if the condition met
|
||||
result_d14[x] = ((hipLaunchKernelStruct_.readint == 1)
|
||||
&& (hipLaunchKernelStruct_.writeint[x] == 1));
|
||||
result_d14[x] =
|
||||
((hipLaunchKernelStruct_.readint == 1) && (hipLaunchKernelStruct_.writeint[x] == 1));
|
||||
}
|
||||
|
||||
// Passing struct with struct with dynamic memory, new int
|
||||
// the heap memory will be accessed from device
|
||||
__global__ void hipLaunchKernelStructFunc15(
|
||||
hipLaunchKernelStruct_t15 hipLaunchKernelStruct_,
|
||||
bool* result_d15) {
|
||||
__global__ void hipLaunchKernelStructFunc15(hipLaunchKernelStruct_t15 hipLaunchKernelStruct_,
|
||||
bool* result_d15) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
result_d15[x] = ((hipLaunchKernelStruct_.c1 == 'c')
|
||||
&& (hipLaunchKernelStruct_.heapmem[x] == 1));
|
||||
result_d15[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.heapmem[x] == 1));
|
||||
}
|
||||
|
||||
// Passing simple template struct
|
||||
__global__ void hipLaunchKernelStructFunc16(
|
||||
hipLaunchKernelStruct_t16<char> hipLaunchKernelStruct_,
|
||||
bool* result_d16) {
|
||||
__global__ void hipLaunchKernelStructFunc16(hipLaunchKernelStruct_t16<char> hipLaunchKernelStruct_,
|
||||
bool* result_d16) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
result_d16[x] = (hipLaunchKernelStruct_.t1 == 'c');
|
||||
result_d16[x] = (hipLaunchKernelStruct_.t1 == 'c');
|
||||
}
|
||||
|
||||
// Passing simple explicit template struct
|
||||
__global__ void hipLaunchKernelStructFunc17(
|
||||
hipLaunchKernelStruct_t17<int> hipLaunchKernelStruct_,
|
||||
bool* result_d17) {
|
||||
__global__ void hipLaunchKernelStructFunc17(hipLaunchKernelStruct_t17<int> hipLaunchKernelStruct_,
|
||||
bool* result_d17) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// set the result to true if the condition met
|
||||
result_d17[x] = (hipLaunchKernelStruct_.t1 == 1);
|
||||
result_d17[x] = (hipLaunchKernelStruct_.t1 == 1);
|
||||
}
|
||||
|
||||
// Passing struct and write to struct memory using __device__ func()
|
||||
__global__ void hipLaunchKernelStructFunc18(
|
||||
hipLaunchKernelStruct_t18 hipLaunchKernelStruct_,
|
||||
bool* result_d18) {
|
||||
__global__ void hipLaunchKernelStructFunc18(hipLaunchKernelStruct_t18 hipLaunchKernelStruct_,
|
||||
bool* result_d18) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
hipLaunchKernelStruct_.setChar('c');
|
||||
// set the result to true if the condition met
|
||||
result_d18[x] = (hipLaunchKernelStruct_.getChar() == 'c');
|
||||
result_d18[x] = (hipLaunchKernelStruct_.getChar() == 'c');
|
||||
}
|
||||
|
||||
// Passing out of order initalized struct, access in-order
|
||||
__global__ void hipLaunchKernelStructFunc20(
|
||||
hipLaunchKernelStruct_t20 hipLaunchKernelStruct_,
|
||||
bool* result_d20) {
|
||||
__global__ void hipLaunchKernelStructFunc20(hipLaunchKernelStruct_t20 hipLaunchKernelStruct_,
|
||||
bool* result_d20) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// accessing struct members in order
|
||||
result_d20[x] = (hipLaunchKernelStruct_.name == 'A'
|
||||
// strcmp(hipLaunchKernelStruct_.name, "AMD") -> strcmp is not broken
|
||||
&& hipLaunchKernelStruct_.age == 42
|
||||
&& hipLaunchKernelStruct_.rank == 2);
|
||||
// strcmp(hipLaunchKernelStruct_.name, "AMD") -> strcmp is not broken
|
||||
&& hipLaunchKernelStruct_.age == 42 && hipLaunchKernelStruct_.rank == 2);
|
||||
}
|
||||
|
||||
// Passing struct with bit fields
|
||||
__global__ void hipLaunchKernelStructFunc21(
|
||||
hipLaunchKernelStruct_t21 hipLaunchKernelStruct_,
|
||||
bool* result_d21) {
|
||||
__global__ void hipLaunchKernelStructFunc21(hipLaunchKernelStruct_t21 hipLaunchKernelStruct_,
|
||||
bool* result_d21) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
// accessing struct members in order
|
||||
result_d21[x] = (hipLaunchKernelStruct_.i == 2
|
||||
&& hipLaunchKernelStruct_.j == 0
|
||||
&& (sizeof(hipLaunchKernelStruct_) == 1));
|
||||
result_d21[x] = (hipLaunchKernelStruct_.i == 2 && hipLaunchKernelStruct_.j == 0 &&
|
||||
(sizeof(hipLaunchKernelStruct_) == 1));
|
||||
}
|
||||
|
||||
__global__ void vAdd(float* a) {}
|
||||
|
||||
template<class T1, class T2>
|
||||
__global__ void myKernel(T1 a, T2 b) {}
|
||||
template <class T1, class T2> __global__ void myKernel(T1 a, T2 b) {}
|
||||
|
||||
|
||||
//---
|
||||
// Some wrapper macro for testing:
|
||||
#define WRAP(...) __VA_ARGS__
|
||||
|
||||
#define MY_LAUNCH_MACRO(cmd, elapsed, quiet) \
|
||||
do { \
|
||||
HIP_CHECK(hipDeviceSynchronize()); \
|
||||
cmd; \
|
||||
HIP_CHECK(hipDeviceSynchronize()); \
|
||||
} while (0);
|
||||
#define MY_LAUNCH_MACRO(cmd, elapsed, quiet) \
|
||||
do { \
|
||||
HIP_CHECK(hipDeviceSynchronize()); \
|
||||
cmd; \
|
||||
HIP_CHECK(hipDeviceSynchronize()); \
|
||||
} while (0);
|
||||
|
||||
|
||||
#define MY_LAUNCH(command, doTrace, msg) \
|
||||
{ \
|
||||
if (doTrace) printf("TRACE: %s %s\n", msg, #command); \
|
||||
command; \
|
||||
}
|
||||
#define MY_LAUNCH(command, doTrace, msg) \
|
||||
{ \
|
||||
if (doTrace) printf("TRACE: %s %s\n", msg, #command); \
|
||||
command; \
|
||||
}
|
||||
|
||||
|
||||
#define MY_LAUNCH_WITH_PAREN(command, doTrace, msg) \
|
||||
{ \
|
||||
if (doTrace) printf("TRACE: %s %s\n", msg, #command); \
|
||||
(command); \
|
||||
}
|
||||
#define MY_LAUNCH_WITH_PAREN(command, doTrace, msg) \
|
||||
{ \
|
||||
if (doTrace) printf("TRACE: %s %s\n", msg, #command); \
|
||||
(command); \
|
||||
}
|
||||
|
||||
/**
|
||||
* @addtogroup hipLaunchKernelGGL hipLaunchKernelGGL
|
||||
@@ -589,10 +543,9 @@ __global__ void myKernel(T1 a, T2 b) {}
|
||||
*/
|
||||
|
||||
TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipMallocError = hipMalloc(reinterpret_cast<void**>(&result_d),
|
||||
BLOCK_DIM_SIZE*sizeof(bool));
|
||||
hipHostMallocError = hipHostMalloc(reinterpret_cast<void**>(&result_h),
|
||||
BLOCK_DIM_SIZE*sizeof(bool));
|
||||
hipMallocError = hipMalloc(reinterpret_cast<void**>(&result_d), BLOCK_DIM_SIZE * sizeof(bool));
|
||||
hipHostMallocError =
|
||||
hipHostMalloc(reinterpret_cast<void**>(&result_h), BLOCK_DIM_SIZE * sizeof(bool));
|
||||
hipMemsetError = hipMemset(result_d, false, BLOCK_DIM_SIZE);
|
||||
|
||||
// Validating memory & initial value, for result_d, result_h
|
||||
@@ -606,10 +559,8 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_h1.li = 1;
|
||||
hipLaunchKernelStruct_h1.lf = 1.0;
|
||||
hipLaunchKernelStruct_h1.result = false;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc1),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h1,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc1), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h1, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -621,10 +572,8 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_h2.c2 = 'b';
|
||||
hipLaunchKernelStruct_h2.l2 = 2.0;
|
||||
hipLaunchKernelStruct_h2.result = false;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc2),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h2,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc2), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h2, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -636,22 +585,18 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_h3.l1 = 1.0;
|
||||
hipLaunchKernelStruct_h3.bf3 = 1;
|
||||
hipLaunchKernelStruct_h3.result = false;
|
||||
// initialize to false, will be set to
|
||||
// true if the struct size is 1Byte, from device size
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc3),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h3,
|
||||
result_d);
|
||||
// initialize to false, will be set to
|
||||
// true if the struct size is 1Byte, from device size
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc3), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h3, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
SECTION("Empty struct") {
|
||||
ResetValidationMem();
|
||||
hipLaunchKernelStruct_t4 hipLaunchKernelStruct_h4;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc4),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h4,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc4), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h4, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -664,10 +609,8 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
HIP_CHECK(hipMemset(cp_d5, 'p', sizeof(char)));
|
||||
hipLaunchKernelStruct_h5.c1 = 'c';
|
||||
hipLaunchKernelStruct_h5.cp = cp_d5;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc5),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h5,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc5), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h5, result_d);
|
||||
ResultValidation();
|
||||
HIP_CHECK(hipFree(reinterpret_cast<void*>(cp_d5)));
|
||||
}
|
||||
@@ -677,14 +620,12 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_t6 hipLaunchKernelStruct_h6;
|
||||
hipLaunchKernelStruct_h6.c1 = 'c';
|
||||
hipLaunchKernelStruct_h6.si = 1;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc6),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h6,
|
||||
result_d);
|
||||
// alignment is broken hence disabled the validation part
|
||||
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc6), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h6, result_d);
|
||||
// alignment is broken hence disabled the validation part
|
||||
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("Passing struct with aligned(16)") {
|
||||
@@ -692,13 +633,11 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_t7 hipLaunchKernelStruct_h7;
|
||||
hipLaunchKernelStruct_h7.c1 = 'c';
|
||||
hipLaunchKernelStruct_h7.si = 1;
|
||||
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR // This is broken on small bar
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc7),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h7,
|
||||
result_d);
|
||||
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR // This is broken on small bar
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc7), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h7, result_d);
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("Passing struct with packed aligned to 4bytes") {
|
||||
@@ -706,14 +645,12 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_t8 hipLaunchKernelStruct_h8;
|
||||
hipLaunchKernelStruct_h8.c1 = 'c';
|
||||
hipLaunchKernelStruct_h8.si = 1;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h8,
|
||||
result_d);
|
||||
// packed member broken on large and small bar setup.
|
||||
#if ENABLE_PACKED_TEST
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h8, result_d);
|
||||
// packed member broken on large and small bar setup.
|
||||
#if ENABLE_PACKED_TEST
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("Passing struct with packed to 4Bytes") {
|
||||
@@ -721,14 +658,12 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_t8A hipLaunchKernelStruct_h8A;
|
||||
hipLaunchKernelStruct_h8A.c1 = 'c';
|
||||
hipLaunchKernelStruct_h8A.si = 1;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8A),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h8A,
|
||||
result_d);
|
||||
// packed member broken on large and small bar setup.
|
||||
#if ENABLE_PACKED_TEST
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8A), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h8A, result_d);
|
||||
// packed member broken on large and small bar setup.
|
||||
#if ENABLE_PACKED_TEST
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("Passing struct with aligned(4) to 4Bytes") {
|
||||
@@ -736,14 +671,12 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_t8B hipLaunchKernelStruct_h8B;
|
||||
hipLaunchKernelStruct_h8B.c1 = 'c';
|
||||
hipLaunchKernelStruct_h8B.si = 1;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8B),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h8B,
|
||||
result_d);
|
||||
// alignment is broken hence disabled the validation part
|
||||
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8B), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h8B, result_d);
|
||||
// alignment is broken hence disabled the validation part
|
||||
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("Passing const struct object") {
|
||||
@@ -754,13 +687,11 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
HIP_CHECK(hipMemset(ip_d9, 1, sizeof(uint32_t)));
|
||||
// ip_d9 passed as pointer to struct member, struct.ip = &ip_d9
|
||||
const hipLaunchKernelStruct_t9 hipLaunchKernelStruct_h9 = {'c', ip_d9};
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc9),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h9,
|
||||
result_d);
|
||||
#if ENABLE_DECLARE_INITIALIZATION_POINTER
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc9), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h9, result_d);
|
||||
#if ENABLE_DECLARE_INITIALIZATION_POINTER
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
HIP_CHECK(hipFree(reinterpret_cast<void*>(ip_d9)));
|
||||
}
|
||||
|
||||
@@ -770,10 +701,8 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_h10.u64 = UINT64_MAX;
|
||||
hipLaunchKernelStruct_h10.u32 = 1;
|
||||
hipLaunchKernelStruct_h10.u8 = UINT8_MAX;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc10),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h10,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc10), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h10, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -782,10 +711,8 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_t11 hipLaunchKernelStruct_h11;
|
||||
hipLaunchKernelStruct_h11.i1 = 1;
|
||||
hipLaunchKernelStruct_h11.vint = 0;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc11),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h11,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc11), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h11, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -793,24 +720,20 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
ResetValidationMem();
|
||||
hipLaunchKernelStruct_t12 hipLaunchKernelStruct_h12;
|
||||
hipLaunchKernelStruct_h12.c1 = 'c';
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc12),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h12,
|
||||
result_d);
|
||||
#if ENABLE_CLASS_OBJ_ACCESS // access class obj from device broken
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc12), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h12, result_d);
|
||||
#if ENABLE_CLASS_OBJ_ACCESS // access class obj from device broken
|
||||
// Validation part of the struct, hipLaunchKernelStructFunc12
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("Passing struct with simple __device__ func()") {
|
||||
ResetValidationMem();
|
||||
hipLaunchKernelStruct_t13 hipLaunchKernelStruct_h13;
|
||||
hipLaunchKernelStruct_h13.i1 = 1;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc13),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h13,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc13), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h13, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -818,10 +741,8 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
ResetValidationMem();
|
||||
hipLaunchKernelStruct_t14 hipLaunchKernelStruct_h14;
|
||||
hipLaunchKernelStruct_h14.readint = 1;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc14),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h14,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc14), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h14, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -830,16 +751,12 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_t15 hipLaunchKernelStruct_h15;
|
||||
hipLaunchKernelStruct_h15.c1 = 'c';
|
||||
|
||||
#if ENABLE_HEAP_MEMORY_ACCESS // causing page fault here,
|
||||
// on small bar set
|
||||
HIP_CHECK(hipMalloc(&hipLaunchKernelStruct_h15.heapmem,
|
||||
BLOCK_DIM_SIZE*sizeof(int)));
|
||||
HIP_CHECK(hipMemset(&hipLaunchKernelStruct_h15.heapmem,
|
||||
0, BLOCK_DIM_SIZE));
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc15),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h15,
|
||||
result_d);
|
||||
#if ENABLE_HEAP_MEMORY_ACCESS // causing page fault here,
|
||||
// on small bar set
|
||||
HIP_CHECK(hipMalloc(&hipLaunchKernelStruct_h15.heapmem, BLOCK_DIM_SIZE * sizeof(int)));
|
||||
HIP_CHECK(hipMemset(&hipLaunchKernelStruct_h15.heapmem, 0, BLOCK_DIM_SIZE));
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc15), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h15, result_d);
|
||||
ResultValidation();
|
||||
HIP_CHECK(hipFree(reinterpret_cast<void*>(hipLaunchKernelStruct_h15.heapmem)));
|
||||
#endif
|
||||
@@ -849,10 +766,8 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
ResetValidationMem();
|
||||
hipLaunchKernelStruct_t16<char> hipLaunchKernelStruct_h16;
|
||||
hipLaunchKernelStruct_h16.t1 = 'c';
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc16),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h16,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc16), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h16, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -860,20 +775,16 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
ResetValidationMem();
|
||||
hipLaunchKernelStruct_t17<int> hipLaunchKernelStruct_h17;
|
||||
hipLaunchKernelStruct_h17.t1 = 1;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc17),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h17,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc17), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h17, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
SECTION("Passing struct with simple __device__ func()") {
|
||||
ResetValidationMem();
|
||||
hipLaunchKernelStruct_t18 hipLaunchKernelStruct_h18;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc18),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h18,
|
||||
result_d);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc18), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h18, result_d);
|
||||
ResultValidation();
|
||||
}
|
||||
|
||||
@@ -886,26 +797,24 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelStruct_h20.rank = 2;
|
||||
hipLaunchKernelStruct_h20.age = 42;
|
||||
bool *result_d20, *result_h20;
|
||||
#if ENABLE_OUT_OF_ORDER_INITIALIZATION
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc20),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h20, result_d);
|
||||
#if ENABLE_OUT_OF_ORDER_INITIALIZATION
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc20), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h20, result_d);
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("Passing struct with bit fields operation") {
|
||||
ResetValidationMem();
|
||||
hipLaunchKernelStruct_t21 hipLaunchKernelStruct_h21 =
|
||||
// out of order initalization
|
||||
{2, 0};
|
||||
// out of order initalization
|
||||
{2, 0};
|
||||
bool *result_d21, *result_h21;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc21),
|
||||
dim3(BLOCK_DIM_SIZE),
|
||||
dim3(1), 0, 0, hipLaunchKernelStruct_h21, result_d);
|
||||
#if ENABLE_BIT_FIELDS
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc21), dim3(BLOCK_DIM_SIZE), dim3(1),
|
||||
0, 0, hipLaunchKernelStruct_h21, result_d);
|
||||
#if ENABLE_BIT_FIELDS
|
||||
ResultValidation();
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("Passing the different hipLaunchParm options") {
|
||||
@@ -917,42 +826,36 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(vAdd), dim3(1024), dim3(1), 0, 0, Ad);
|
||||
|
||||
// Test: Passing macro to hipLaunchKernelGGL
|
||||
#define KERNEL_CONFIG dim3(1024), dim3(1), 0, 0
|
||||
#define KERNEL_CONFIG dim3(1024), dim3(1), 0, 0
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(vAdd), KERNEL_CONFIG, Ad);
|
||||
|
||||
// Test: Same thing with templates:
|
||||
int a;
|
||||
float b;
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(myKernel<int, float>),
|
||||
KERNEL_CONFIG, a, b);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(myKernel<int, float>), KERNEL_CONFIG, a, b);
|
||||
|
||||
#define TYPE_PARAM_CONFIG int, float
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(myKernel<TYPE_PARAM_CONFIG>),
|
||||
KERNEL_CONFIG, a, b);
|
||||
hipLaunchKernelGGL(HIP_KERNEL_NAME(myKernel<TYPE_PARAM_CONFIG>), KERNEL_CONFIG, a, b);
|
||||
|
||||
// Test: Passing hipLaunchKernelGGL inside another macro:
|
||||
float e0;
|
||||
MY_LAUNCH_MACRO(hipLaunchKernelGGL(vAdd, dim3(1024),
|
||||
dim3(1), 0, 0, Ad), e0, j);
|
||||
MY_LAUNCH_MACRO(WRAP(hipLaunchKernelGGL(vAdd, dim3(1024),
|
||||
dim3(1), 0, 0, Ad)), e0, j);
|
||||
MY_LAUNCH_MACRO(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad), e0, j);
|
||||
MY_LAUNCH_MACRO(WRAP(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad)), e0, j);
|
||||
|
||||
#ifdef EXTRA_PARENS_1
|
||||
// Don't wrap hipLaunchKernelGGL in extra set of parens:
|
||||
MY_LAUNCH_MACRO((hipLaunchKernelGGL(vAdd, dim3(1024),
|
||||
dim3(1), 0, 0, Ad)), e0, j);
|
||||
MY_LAUNCH_MACRO((hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad)), e0, j);
|
||||
#endif
|
||||
|
||||
MY_LAUNCH(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1),
|
||||
0, 0, Ad), true, "firstCall");
|
||||
MY_LAUNCH(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad), true, "firstCall");
|
||||
float* A;
|
||||
float e1;
|
||||
MY_LAUNCH_WITH_PAREN(static_cast<void>(hipMalloc(&A, 100)), true, "launch2");
|
||||
|
||||
#ifdef EXTRA_PARENS_2
|
||||
// MY_LAUNCH_WITH_PAREN wraps cmd in () which can cause issues.
|
||||
MY_LAUNCH_WITH_PAREN(hipLaunchKernelGGL(vAdd, dim3(1024),
|
||||
dim3(1), 0, 0, Ad), true, "firstCall");
|
||||
MY_LAUNCH_WITH_PAREN(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad), true,
|
||||
"firstCall");
|
||||
#endif
|
||||
HIP_CHECK(hipFree(reinterpret_cast<void*>(A)));
|
||||
HIP_CHECK(hipFree(reinterpret_cast<void*>(Ad)));
|
||||
@@ -962,6 +865,6 @@ TEST_CASE("Unit_hipLaunchParm") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -20,34 +20,34 @@ THE SOFTWARE.
|
||||
#include <hip_test_kernels.hh>
|
||||
#include <hip_test_checkers.hh>
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
|
||||
class HipFunctorTests {
|
||||
public:
|
||||
// Test that a class functor can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForSimpleClassFunctor(void);
|
||||
// Test that a templated class functor can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForClassTemplateFunctor(void);
|
||||
// Test that a class functor object ptr can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForClassObjPtrFunctor(void);
|
||||
// Test that a class object containing functor can be passed
|
||||
// to hiplaunchparam and can be used in kernel
|
||||
void TestForFunctorContainInClassObj(void);
|
||||
// Test that a stuct functor can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForSimpleStructFunctor(void);
|
||||
// Test that a stuct functor object ptr can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForStructObjPtrFunctor(void);
|
||||
// Test that a templated struct functor can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForStructTemplateFunctor(void);
|
||||
// Test that a struct object containing functor can be
|
||||
// passed to hiplaunchparam and can be used in kernel
|
||||
void TestForFunctorContainInStructObj(void);
|
||||
// Test that a class functor can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForSimpleClassFunctor(void);
|
||||
// Test that a templated class functor can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForClassTemplateFunctor(void);
|
||||
// Test that a class functor object ptr can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForClassObjPtrFunctor(void);
|
||||
// Test that a class object containing functor can be passed
|
||||
// to hiplaunchparam and can be used in kernel
|
||||
void TestForFunctorContainInClassObj(void);
|
||||
// Test that a stuct functor can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForSimpleStructFunctor(void);
|
||||
// Test that a stuct functor object ptr can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForStructObjPtrFunctor(void);
|
||||
// Test that a templated struct functor can be passed to hiplaunchparam
|
||||
// and can be used in kernel
|
||||
void TestForStructTemplateFunctor(void);
|
||||
// Test that a struct object containing functor can be
|
||||
// passed to hiplaunchparam and can be used in kernel
|
||||
void TestForFunctorContainInStructObj(void);
|
||||
};
|
||||
|
||||
static const int BLOCK_DIM_SIZE = 1024;
|
||||
@@ -56,15 +56,13 @@ static const int THREADS_PER_BLOCK = 1;
|
||||
// class functor tests
|
||||
|
||||
// Simple doubler Functor
|
||||
class DoublerFunctor{
|
||||
class DoublerFunctor {
|
||||
public:
|
||||
__device__ int operator()(int x) { return x * 2;}
|
||||
__device__ int operator()(int x) { return x * 2; }
|
||||
};
|
||||
|
||||
// simple doubler functor passed to kernel
|
||||
__global__ void DoublerFunctorKernel(
|
||||
DoublerFunctor doubler_,
|
||||
bool* deviceResult) {
|
||||
__global__ void DoublerFunctorKernel(DoublerFunctor doubler_, bool* deviceResult) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int result = doubler_(5);
|
||||
deviceResult[x] = (result == 10);
|
||||
@@ -73,32 +71,29 @@ __global__ void DoublerFunctorKernel(
|
||||
void HipFunctorTests::TestForSimpleClassFunctor(void) {
|
||||
DoublerFunctor doubler;
|
||||
bool *deviceResults, *hostResults;
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
// initialize to false, will be set to
|
||||
// true if the functor is called in device code
|
||||
hostResults[k] = false;
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(DoublerFunctorKernel, dim3(BLOCK_DIM_SIZE),
|
||||
dim3(THREADS_PER_BLOCK), 0, 0, doubler, deviceResults);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(DoublerFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0, 0,
|
||||
doubler, deviceResults);
|
||||
|
||||
// Validation part of TestForSimpleClassFunctor
|
||||
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
|
||||
REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(hipHostFree(hostResults));
|
||||
HIP_CHECK(hipFree(deviceResults));
|
||||
}
|
||||
|
||||
// pointer functor passed to kernel
|
||||
__global__ void PtrDoublerFunctorKernel(
|
||||
DoublerFunctor *doubler_,
|
||||
bool* deviceResult) {
|
||||
__global__ void PtrDoublerFunctorKernel(DoublerFunctor* doubler_, bool* deviceResult) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int result = (*doubler_)(5);
|
||||
deviceResult[x] = (result == 10);
|
||||
@@ -107,24 +102,23 @@ __global__ void PtrDoublerFunctorKernel(
|
||||
void HipFunctorTests::TestForClassObjPtrFunctor(void) {
|
||||
DoublerFunctor* ptrdoubler = new DoublerFunctor[sizeof(int)];
|
||||
bool *deviceResults, *hostResults;
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
// initialize to false, will be set to
|
||||
// true if the functor is called in device code
|
||||
hostResults[k] = false;
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(PtrDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE),
|
||||
dim3(THREADS_PER_BLOCK), 0, 0, ptrdoubler, deviceResults);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(PtrDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0, 0,
|
||||
ptrdoubler, deviceResults);
|
||||
|
||||
// Validation part of TestForClassObjPtrFunctor
|
||||
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
|
||||
REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(hipHostFree(hostResults));
|
||||
HIP_CHECK(hipFree(deviceResults));
|
||||
delete[] ptrdoubler;
|
||||
@@ -132,16 +126,13 @@ void HipFunctorTests::TestForClassObjPtrFunctor(void) {
|
||||
|
||||
class compare {
|
||||
public:
|
||||
template<typename T1, typename T2>
|
||||
__device__ bool operator()(const T1& v1, const T2& v2) {
|
||||
return v1 > v2;
|
||||
}
|
||||
template <typename T1, typename T2> __device__ bool operator()(const T1& v1, const T2& v2) {
|
||||
return v1 > v2;
|
||||
}
|
||||
};
|
||||
|
||||
// template functor passed to kernel
|
||||
__global__ void TemplateFunctorKernel(
|
||||
compare compare_,
|
||||
bool* deviceResult) {
|
||||
__global__ void TemplateFunctorKernel(compare compare_, bool* deviceResult) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
deviceResult[x] = compare_(2.2, 2.1);
|
||||
deviceResult[x] = compare_(2, 1);
|
||||
@@ -151,24 +142,23 @@ __global__ void TemplateFunctorKernel(
|
||||
void HipFunctorTests::TestForClassTemplateFunctor(void) {
|
||||
compare comparefunctor;
|
||||
bool *deviceResults, *hostResults;
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
// initialize to false, will be set to
|
||||
// true if the functor is called in device code
|
||||
hostResults[k] = false;
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(TemplateFunctorKernel, dim3(BLOCK_DIM_SIZE),
|
||||
dim3(THREADS_PER_BLOCK), 0, 0, comparefunctor, deviceResults);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(TemplateFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0, 0,
|
||||
comparefunctor, deviceResults);
|
||||
|
||||
// Validation part of TestForClassTemplateFunctor
|
||||
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
|
||||
REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(hipHostFree(hostResults));
|
||||
HIP_CHECK(hipFree(deviceResults));
|
||||
}
|
||||
@@ -177,15 +167,13 @@ void HipFunctorTests::TestForClassTemplateFunctor(void) {
|
||||
// Doubler calculator
|
||||
class DoublerCalculator {
|
||||
public:
|
||||
int a, result;
|
||||
// fucntor contained in class object
|
||||
DoublerFunctor doubler;
|
||||
int a, result;
|
||||
// fucntor contained in class object
|
||||
DoublerFunctor doubler;
|
||||
};
|
||||
|
||||
// doubler functor conatined in class obj passed to kernel
|
||||
__global__ void DoublerCalculatorFunctorKernel(
|
||||
DoublerCalculator doubler_,
|
||||
bool* deviceResult) {
|
||||
__global__ void DoublerCalculatorFunctorKernel(DoublerCalculator doubler_, bool* deviceResult) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int result = doubler_.doubler(doubler_.a);
|
||||
deviceResult[x] = (doubler_.result == result);
|
||||
@@ -194,8 +182,8 @@ __global__ void DoublerCalculatorFunctorKernel(
|
||||
void HipFunctorTests::TestForFunctorContainInClassObj(void) {
|
||||
DoublerCalculator Doubler;
|
||||
bool *deviceResults, *hostResults;
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
// initialize to false, will be set to
|
||||
// true if the functor is called in device code
|
||||
@@ -206,16 +194,15 @@ void HipFunctorTests::TestForFunctorContainInClassObj(void) {
|
||||
Doubler.result = 10;
|
||||
// pass comparefunctor to hipLaunchParm
|
||||
|
||||
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(DoublerCalculatorFunctorKernel, dim3(BLOCK_DIM_SIZE),
|
||||
dim3(THREADS_PER_BLOCK), 0, 0, Doubler, deviceResults);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(DoublerCalculatorFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK),
|
||||
0, 0, Doubler, deviceResults);
|
||||
|
||||
// Validation part of TestForStructTemplateFunctor
|
||||
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
|
||||
REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(hipHostFree(hostResults));
|
||||
HIP_CHECK(hipFree(deviceResults));
|
||||
}
|
||||
@@ -225,14 +212,12 @@ void HipFunctorTests::TestForFunctorContainInClassObj(void) {
|
||||
// Simple doubler Functor
|
||||
struct sDoublerFunctor {
|
||||
public:
|
||||
__device__ int operator()(int x) { return x * 2;}
|
||||
__device__ int operator()(int x) { return x * 2; }
|
||||
};
|
||||
|
||||
|
||||
// simple sturct doubler functor passed to kernel
|
||||
__global__ void structDoublerFunctorKernel(
|
||||
sDoublerFunctor doubler_,
|
||||
bool* deviceResult) {
|
||||
__global__ void structDoublerFunctorKernel(sDoublerFunctor doubler_, bool* deviceResult) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int result = doubler_(5);
|
||||
deviceResult[x] = (result == 10);
|
||||
@@ -241,32 +226,29 @@ __global__ void structDoublerFunctorKernel(
|
||||
void HipFunctorTests::TestForSimpleStructFunctor(void) {
|
||||
sDoublerFunctor doubler;
|
||||
bool *deviceResults, *hostResults;
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
// initialize to false, will be set to
|
||||
// true if the functor is called in device code
|
||||
hostResults[k] = false;
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(structDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE),
|
||||
dim3(THREADS_PER_BLOCK), 0, 0, doubler, deviceResults);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(structDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0,
|
||||
0, doubler, deviceResults);
|
||||
|
||||
// Validation part of TestForSimpleStructFunctor
|
||||
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
|
||||
REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(hipHostFree(hostResults));
|
||||
HIP_CHECK(hipFree(deviceResults));
|
||||
}
|
||||
|
||||
// ptr functor passed to kernel
|
||||
__global__ void structPtrDoublerFunctorKernel(
|
||||
sDoublerFunctor *doubler_,
|
||||
bool* deviceResult) {
|
||||
__global__ void structPtrDoublerFunctorKernel(sDoublerFunctor* doubler_, bool* deviceResult) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int result = (*doubler_)(5);
|
||||
deviceResult[x] = (result == 10);
|
||||
@@ -275,24 +257,23 @@ __global__ void structPtrDoublerFunctorKernel(
|
||||
void HipFunctorTests::TestForStructObjPtrFunctor(void) {
|
||||
sDoublerFunctor* ptrdoubler = new sDoublerFunctor[sizeof(int)];
|
||||
bool *deviceResults, *hostResults;
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
// initialize to false, will be set to
|
||||
// true if the functor is called in device code
|
||||
hostResults[k] = false;
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(structPtrDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE),
|
||||
dim3(THREADS_PER_BLOCK), 0, 0, ptrdoubler, deviceResults);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(structPtrDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK),
|
||||
0, 0, ptrdoubler, deviceResults);
|
||||
|
||||
// Validation part of TestForStructObjPtrFunctor
|
||||
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
|
||||
REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(hipHostFree(hostResults));
|
||||
HIP_CHECK(hipFree(deviceResults));
|
||||
delete[] ptrdoubler;
|
||||
@@ -300,16 +281,13 @@ void HipFunctorTests::TestForStructObjPtrFunctor(void) {
|
||||
|
||||
struct sCompare {
|
||||
public:
|
||||
template< typename T1, typename T2 >
|
||||
__device__ bool operator()(const T1& v1, const T2& v2) {
|
||||
template <typename T1, typename T2> __device__ bool operator()(const T1& v1, const T2& v2) {
|
||||
return v1 > v2;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// template functor passed to kernel
|
||||
__global__ void structTemplateFunctorKernel(
|
||||
sCompare compare_,
|
||||
bool* deviceResult) {
|
||||
__global__ void structTemplateFunctorKernel(sCompare compare_, bool* deviceResult) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
deviceResult[x] = compare_(2.2, 2.1);
|
||||
deviceResult[x] = compare_(2, 1);
|
||||
@@ -319,26 +297,25 @@ __global__ void structTemplateFunctorKernel(
|
||||
void HipFunctorTests::TestForStructTemplateFunctor(void) {
|
||||
sCompare comparefunctor;
|
||||
bool *deviceResults, *hostResults;
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
// initialize to false, will be set to
|
||||
// true if the functor is called in device code
|
||||
hostResults[k] = false;
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyHostToDevice));
|
||||
HIP_CHECK(
|
||||
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
|
||||
|
||||
// pass comparefunctor to hipLaunchKernelGGL
|
||||
hipLaunchKernelGGL(structTemplateFunctorKernel, dim3(BLOCK_DIM_SIZE),
|
||||
dim3(THREADS_PER_BLOCK), 0, 0, comparefunctor, deviceResults);
|
||||
hipLaunchKernelGGL(structTemplateFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0,
|
||||
0, comparefunctor, deviceResults);
|
||||
|
||||
// Validation part of TestForStructTemplateFunctor
|
||||
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
|
||||
REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(hipHostFree(hostResults));
|
||||
HIP_CHECK(hipFree(deviceResults));
|
||||
}
|
||||
@@ -346,17 +323,14 @@ void HipFunctorTests::TestForStructTemplateFunctor(void) {
|
||||
// Doubler calculator struct
|
||||
struct sDoublerCalculator {
|
||||
public:
|
||||
int a, result;
|
||||
// fucntor contained in class object
|
||||
DoublerFunctor doubler;
|
||||
int a, result;
|
||||
// fucntor contained in class object
|
||||
DoublerFunctor doubler;
|
||||
};
|
||||
|
||||
|
||||
|
||||
// doubler functor contained in struct passed to kernel
|
||||
__global__ void DoublerCalculatorFunctorKernel(
|
||||
sDoublerCalculator doubler_,
|
||||
bool* deviceResult) {
|
||||
__global__ void DoublerCalculatorFunctorKernel(sDoublerCalculator doubler_, bool* deviceResult) {
|
||||
int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int result = doubler_.doubler(doubler_.a);
|
||||
deviceResult[x] = (doubler_.result == result);
|
||||
@@ -365,8 +339,8 @@ __global__ void DoublerCalculatorFunctorKernel(
|
||||
void HipFunctorTests::TestForFunctorContainInStructObj(void) {
|
||||
sDoublerCalculator Doubler;
|
||||
bool *deviceResults, *hostResults;
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
|
||||
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
|
||||
// initialize to false, will be set to
|
||||
// true if the functor is called in device code
|
||||
@@ -375,19 +349,18 @@ void HipFunctorTests::TestForFunctorContainInStructObj(void) {
|
||||
|
||||
Doubler.a = 5;
|
||||
Doubler.result = 10;
|
||||
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyHostToDevice));
|
||||
HIP_CHECK(
|
||||
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
|
||||
|
||||
|
||||
// pass comparefunctor to hipLaunchKernelGGL
|
||||
hipLaunchKernelGGL(DoublerCalculatorFunctorKernel, dim3(BLOCK_DIM_SIZE),
|
||||
dim3(THREADS_PER_BLOCK), 0, 0, Doubler, deviceResults);
|
||||
hipLaunchKernelGGL(DoublerCalculatorFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK),
|
||||
0, 0, Doubler, deviceResults);
|
||||
|
||||
// Validation part of TestForStructTemplateFunctor
|
||||
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
|
||||
REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(
|
||||
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
|
||||
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
|
||||
HIP_CHECK(hipHostFree(hostResults));
|
||||
HIP_CHECK(hipFree(deviceResults));
|
||||
}
|
||||
@@ -432,24 +405,12 @@ void HipFunctorTests::TestForFunctorContainInStructObj(void) {
|
||||
TEST_CASE("Unit_hipLaunchParmFunctor") {
|
||||
HipFunctorTests FunctorTests;
|
||||
|
||||
SECTION("test for simple class functor") {
|
||||
FunctorTests.TestForSimpleClassFunctor();
|
||||
}
|
||||
SECTION("test for class objptr functor") {
|
||||
FunctorTests.TestForClassObjPtrFunctor();
|
||||
}
|
||||
SECTION("test for class templete functor") {
|
||||
FunctorTests.TestForClassTemplateFunctor();
|
||||
}
|
||||
SECTION("test for simple struct functor") {
|
||||
FunctorTests.TestForSimpleStructFunctor();
|
||||
}
|
||||
SECTION("test for struct objptr functor") {
|
||||
FunctorTests.TestForStructObjPtrFunctor();
|
||||
}
|
||||
SECTION("test for struct templete functor") {
|
||||
FunctorTests.TestForStructTemplateFunctor();
|
||||
}
|
||||
SECTION("test for simple class functor") { FunctorTests.TestForSimpleClassFunctor(); }
|
||||
SECTION("test for class objptr functor") { FunctorTests.TestForClassObjPtrFunctor(); }
|
||||
SECTION("test for class templete functor") { FunctorTests.TestForClassTemplateFunctor(); }
|
||||
SECTION("test for simple struct functor") { FunctorTests.TestForSimpleStructFunctor(); }
|
||||
SECTION("test for struct objptr functor") { FunctorTests.TestForStructObjPtrFunctor(); }
|
||||
SECTION("test for struct templete functor") { FunctorTests.TestForStructTemplateFunctor(); }
|
||||
SECTION("test for functor contain in classobj") {
|
||||
FunctorTests.TestForFunctorContainInClassObj();
|
||||
}
|
||||
@@ -459,6 +420,6 @@ TEST_CASE("Unit_hipLaunchParmFunctor") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -37,7 +37,7 @@ __global__ void MyKernelConstSize(int* C_d, const int* A_d) {
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < N; ++i) {
|
||||
C_d[i] = A_d[i] + A1[i%A1size];
|
||||
C_d[i] = A_d[i] + A1[i % A1size];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -50,13 +50,13 @@ __global__ void MyKernelVariableSize(int* C_d, const int* A_d) {
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < N; ++i) {
|
||||
C_d[i] = A_d[i] + A1[i%A1size];
|
||||
C_d[i] = A_d[i] + A1[i % A1size];
|
||||
}
|
||||
}
|
||||
|
||||
static bool verify(const int* C_d, const int* A_d) {
|
||||
for (size_t i = 0; i < N; i++) {
|
||||
if (C_d[i] != A_d[i] + i%1024) {
|
||||
if (C_d[i] != A_d[i] + i % 1024) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -68,7 +68,7 @@ TEST_CASE("Unit_hipMemFaultStackAllocation_Check") {
|
||||
int *A_d, *C_d;
|
||||
const size_t Nbytes = N * sizeof(int);
|
||||
const unsigned threadsPerBlock = 256;
|
||||
const unsigned blocks = (N + threadsPerBlock - 1)/threadsPerBlock;
|
||||
const unsigned blocks = (N + threadsPerBlock - 1) / threadsPerBlock;
|
||||
|
||||
HIP_CHECK(hipMallocManaged(&A_d, Nbytes));
|
||||
REQUIRE(A_d != nullptr);
|
||||
@@ -76,20 +76,18 @@ TEST_CASE("Unit_hipMemFaultStackAllocation_Check") {
|
||||
REQUIRE(C_d != nullptr);
|
||||
|
||||
for (size_t i = 0; i < N; i++) {
|
||||
A_d[i] = i%1024;
|
||||
A_d[i] = i % 1024;
|
||||
}
|
||||
|
||||
SECTION("Calling Kernel which allocate ConstSize to local array") {
|
||||
hipLaunchKernelGGL(MyKernelConstSize, dim3(blocks),
|
||||
dim3(threadsPerBlock), 0, 0, C_d, A_d);
|
||||
hipLaunchKernelGGL(MyKernelConstSize, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d);
|
||||
ret = hipGetLastError();
|
||||
REQUIRE(hipSuccess == ret);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
REQUIRE(true == verify(C_d, A_d));
|
||||
}
|
||||
SECTION("Calling Kernel which allocate VariableSize to local array") {
|
||||
hipLaunchKernelGGL(MyKernelVariableSize, dim3(blocks),
|
||||
dim3(threadsPerBlock), 0, 0, C_d, A_d);
|
||||
hipLaunchKernelGGL(MyKernelVariableSize, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d);
|
||||
ret = hipGetLastError();
|
||||
REQUIRE(hipSuccess == ret);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
@@ -99,4 +97,3 @@ TEST_CASE("Unit_hipMemFaultStackAllocation_Check") {
|
||||
HIP_CHECK(hipFree(C_d));
|
||||
HIP_CHECK(hipFree(A_d));
|
||||
}
|
||||
|
||||
|
||||
@@ -17,15 +17,13 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
#include <cstring>
|
||||
#include "../kernel/printf_common.h"
|
||||
|
||||
#define HIP_ENABLE_PRINTF
|
||||
|
||||
__global__ void run_printf() {
|
||||
printf("Hello World\n");
|
||||
}
|
||||
__global__ void run_printf() { printf("Hello World\n"); }
|
||||
/**
|
||||
* @addtogroup hipLaunchKernelGGL
|
||||
* @{
|
||||
@@ -52,21 +50,22 @@ TEST_CASE("Unit_kernel_ChkPrintf") {
|
||||
CaptureStream capture(stdout);
|
||||
HIP_CHECK(hipGetDeviceCount(&device_count));
|
||||
std::string st = "Hello World";
|
||||
const char * check = st.c_str();
|
||||
const char* check = st.c_str();
|
||||
for (int i = 0; i < device_count; ++i) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
hipLaunchKernelGGL(run_printf, dim3(1), dim3(1), 0, 0);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
char* data = new char[st.size()];;
|
||||
char* data = new char[st.size()];
|
||||
;
|
||||
std::ifstream CapturedData = capture.getCapturedData();
|
||||
CapturedData.getline(data, st.size()+1);
|
||||
CapturedData.getline(data, st.size() + 1);
|
||||
int result = strcmp(data, check);
|
||||
REQUIRE(result == 0);
|
||||
delete [] data;
|
||||
delete[] data;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -41,8 +41,8 @@ TEST_CASE("Unit_hipSetupArgument_Simple") {
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Verifies that arguments sent to the kernel with hipSetupArgument are correct by executing
|
||||
* kernel that calculates sum of two vectors, doing the same calculation on CPU and checking if the
|
||||
* results are the same, which proves that the arguments used in kernel are the proper ones
|
||||
* kernel that calculates sum of two vectors, doing the same calculation on CPU and checking if
|
||||
* the results are the same, which proves that the arguments used in kernel are the proper ones
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
|
||||
@@ -17,7 +17,7 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
|
||||
#define LEN 512
|
||||
#define SIZE 2048
|
||||
@@ -61,21 +61,20 @@ TEST_CASE("Unit_kernel_chkConstantViaKernel") {
|
||||
|
||||
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&Ad), SIZE));
|
||||
|
||||
HIP_CHECK(hipMemcpyToSymbol(HIP_SYMBOL(Value), A, SIZE, 0,
|
||||
hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpyToSymbol(HIP_SYMBOL(Value), A, SIZE, 0, hipMemcpyHostToDevice));
|
||||
hipLaunchKernelGGL(Get, dim3(1, 1, 1), dim3(LEN, 1, 1), 0, 0, Ad);
|
||||
HIP_CHECK(hipMemcpy(B, Ad, SIZE, hipMemcpyDeviceToHost));
|
||||
|
||||
for (unsigned i = 0; i < LEN; i++) {
|
||||
REQUIRE(A[i] == B[i]);
|
||||
}
|
||||
delete [] A;
|
||||
delete [] B;
|
||||
delete[] A;
|
||||
delete[] B;
|
||||
HIP_CHECK(hipFree(Ad));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -17,7 +17,7 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
|
||||
#define LEN 512
|
||||
#define SIZE 2048
|
||||
@@ -62,7 +62,7 @@ void runTestConstantGlobalVar() {
|
||||
for (unsigned i = 0; i < LEN; i++) {
|
||||
REQUIRE(123 == A[i]);
|
||||
}
|
||||
delete [] A;
|
||||
delete[] A;
|
||||
HIP_CHECK(hipFree(Ad));
|
||||
}
|
||||
|
||||
@@ -91,7 +91,7 @@ void runTestGlobalArray() {
|
||||
for (unsigned i = 0; i < LEN; i++) {
|
||||
REQUIRE(i == A[i]);
|
||||
}
|
||||
delete [] A;
|
||||
delete[] A;
|
||||
HIP_CHECK(hipFree(Ad));
|
||||
}
|
||||
|
||||
@@ -101,6 +101,6 @@ TEST_CASE("Unit_kernel_chkGlobalArrAndGlobalVaribleViaKernelFn") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -17,7 +17,7 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
|
||||
#define LEN8 8 * 4
|
||||
#define LEN9 9 * 4
|
||||
@@ -26,53 +26,53 @@ THE SOFTWARE.
|
||||
#define LEN12 12 * 4
|
||||
|
||||
__global__ void MemCpy8(uint8_t* In, uint8_t* Out) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 8, In + tid * 8, 8);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 8, In + tid * 8, 8);
|
||||
}
|
||||
|
||||
__global__ void MemCpy9(uint8_t* In, uint8_t* Out) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 9, In + tid * 9, 9);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 9, In + tid * 9, 9);
|
||||
}
|
||||
|
||||
__global__ void MemCpy10(uint8_t* In, uint8_t* Out) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 10, In + tid * 10, 10);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 10, In + tid * 10, 10);
|
||||
}
|
||||
|
||||
__global__ void MemCpy11(uint8_t* In, uint8_t* Out) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 11, In + tid * 11, 11);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 11, In + tid * 11, 11);
|
||||
}
|
||||
|
||||
__global__ void MemCpy12(uint8_t* In, uint8_t* Out) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 12, In + tid * 12, 12);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memcpy(Out + tid * 12, In + tid * 12, 12);
|
||||
}
|
||||
|
||||
__global__ void MemSet8(uint8_t* In) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 8, 1, 8);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 8, 1, 8);
|
||||
}
|
||||
|
||||
__global__ void MemSet9(uint8_t* In) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 9, 1, 9);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 9, 1, 9);
|
||||
}
|
||||
|
||||
__global__ void MemSet10(uint8_t* In) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 10, 1, 10);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 10, 1, 10);
|
||||
}
|
||||
|
||||
__global__ void MemSet11(uint8_t* In) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 11, 1, 11);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 11, 1, 11);
|
||||
}
|
||||
|
||||
__global__ void MemSet12(uint8_t* In) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 12, 1, 12);
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
memset(In + tid * 12, 1, 12);
|
||||
}
|
||||
/**
|
||||
* @addtogroup hipLaunchKernelGGL
|
||||
@@ -161,9 +161,9 @@ TEST_CASE("Unit_kernel_MemoryOperationsViaKernels") {
|
||||
B = new uint8_t[LEN10];
|
||||
C = new uint8_t[LEN10];
|
||||
for (uint32_t i = 0; i < LEN10; i++) {
|
||||
A[i] = i;
|
||||
B[i] = 0;
|
||||
C[i] = 0;
|
||||
A[i] = i;
|
||||
B[i] = 0;
|
||||
C[i] = 0;
|
||||
}
|
||||
HIP_CHECK(hipMalloc(&Ad, LEN10));
|
||||
HIP_CHECK(hipMalloc(&Bd, LEN10));
|
||||
@@ -248,6 +248,6 @@ TEST_CASE("Unit_kernel_MemoryOperationsViaKernels") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -17,19 +17,17 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
|
||||
|
||||
constexpr size_t N = 1024;
|
||||
int p_blockSize = 256;
|
||||
|
||||
__global__ void
|
||||
__launch_bounds__(256, 2)
|
||||
myKern(int* C, const int* A, int N) {
|
||||
int tid = (blockIdx.x * blockDim.x + threadIdx.x);
|
||||
__global__ void __launch_bounds__(256, 2) myKern(int* C, const int* A, int N) {
|
||||
int tid = (blockIdx.x * blockDim.x + threadIdx.x);
|
||||
|
||||
if (tid < N) {
|
||||
C[tid] = A[tid];
|
||||
}
|
||||
if (tid < N) {
|
||||
C[tid] = A[tid];
|
||||
}
|
||||
}
|
||||
/**
|
||||
* @addtogroup hipLaunchKernelGGL
|
||||
@@ -70,8 +68,7 @@ TEST_CASE("Unit_kernel_LaunchBounds_Functional") {
|
||||
|
||||
HIPCHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
|
||||
HIPCHECK(hipGetLastError());
|
||||
hipLaunchKernelGGL(myKern, dim3(blocks), dim3(p_blockSize), 0,
|
||||
0, C_d, A_d, N);
|
||||
hipLaunchKernelGGL(myKern, dim3(blocks), dim3(p_blockSize), 0, 0, C_d, A_d, N);
|
||||
|
||||
#ifdef __HIP_PLATFORM_NVIDIA__
|
||||
cudaFuncAttributes attrib;
|
||||
@@ -102,6 +99,6 @@ TEST_CASE("Unit_kernel_LaunchBounds_Functional") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group KernelTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -42,7 +42,7 @@ struct CaptureStream {
|
||||
|
||||
char tempname[13] = "mytestXXXXXX";
|
||||
|
||||
explicit CaptureStream(FILE *original) {
|
||||
explicit CaptureStream(FILE* original) {
|
||||
orig_fd = fileno(original);
|
||||
saved_fd = dup(orig_fd);
|
||||
|
||||
@@ -63,8 +63,7 @@ struct CaptureStream {
|
||||
}
|
||||
|
||||
void restoreStream() {
|
||||
if (saved_fd == -1)
|
||||
return;
|
||||
if (saved_fd == -1) return;
|
||||
fflush(nullptr);
|
||||
if (dup2(saved_fd, orig_fd) == -1) {
|
||||
error(0, errno, "Error");
|
||||
@@ -77,9 +76,7 @@ struct CaptureStream {
|
||||
saved_fd = -1;
|
||||
}
|
||||
|
||||
const char *getTempFilename() {
|
||||
return (const char*)tempname;
|
||||
}
|
||||
const char* getTempFilename() { return (const char*)tempname; }
|
||||
|
||||
std::ifstream getCapturedData() {
|
||||
restoreStream();
|
||||
|
||||
Yeni konuda referans
Bir kullanıcı engelle