SWDEV-470698 - fix formatting, add format check workflow (#657)

Bu işleme şunda yer alıyor:
Danylo Lytovchenko
2025-08-20 16:28:06 +02:00
işlemeyi yapan: GitHub
ebeveyn 5840940caa
işleme f7338717ae
1574 değiştirilmiş dosya ile 162972 ekleme ve 199346 silme
+3 -4
Dosyayı Görüntüle
@@ -51,7 +51,7 @@ static __global__ void vecSqrSingBlk(int* A_d, size_t NELEM) {
* - HIP_VERSION >= 6.1
*/
TEST_CASE("Unit_kernel_Assign_threadIdx_to_auto") {
int *A_d;
int* A_d;
const unsigned blocks = 256;
const unsigned threadsPerBlock = 128;
size_t N = (blocks * threadsPerBlock);
@@ -66,12 +66,11 @@ TEST_CASE("Unit_kernel_Assign_threadIdx_to_auto") {
// Transfer data and perform operations on GPU
HIP_CHECK(hipMalloc(&A_d, Nbytes));
HIP_CHECK(hipMemcpy(A_d, A_h.data(), Nbytes, hipMemcpyHostToDevice));
hipLaunchKernelGGL(vecSqrSingBlk, dim3(blocks), dim3(threadsPerBlock),
0, 0, A_d, N);
hipLaunchKernelGGL(vecSqrSingBlk, dim3(blocks), dim3(threadsPerBlock), 0, 0, A_d, N);
HIP_CHECK(hipMemcpy(C_h.data(), A_d, Nbytes, hipMemcpyDeviceToHost));
HIP_CHECK(hipDeviceSynchronize());
for (size_t i = 0; i < N; i++) {
REQUIRE(C_h[i] == (i*i));
REQUIRE(C_h[i] == (i * i));
}
HIP_CHECK(hipFree(A_d));
}
+6 -6
Dosyayı Görüntüle
@@ -31,8 +31,8 @@ THE SOFTWARE.
* - HIP_VERSION >= 6.0
*/
TEST_CASE("Unit_ConfigureCall") {
struct dim3 grid_dim {};
struct dim3 block_dim {};
struct dim3 grid_dim{};
struct dim3 block_dim{};
size_t shared_memory_size = 1024;
HIP_CHECK(hipConfigureCall(grid_dim, block_dim, shared_memory_size));
@@ -50,10 +50,10 @@ TEST_CASE("Unit_ConfigureCall") {
* - HIP_VERSION >= 6.0
*/
TEST_CASE("Unit_ConfigureCall_CheckParams") {
struct dim3 grid_dim { 16, 8, 1 };
struct dim3 test_grid_dim {};
struct dim3 block_dim { 16, 8, 1 };
struct dim3 test_block_dim {};
struct dim3 grid_dim{16, 8, 1};
struct dim3 test_grid_dim{};
struct dim3 block_dim{16, 8, 1};
struct dim3 test_block_dim{};
size_t shmem_size = 1024;
size_t test_shmem_size = 0;
hipStream_t test_stream;
+16 -22
Dosyayı Görüntüle
@@ -20,7 +20,7 @@ THE SOFTWARE.
#include <hip_test_kernels.hh>
#include <hip_test_checkers.hh>
#include <hip_test_common.hh>
#pragma clang diagnostic ignored "-Wunused-parameter"
@@ -29,32 +29,29 @@ unsigned threadsPerBlock = 256;
template <unsigned batch, typename T>
__device__ void sum(T* sdata, unsigned groupElements, unsigned tid) {
T tmp;
if (groupElements < batch)
return;
if (groupElements < batch) return;
// sdata[tid] += sdata[tid - batch/2] does not work when block size is
// greater than wave size because one wave may complete before another
// wave.
if (tid >= batch/2 && tid < groupElements)
tmp = sdata[tid - batch/2];
if (tid >= batch / 2 && tid < groupElements) tmp = sdata[tid - batch / 2];
__syncthreads();
if (tid >= batch/2 && tid < groupElements)
sdata[tid] += tmp;
if (tid >= batch / 2 && tid < groupElements) sdata[tid] += tmp;
__syncthreads();
}
template <typename T>
__global__ void testExternSharedKernel(const T* A_d, const T* B_d, T* C_d,
size_t numElements, size_t groupElements) {
__global__ void testExternSharedKernel(const T* A_d, const T* B_d, T* C_d, size_t numElements,
size_t groupElements) {
// declare dynamic shared memory
extern __shared__ double sdata0[];
T* sdata = reinterpret_cast<T *>(sdata0);
T* sdata = reinterpret_cast<T*>(sdata0);
size_t gid = (blockIdx.x * blockDim.x + threadIdx.x);
size_t tid = threadIdx.x;
// initialize dynamic shared memory
if (tid < groupElements) {
sdata[tid] = static_cast<T>(tid);
sdata[tid] = static_cast<T>(tid);
}
__syncthreads();
@@ -71,15 +68,14 @@ __global__ void testExternSharedKernel(const T* A_d, const T* B_d, T* C_d,
C_d[gid] = A_d[gid] + B_d[gid] + sdata[tid % groupElements];
}
template <typename T>
void testExternShared(size_t N, unsigned groupElements) {
template <typename T> void testExternShared(size_t N, unsigned groupElements) {
size_t Nbytes = N * sizeof(T);
T *A_d, *B_d, *C_d;
T *A_h, *B_h, *C_h;
HipTest::initArrays(&A_d, &B_d, &C_d, &A_h, &B_h, &C_h, N, false);
unsigned blocks = N/threadsPerBlock;
unsigned blocks = N / threadsPerBlock;
assert(N == blocks * threadsPerBlock);
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
@@ -90,8 +86,7 @@ void testExternShared(size_t N, unsigned groupElements) {
// launch kernel with dynamic shared memory
hipLaunchKernelGGL(HIP_KERNEL_NAME(testExternSharedKernel<T>), dim3(blocks),
dim3(threadsPerBlock), groupMemBytes, 0, A_d, B_d, C_d,
N, groupElements);
dim3(threadsPerBlock), groupMemBytes, 0, A_d, B_d, C_d, N, groupElements);
HIP_CHECK(hipDeviceSynchronize());
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
@@ -164,13 +159,12 @@ TEST_CASE("Unit_hipDynamicShared") {
SECTION("test case with float for max LDS size") {
int maxLDS = 0;
HIP_CHECK(hipDeviceGetAttribute(&maxLDS,
hipDeviceAttributeMaxSharedMemoryPerBlock, 0));
testExternShared<float>(1024, maxLDS/sizeof(float));
HIP_CHECK(hipDeviceGetAttribute(&maxLDS, hipDeviceAttributeMaxSharedMemoryPerBlock, 0));
testExternShared<float>(1024, maxLDS / sizeof(float));
}
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+10 -11
Dosyayı Görüntüle
@@ -20,9 +20,9 @@ THE SOFTWARE.
#include <hip_test_kernels.hh>
#include <hip_test_checkers.hh>
#include <hip_test_common.hh>
#define LEN (16 * 1024)
#define LEN (16 * 1024)
#define SIZE (LEN * sizeof(float))
__global__ void vectorAdd(float* Ad, float* Bd) {
@@ -46,7 +46,7 @@ __global__ void vectorAdd(float* Ad, float* Bd) {
/**
* Test Description
* ------------------------
* - Assign max dynamic shared memory to kernel function and
* - Assign max dynamic shared memory to kernel function and
* verify the results.
* Test source
@@ -62,17 +62,16 @@ TEST_CASE("Unit_hipDynamicShared2") {
A = new float[LEN];
B = new float[LEN];
for (int i = 0; i < LEN; i++) {
A[i] = 1.0f;
B[i] = 1.0f;
A[i] = 1.0f;
B[i] = 1.0f;
}
HIP_CHECK(hipMalloc(&Ad, SIZE));
HIP_CHECK(hipMalloc(&Bd, SIZE));
HIP_CHECK(hipMemcpy(Ad, A, SIZE, hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(Bd, B, SIZE, hipMemcpyHostToDevice));
hipError_t ret = hipFuncSetAttribute(
reinterpret_cast<const void*>(&vectorAdd),
hipFuncAttributeMaxDynamicSharedMemorySize, SIZE);
hipError_t ret = hipFuncSetAttribute(reinterpret_cast<const void*>(&vectorAdd),
hipFuncAttributeMaxDynamicSharedMemorySize, SIZE);
REQUIRE(ret == hipSuccess);
hipLaunchKernelGGL(vectorAdd, dim3(1, 1, 1), dim3(64, 1, 1), SIZE, 0, Ad, Bd);
@@ -89,6 +88,6 @@ TEST_CASE("Unit_hipDynamicShared2") {
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+6 -6
Dosyayı Görüntüle
@@ -20,7 +20,7 @@ THE SOFTWARE.
#include <hip_test_kernels.hh>
#include <hip_test_checkers.hh>
#include <hip_test_common.hh>
#pragma clang diagnostic ignored "-Wunused-parameter"
@@ -49,11 +49,11 @@ __global__ void Empty(int param) {}
*/
TEST_CASE("Unit_hipEmptyKernel") {
hipLaunchKernelGGL(HIP_KERNEL_NAME(Empty), dim3(1), dim3(1), 0, 0, 0);
HIP_CHECK(hipDeviceSynchronize());
hipLaunchKernelGGL(HIP_KERNEL_NAME(Empty), dim3(1), dim3(1), 0, 0, 0);
HIP_CHECK(hipDeviceSynchronize());
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+19 -25
Dosyayı Görüntüle
@@ -21,37 +21,35 @@ THE SOFTWARE.
#include <hip_test_kernels.hh>
#include <hip_test_checkers.hh>
#include <hip_test_common.hh>
#include "hip/hip_ext.h"
static unsigned threadsPerBlock = 256;
static unsigned blocksPerCU = 6;
struct _t {
double _a, _b, _c, _d, _e, _f, _g, _h, _i, _j;
double _a, _b, _c, _d, _e, _f, _g, _h, _i, _j;
};
typedef struct _t _T;
__global__ void sKernel(_T s, double *a) {
*a = s._a + s._b + s._c + s._d + s._e + s._f + s._g + s._h + s._i + s._j;
__global__ void sKernel(_T s, double* a) {
*a = s._a + s._b + s._c + s._d + s._e + s._f + s._g + s._h + s._i + s._j;
}
__global__ void mKernel(char f, int16_t a, int b, double c,
int16_t d, int e, double* res) {
*res = a + b + c + d + e + f;
__global__ void mKernel(char f, int16_t a, int b, double c, int16_t d, int e, double* res) {
*res = a + b + c + d + e + f;
}
void testMixData() {
double m = 0;
double *d_m;
double* d_m;
HIP_CHECK(hipMalloc(&d_m, sizeof(double)));
int a = 1, e = 10;
int16_t b = 2, d = 4;
double c = 3.0;
char ff = 10;
hipExtLaunchKernelGGL(mKernel, 1, 1, 0, 0, nullptr, nullptr, 0, ff,
b, a, c, d, e, d_m);
hipExtLaunchKernelGGL(mKernel, 1, 1, 0, 0, nullptr, nullptr, 0, ff, b, a, c, d, e, d_m);
HIP_CHECK(hipMemcpy(&m, d_m, sizeof(double), hipMemcpyDeviceToHost));
REQUIRE(m == 30.0);
HIP_CHECK(hipFree(d_m));
@@ -59,7 +57,7 @@ void testMixData() {
void testStruct() {
double m = 0;
double *d_m;
double* d_m;
HIP_CHECK(hipMalloc(&d_m, sizeof(double)));
_T s{1, 2, 3, 4, 5, 6, 7, 8, 9, 10};
hipExtLaunchKernelGGL(sKernel, 1, 1, 0, 0, nullptr, nullptr, 0, s, d_m);
@@ -80,10 +78,9 @@ void test(size_t N) {
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(B_d, B_h, Nbytes, hipMemcpyHostToDevice));
hipExtLaunchKernelGGL(HipTest::vectorADD, dim3(blocks),
dim3(threadsPerBlock), 0, 0, nullptr, nullptr, 0,
static_cast<const int*>(A_d),
static_cast<const int*>(B_d), C_d, N);
hipExtLaunchKernelGGL(HipTest::vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, 0, nullptr,
nullptr, 0, static_cast<const int*>(A_d), static_cast<const int*>(B_d), C_d,
N);
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
HIP_CHECK(hipDeviceSynchronize());
@@ -99,7 +96,8 @@ void test(size_t N) {
std::uint32_t sharedMemBytes, hipStream_t stream,
hipEvent_t startEvent, hipEvent_t stopEvent, std::uint32_t flags,
Args... args)` -
* Launches kernel with dimention parameters and shared memory on stream with templated kernel and arguments
* Launches kernel with dimention parameters and shared memory on stream with templated kernel and
arguments
*/
/**
@@ -125,15 +123,11 @@ TEST_CASE("Unit_hipExtLaunchKernelGGL") {
size_t N = 4 * 1024 * 1024;
test(N);
}
SECTION("testStruct run") {
testStruct();
}
SECTION("testMixData run") {
testMixData();
}
SECTION("testStruct run") { testStruct(); }
SECTION("testMixData run") { testMixData(); }
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+14 -20
Dosyayı Görüntüle
@@ -21,7 +21,7 @@ THE SOFTWARE.
#include <hip_test_kernels.hh>
#include <hip_test_checkers.hh>
#include <hip_test_common.hh>
static unsigned threadsPerBlock = 256;
static unsigned blocksPerCU = 6;
@@ -30,15 +30,14 @@ static unsigned blocksPerCU = 6;
__device__ int foo(int i) { return i + 1; }
template <typename T>
__global__ void vectorADD2(T* A_d, T* B_d, T* C_d, size_t N) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
template <typename T> __global__ void vectorADD2(T* A_d, T* B_d, T* C_d, size_t N) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
for (size_t i = offset; i < N; i += stride) {
double foo = __hiloint2double(A_d[i], B_d[i]);
C_d[i] = __double2loint(foo) + __double2hiint(foo);
}
for (size_t i = offset; i < N; i += stride) {
double foo = __hiloint2double(A_d[i], B_d[i]);
C_d[i] = __double2loint(foo) + __double2hiint(foo);
}
}
int test_gl2(size_t N) {
@@ -52,8 +51,7 @@ int test_gl2(size_t N) {
// Full vadd in one large chunk, to get things started:
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(B_d, B_h, Nbytes, hipMemcpyHostToDevice));
hipLaunchKernelGGL(vectorADD2, dim3(blocks), dim3(threadsPerBlock),
0, 0, A_d, B_d, C_d, N);
hipLaunchKernelGGL(vectorADD2, dim3(blocks), dim3(threadsPerBlock), 0, 0, A_d, B_d, C_d, N);
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
HIP_CHECK(hipDeviceSynchronize());
// verify
@@ -107,18 +105,14 @@ int test_triple_chevron(size_t N) {
TEST_CASE("Unit_hipGridLaunch") {
size_t N = 4 * 1024 * 1024;
SECTION("Test test_gl2") {
test_gl2(N);
}
SECTION("Test test_gl2") { test_gl2(N); }
#if __HIP__
SECTION("Test triple_chevron") {
test_triple_chevron(N);
}
SECTION("Test triple_chevron") { test_triple_chevron(N); }
#endif
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+11 -16
Dosyayı Görüntüle
@@ -20,7 +20,7 @@ THE SOFTWARE.
#include <hip_test_kernels.hh>
#include <hip_test_common.hh>
#include <hip_test_checkers.hh>
#include <hip/math_functions.h>
#pragma clang diagnostic ignored "-Wunused-variable"
@@ -43,11 +43,10 @@ __device__ __forceinline__ int sum1_forceinline(int a) { return a + 1; }
__device__ __host__ float PlusOne(float x) { return x + 1.0; }
__global__ void MyKernel(const float* a, const float* b, float* c,
unsigned N) {
__global__ void MyKernel(const float* a, const float* b, float* c, unsigned N) {
unsigned gid = threadIdx.x;
if (gid < N) {
c[gid] = a[gid] + PlusOne(b[gid]);
c[gid] = a[gid] + PlusOne(b[gid]);
}
}
@@ -55,18 +54,16 @@ void callMyKernel() {
float *a, *b, *c;
const unsigned blockSize = 256;
unsigned N = blockSize;
hipLaunchKernelGGL(MyKernel, dim3(N / blockSize), dim3(blockSize),
0, 0, a, b, c, N);
hipLaunchKernelGGL(MyKernel, dim3(N / blockSize), dim3(blockSize), 0, 0, a, b, c, N);
}
template <typename T>
__global__ void vectorADD(T __restrict__* A_d, T* B_d, T* C_d, size_t N) {
template <typename T> __global__ void vectorADD(T __restrict__* A_d, T* B_d, T* C_d, size_t N) {
#ifdef NOT_YET
int a = __shfl_up(x, 1);
#endif
float x = 1.0;
#ifdef NOT_YET
float fastZ = __sin(x);
float fastZ = __sin(x);
#endif
__syncthreads();
@@ -74,7 +71,7 @@ __global__ void vectorADD(T __restrict__* A_d, T* B_d, T* C_d, size_t N) {
size_t stride = blockDim.x * gridDim.x;
for (size_t i = offset; i < N; i += stride) {
C_d[i] = A_d[i] + B_d[i];
C_d[i] = A_d[i] + B_d[i];
}
}
@@ -101,11 +98,9 @@ __global__ void vectorADD(T __restrict__* A_d, T* B_d, T* C_d, size_t N) {
* - HIP_VERSION >= 5.5
*/
TEST_CASE("Unit_hipLanguageExtensions") {
REQUIRE(true);
}
TEST_CASE("Unit_hipLanguageExtensions") { REQUIRE(true); }
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+7 -12
Dosyayı Görüntüle
@@ -37,23 +37,19 @@ Testcase Scenarios : hipLaunchBounds_With_maxThreadsPerBlock_blocksPerCU
#include <hip_test_common.hh>
#include <hip_test_kernels.hh>
__global__ void
__launch_bounds__(128, 2)
MyKernel(int N, int *x, int val) {
__global__ void __launch_bounds__(128, 2) MyKernel(int N, int* x, int val) {
for (int i = 0; i < N; i++) {
x[i] = val;
}
}
__global__ void
__launch_bounds__(64)
MyKernel_2(int N, int *x, int val) {
__global__ void __launch_bounds__(64) MyKernel_2(int N, int* x, int val) {
for (int i = 0; i < N; i++) {
x[i] = val;
}
}
static bool verify(int N, int *x, int val) {
static bool verify(int N, int* x, int val) {
for (int i = 0; i < N; i++) {
if (x[i] != val) {
return false;
@@ -65,9 +61,9 @@ static bool verify(int N, int *x, int val) {
TEST_CASE("Unit_hipLaunchBounds_With_maxThreadsPerBlock_Check") {
constexpr size_t N = 10000;
hipError_t ret;
int *x;
int* x;
HIP_CHECK(hipMallocManaged(&x, N*sizeof(int)));
HIP_CHECK(hipMallocManaged(&x, N * sizeof(int)));
REQUIRE(x != nullptr);
SECTION("Passing threadsPerBlock same as kernel launch_bounds") {
@@ -105,9 +101,9 @@ TEST_CASE("Unit_hipLaunchBounds_With_maxThreadsPerBlock_Check") {
TEST_CASE("Unit_hipLaunchBounds_With_maxThreadsPerBlock_blocksPerCU_Check") {
constexpr size_t N = 10000;
hipError_t ret;
int *x;
int* x;
HIP_CHECK(hipMallocManaged(&x, N*sizeof(int)));
HIP_CHECK(hipMallocManaged(&x, N * sizeof(int)));
REQUIRE(x != nullptr);
SECTION("Passing threadsPerBlock same as kernel launch_bounds") {
@@ -170,4 +166,3 @@ TEST_CASE("Unit_hipLaunchBounds_With_maxThreadsPerBlock_blocksPerCU_Check") {
HIP_CHECK(hipFree(x));
}
+50 -82
Dosyayı Görüntüle
@@ -27,7 +27,7 @@ static __device__ int devArr[N];
//------------------------------------------------------------------------------
// Kernel using hipLaunchKernelEx
//------------------------------------------------------------------------------
__global__ void cooperativeKernelEx(int *output, int totalThreads) {
__global__ void cooperativeKernelEx(int* output, int totalThreads) {
cooperative_groups::grid_group grid = cooperative_groups::this_grid();
int tid = threadIdx.x + blockDim.x * blockIdx.x;
if (tid < totalThreads) {
@@ -41,7 +41,7 @@ __global__ void cooperativeKernelEx(int *output, int totalThreads) {
//------------------------------------------------------------------------------
// Kernel using hipLaunchKernelExC
//------------------------------------------------------------------------------
__global__ void cooperativeKernelExC(int *output, int totalThreads) {
__global__ void cooperativeKernelExC(int* output, int totalThreads) {
cooperative_groups::grid_group grid = cooperative_groups::this_grid();
int tid = threadIdx.x + blockDim.x * blockIdx.x;
if (tid < totalThreads) {
@@ -61,7 +61,7 @@ static __global__ void emptyKernel() {}
* Kernel which doesn't use cooperative groups and takes an argument
* and updates the value with 100
*/
static __global__ void argKernel(int *val) { *val = 100; }
static __global__ void argKernel(int* val) { *val = 100; }
/*
* Kernel which uses cooperative groups and without any arguments
@@ -79,7 +79,7 @@ static __global__ void coopEmptykernel() {
* 2) Wait for all the blocks completes it's operations
* 3) Fill each element in the output array with sum of elements in devArr
*/
static __global__ void coopFillArrayKernel(int *output) {
static __global__ void coopFillArrayKernel(int* output) {
cooperative_groups::grid_group grid = cooperative_groups::this_grid();
if (blockIdx.x == 0)
@@ -110,7 +110,7 @@ static __global__ void coopFillArrayKernel(int *output) {
}
}
__global__ void normalKernel(int *output, int totalThreads) {
__global__ void normalKernel(int* output, int totalThreads) {
int tid = threadIdx.x + blockDim.x * blockIdx.x;
if (tid < totalThreads) {
output[tid] = tid * 3;
@@ -157,10 +157,10 @@ TEST_CASE("Unit_hipLaunchKernelExC_NegetiveTsts") {
config.attrs = &attr;
config.numAttrs = 1;
int *d_output = nullptr;
int* d_output = nullptr;
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
void *kernelArgs[] = {&d_output, (void *)&totalThreads};
void* kernelArgs[] = {&d_output, (void*)&totalThreads};
SECTION("Kernel function as nullptr") {
HIP_CHECK_ERROR(hipLaunchKernelExC(&config, nullptr, kernelArgs),
@@ -168,15 +168,12 @@ TEST_CASE("Unit_hipLaunchKernelExC_NegetiveTsts") {
}
SECTION("Kernel args as nullptr") {
HIP_CHECK_ERROR(
hipLaunchKernelExC(&config, (void *)cooperativeKernelExC, nullptr),
hipErrorInvalidValue);
HIP_CHECK_ERROR(hipLaunchKernelExC(&config, (void*)cooperativeKernelExC, nullptr),
hipErrorInvalidValue);
}
SECTION("Non Cooparative Kernel") {
HIP_CHECK_ERROR(
hipLaunchKernelExC(&config, (void *)normalKernel, kernelArgs),
hipSuccess);
HIP_CHECK_ERROR(hipLaunchKernelExC(&config, (void*)normalKernel, kernelArgs), hipSuccess);
}
hipLaunchConfig_t invalidConfig = {};
@@ -192,9 +189,7 @@ TEST_CASE("Unit_hipLaunchKernelExC_NegetiveTsts") {
invalidConfig.numAttrs = 1;
SECTION("Invalid Kernel Config") {
HIP_CHECK_ERROR(hipLaunchKernelExC(&invalidConfig,
(void *)cooperativeKernelExC,
kernelArgs),
HIP_CHECK_ERROR(hipLaunchKernelExC(&invalidConfig, (void*)cooperativeKernelExC, kernelArgs),
hipErrorInvalidConfiguration);
}
}
@@ -231,7 +226,7 @@ TEST_CASE("Unit_hipLaunchKernelEx_NegetiveTsts") {
config.attrs = &attr;
config.numAttrs = 1;
int *d_output = nullptr;
int* d_output = nullptr;
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
@@ -240,10 +235,9 @@ TEST_CASE("Unit_hipLaunchKernelEx_NegetiveTsts") {
}
SECTION("Non Cooparative Kernel") {
HIP_CHECK_ERROR(hipLaunchKernelEx(&config,
(void (*)(int *, int))normalKernel,
d_output, totalThreads),
hipSuccess);
HIP_CHECK_ERROR(
hipLaunchKernelEx(&config, (void (*)(int*, int))normalKernel, d_output, totalThreads),
hipSuccess);
}
hipLaunchConfig_t invalidConfig = {};
@@ -259,18 +253,17 @@ TEST_CASE("Unit_hipLaunchKernelEx_NegetiveTsts") {
invalidConfig.numAttrs = 1;
SECTION("Invalid Kernel Config") {
HIP_CHECK_ERROR(hipLaunchKernelEx(&invalidConfig,
(void (*)(int *, int))cooperativeKernelEx,
HIP_CHECK_ERROR(hipLaunchKernelEx(&invalidConfig, (void (*)(int*, int))cooperativeKernelEx,
d_output, totalThreads),
hipErrorInvalidConfiguration);
}
}
bool runTest(const char *testName, const void *kernelFunc, int totalThreads,
int blockSize, int flagValue, bool useTemplate) {
bool runTest(const char* testName, const void* kernelFunc, int totalThreads, int blockSize,
int flagValue, bool useTemplate) {
const int numBlocks = (totalThreads + blockSize - 1) / blockSize;
int *d_output = nullptr;
int* d_output = nullptr;
HIP_CHECK(hipMalloc(&d_output, totalThreads * sizeof(int)));
HIP_CHECK(hipMemset(d_output, 0, totalThreads * sizeof(int)));
@@ -288,33 +281,31 @@ bool runTest(const char *testName, const void *kernelFunc, int totalThreads,
// For a kernel parameter declared as "int* output", pass the address of the
// device pointer.
void *kernelArgs[] = {&d_output, (void *)&totalThreads};
void* kernelArgs[] = {&d_output, (void*)&totalThreads};
if (useTemplate) {
HIP_CHECK(hipLaunchKernelEx(&config, (void (*)(int *, int))kernelFunc,
d_output, totalThreads));
HIP_CHECK(hipLaunchKernelEx(&config, (void (*)(int*, int))kernelFunc, d_output, totalThreads));
} else {
HIP_CHECK(hipLaunchKernelExC(&config, kernelFunc, kernelArgs));
}
HIP_CHECK(hipDeviceSynchronize());
int *h_output = (int *)malloc(totalThreads * sizeof(int));
HIP_CHECK(hipMemcpy(h_output, d_output, totalThreads * sizeof(int),
hipMemcpyDeviceToHost));
int* h_output = (int*)malloc(totalThreads * sizeof(int));
HIP_CHECK(hipMemcpy(h_output, d_output, totalThreads * sizeof(int), hipMemcpyDeviceToHost));
// Verify results.
bool success = true;
if (h_output[0] != flagValue) {
printf("%s test failed: Expected flag %d at index 0, got %d\n", testName,
flagValue, h_output[0]);
printf("%s test failed: Expected flag %d at index 0, got %d\n", testName, flagValue,
h_output[0]);
success = false;
}
for (int i = 1; i < totalThreads; i++) {
int expectedValue = (flagValue == 1111) ? i : (i * 3);
if (h_output[i] != expectedValue) {
printf("%s test failed at index %d: Expected %d, got %d\n", testName, i,
expectedValue, h_output[i]);
printf("%s test failed at index %d: Expected %d, got %d\n", testName, i, expectedValue,
h_output[i]);
success = false;
break;
}
@@ -344,12 +335,10 @@ TEST_CASE("Unit_hipLaunchKernelEx_Functional") {
}
std::string api_type = GENERATE("hipLaunchKernelEx", "hipLaunchKernelExC");
if (api_type == "hipLaunchKernelEx") {
REQUIRE(runTest(api_type.c_str(), (void *)cooperativeKernelEx, 64, 16, 2222,
true) == true);
REQUIRE(runTest(api_type.c_str(), (void*)cooperativeKernelEx, 64, 16, 2222, true) == true);
}
if (api_type == "hipLaunchKernelExC") {
REQUIRE(runTest(api_type.c_str(), (void *)cooperativeKernelExC, 64, 16,
1111, false) == true);
REQUIRE(runTest(api_type.c_str(), (void*)cooperativeKernelExC, 64, 16, 1111, false) == true);
}
}
/**
@@ -391,13 +380,10 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_Different_Kernels") {
config.numAttrs = 1;
SECTION("Normal kernel with no arguments") {
SECTION("hipLaunchKernelEx") {
HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel));
}
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel)); }
SECTION("hipLaunchKernelExC") {
HIP_CHECK(hipLaunchKernelExC(
&config, reinterpret_cast<void *>(emptyKernel), nullptr));
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(emptyKernel), nullptr));
}
HIP_CHECK(hipDeviceSynchronize());
}
@@ -406,17 +392,14 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_Different_Kernels") {
config.gridDim = dim3{1, 1, 1};
config.blockDim = dim3{1, 1, 1};
int *devMem = nullptr;
int* devMem = nullptr;
HIP_CHECK(hipMalloc(&devMem, sizeof(int)));
SECTION("hipLaunchKernelEx") {
HIP_CHECK(hipLaunchKernelEx(&config, argKernel, devMem));
}
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, argKernel, devMem)); }
SECTION("hipLaunchKernelExC") {
void *kernel_args[1] = {&devMem};
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void *>(argKernel),
kernel_args));
void* kernel_args[1] = {&devMem};
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(argKernel), kernel_args));
}
HIP_CHECK(hipDeviceSynchronize());
@@ -426,13 +409,10 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_Different_Kernels") {
}
SECTION("Cooperative kernel with no arguments") {
SECTION("hipLaunchKernelEx") {
HIP_CHECK(hipLaunchKernelEx(&config, coopEmptykernel));
}
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, coopEmptykernel)); }
SECTION("hipLaunchKernelExC") {
HIP_CHECK(hipLaunchKernelExC(
&config, reinterpret_cast<void *>(coopEmptykernel), nullptr));
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(coopEmptykernel), nullptr));
}
HIP_CHECK(hipDeviceSynchronize());
}
@@ -475,7 +455,7 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_CooperativeKernelWithArgs") {
hostMem[i] = 0;
}
int *devMem = nullptr;
int* devMem = nullptr;
HIP_CHECK(hipMalloc(&devMem, N * sizeof(int)));
HIP_CHECK(hipMemcpy(devMem, hostMem, N * sizeof(int), hipMemcpyDefault));
@@ -484,9 +464,9 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_CooperativeKernelWithArgs") {
}
SECTION("hipLaunchKernelExC") {
void *kernel_args[1] = {&devMem};
HIP_CHECK(hipLaunchKernelExC(
&config, reinterpret_cast<void *>(coopFillArrayKernel), kernel_args));
void* kernel_args[1] = {&devMem};
HIP_CHECK(
hipLaunchKernelExC(&config, reinterpret_cast<void*>(coopFillArrayKernel), kernel_args));
}
HIP_CHECK(hipDeviceSynchronize());
HIP_CHECK(hipMemcpy(hostMem, devMem, N * sizeof(int), hipMemcpyDefault));
@@ -534,47 +514,35 @@ TEST_CASE("Unit_hipLaunchKernelEx_With_MaxBlockDims") {
config.numAttrs = 1;
SECTION("blockDim.x == maxBlockDimX") {
const unsigned int x =
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimX, 0);
const unsigned int x = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimX, 0);
config.blockDim = dim3{x, 1, 1};
SECTION("hipLaunchKernelEx") {
HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel));
}
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel)); }
SECTION("hipLaunchKernelExC") {
HIP_CHECK(hipLaunchKernelExC(
&config, reinterpret_cast<void *>(emptyKernel), nullptr));
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(emptyKernel), nullptr));
}
}
SECTION("blockDim.y == maxBlockDimY") {
const unsigned int y =
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimY, 0);
const unsigned int y = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimY, 0);
config.blockDim = dim3{1, y, 1};
SECTION("hipLaunchKernelEx") {
HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel));
}
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel)); }
SECTION("hipLaunchKernelExC") {
HIP_CHECK(hipLaunchKernelExC(
&config, reinterpret_cast<void *>(emptyKernel), nullptr));
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(emptyKernel), nullptr));
}
}
SECTION("blockDim.z == maxBlockDimZ") {
const unsigned int z =
GetDeviceAttribute(hipDeviceAttributeMaxBlockDimZ, 0);
const unsigned int z = GetDeviceAttribute(hipDeviceAttributeMaxBlockDimZ, 0);
config.blockDim = dim3{1, 1, z};
SECTION("hipLaunchKernelEx") {
HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel));
}
SECTION("hipLaunchKernelEx") { HIP_CHECK(hipLaunchKernelEx(&config, emptyKernel)); }
SECTION("hipLaunchKernelExC") {
HIP_CHECK(hipLaunchKernelExC(
&config, reinterpret_cast<void *>(emptyKernel), nullptr));
HIP_CHECK(hipLaunchKernelExC(&config, reinterpret_cast<void*>(emptyKernel), nullptr));
}
}
HIP_CHECK(hipDeviceSynchronize());
+208 -305
Dosyayı Görüntüle
@@ -54,7 +54,7 @@ THE SOFTWARE.
// Bit fields are broken
#define ENABLE_BIT_FIELDS 0
static const int BLOCK_DIM_SIZE = 512;
static const int BLOCK_DIM_SIZE = 512;
// allocate memory on device and host for result validation
static bool *result_d, *result_h;
@@ -64,8 +64,7 @@ static hipError_t hipHostMallocError = hipErrorUnknown;
static hipError_t hipMemsetError = hipErrorUnknown;
static void ResultValidation() {
HIP_CHECK(hipMemcpy(result_h, result_d, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
HIP_CHECK(hipMemcpy(result_h, result_d, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
REQUIRE(result_h[k] == true);
@@ -86,8 +85,8 @@ static void ResetValidationMem() {
// This test is to verify Struct with variables
// support, read from device.
typedef struct hipLaunchKernelStruct1 {
int li; // local int
float lf; // local float
int li; // local int
float lf; // local float
bool result; // local bool
} hipLaunchKernelStruct_t1;
@@ -126,14 +125,14 @@ typedef struct hipLaunchKernelStruct5 {
typedef struct hipLaunchKernelStruct6 {
char c1;
int16_t si;
} __attribute__((aligned(8))) hipLaunchKernelStruct_t6;
} __attribute__((aligned(8))) hipLaunchKernelStruct_t6;
// This test is to verify struct with aligned(16),
// right now it's brokenon hcc & hip-clang
typedef struct hipLaunchKernelStruct7 {
char c1;
int16_t si;
} __attribute__((aligned(16))) hipLaunchKernelStruct_t7;
} __attribute__((aligned(16))) hipLaunchKernelStruct_t7;
// This test is to verify struct with packed & aligned,
// size should be 4Bytes right now it's broken on hcc & hip-clang
@@ -141,7 +140,7 @@ typedef struct hipLaunchKernelStruct8 {
char c1;
int16_t si;
bool b;
}__attribute__((packed, aligned(4))) hipLaunchKernelStruct_t8;
} __attribute__((packed, aligned(4))) hipLaunchKernelStruct_t8;
// This test is to verify struct with packed, no alignment as Sam suggested
// size should be 4Bytes, right now it's broken on hcc & hip-clang
@@ -149,7 +148,7 @@ typedef struct hipLaunchKernelStruct8A {
char c1;
int16_t si;
bool b;
}__attribute__((packed)) hipLaunchKernelStruct_t8A;
} __attribute__((packed)) hipLaunchKernelStruct_t8A;
// This test is to verify struct with alignment, no packing as Sam suggested
// size should be 8Bytes as no packing, right now it's broken on hcc & hip-clang
@@ -157,7 +156,7 @@ typedef struct hipLaunchKernelStruct8B {
char c1;
int16_t si;
bool b;
}__attribute__((aligned(8))) hipLaunchKernelStruct_t8B;
} __attribute__((aligned(8))) hipLaunchKernelStruct_t8B;
// This test is to verify const struct object
typedef struct hipLaunchKernelStruct9 {
@@ -181,8 +180,8 @@ typedef struct hipLaunchKernelStruct11 {
// This test is to verify struct with simple class object
class base {
public:
int i = 0;
base() {}
int i = 0;
base() {}
};
typedef struct hipLaunchKernelStruct12 {
base b;
@@ -210,14 +209,13 @@ typedef struct hipLaunchKernelStruct15 {
} hipLaunchKernelStruct_t15;
// This test is to verify simple template struct
template<typename T>
struct hipLaunchKernelStruct_t16 {
template <typename T> struct hipLaunchKernelStruct_t16 {
T t1;
};
// This test is to verify simple explicity template struct
template<typename T> struct hipLaunchKernelStruct_t17 {};
template<> // explicit template
template <typename T> struct hipLaunchKernelStruct_t17 {};
template <> // explicit template
struct hipLaunchKernelStruct_t17<int> {
int t1;
};
@@ -246,301 +244,257 @@ typedef struct hipLaunchKernelStruct21 {
// Passing struct to a hipLaunchKernelGGL(),
// read and write into the same struct
__global__ void hipLaunchKernelStructFunc1(
hipLaunchKernelStruct_t1 hipLaunchKernelStruct_,
bool* result_d1) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
__global__ void hipLaunchKernelStructFunc1(hipLaunchKernelStruct_t1 hipLaunchKernelStruct_,
bool* result_d1) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d1[x] = ((hipLaunchKernelStruct_.li == 1)
&& (hipLaunchKernelStruct_.lf == 1.0)
&& (hipLaunchKernelStruct_.result == false));
// set the result to true if the condition met
result_d1[x] = ((hipLaunchKernelStruct_.li == 1) && (hipLaunchKernelStruct_.lf == 1.0) &&
(hipLaunchKernelStruct_.result == false));
}
// Passing struct to a hipLaunchKernelGGL(), checks padding,
// read and write into the same struct
__global__ void hipLaunchKernelStructFunc2(
hipLaunchKernelStruct_t2 hipLaunchKernelStruct_,
bool* result_d2) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
__global__ void hipLaunchKernelStructFunc2(hipLaunchKernelStruct_t2 hipLaunchKernelStruct_,
bool* result_d2) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d2[x] = ((hipLaunchKernelStruct_.c1 == 'a')
&& (hipLaunchKernelStruct_.l1 == 1.0)
&& (hipLaunchKernelStruct_.c2 == 'b')
&& (hipLaunchKernelStruct_.l2 == 2.0) );
// set the result to true if the condition met
result_d2[x] = ((hipLaunchKernelStruct_.c1 == 'a') && (hipLaunchKernelStruct_.l1 == 1.0) &&
(hipLaunchKernelStruct_.c2 == 'b') && (hipLaunchKernelStruct_.l2 == 2.0));
}
// Passing struct to a hipLaunchKernelGGL(), checks padding,
// read and write into the same struct
__global__ void hipLaunchKernelStructFunc3(
hipLaunchKernelStruct_t3 hipLaunchKernelStruct_,
bool* result_d3) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
__global__ void hipLaunchKernelStructFunc3(hipLaunchKernelStruct_t3 hipLaunchKernelStruct_,
bool* result_d3) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d3[x] = ((hipLaunchKernelStruct_.bf1 == 1)
&& (hipLaunchKernelStruct_.bf2 == 1)
&& (hipLaunchKernelStruct_.l1 == 1.0)
&& (hipLaunchKernelStruct_.bf3 == 1) );
// set the result to true if the condition met
result_d3[x] = ((hipLaunchKernelStruct_.bf1 == 1) && (hipLaunchKernelStruct_.bf2 == 1) &&
(hipLaunchKernelStruct_.l1 == 1.0) && (hipLaunchKernelStruct_.bf3 == 1));
}
// Passing empty struct to a hipLaunchKernelGGL(),
// check the size of 1Byte, set result_d4 to true if condition met
__global__ void hipLaunchKernelStructFunc4(
hipLaunchKernelStruct_t4 hipLaunchKernelStruct_,
bool* result_d4) {
__global__ void hipLaunchKernelStructFunc4(hipLaunchKernelStruct_t4 hipLaunchKernelStruct_,
bool* result_d4) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d4[x] = (sizeof(hipLaunchKernelStruct_) == 1);
result_d4[x] = (sizeof(hipLaunchKernelStruct_) == 1);
}
// Passing struct with pointer object to a hipLaunchKernelGGL()
__global__ void hipLaunchKernelStructFunc5(
hipLaunchKernelStruct_t5 hipLaunchKernelStruct_,
bool* result_d5) {
__global__ void hipLaunchKernelStructFunc5(hipLaunchKernelStruct_t5 hipLaunchKernelStruct_,
bool* result_d5) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d5[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (*hipLaunchKernelStruct_.cp == 'p'));
result_d5[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (*hipLaunchKernelStruct_.cp == 'p'));
}
// Passing struct which is aligned to 8Byte to a hipLaunchKernelGGL(),
// set the result_d6 to true if condition met
__global__ void hipLaunchKernelStructFunc6(
hipLaunchKernelStruct_t6 hipLaunchKernelStruct_,
bool* result_d6) {
__global__ void hipLaunchKernelStructFunc6(hipLaunchKernelStruct_t6 hipLaunchKernelStruct_,
bool* result_d6) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
// get the address of the struct
// size_t(p)%8 will be 0 if aligned to 8Byte address space
int *p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
result_d6[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (hipLaunchKernelStruct_.si == 1)
&& ((size_t(p))%8 ==0));
int* p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
result_d6[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
((size_t(p)) % 8 == 0));
}
// Passing struct which is aligned to 16Byte,
// set the result_d7 to true if condition met
__global__ void hipLaunchKernelStructFunc7(
hipLaunchKernelStruct_t7 hipLaunchKernelStruct_,
bool* result_d7) {
__global__ void hipLaunchKernelStructFunc7(hipLaunchKernelStruct_t7 hipLaunchKernelStruct_,
bool* result_d7) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
// get the address of the struct
// size_t(p)%16 will be 0 if aligned to 16Byte address space
int *p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
result_d7[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (hipLaunchKernelStruct_.si == 1)
&& ((size_t(p))%16 ==0) );
int* p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
result_d7[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
((size_t(p)) % 16 == 0));
}
// Passing struct which is packed & aligned to 4Byte,
// set the result_d8 to true if condition met
__global__ void hipLaunchKernelStructFunc8(
hipLaunchKernelStruct_t8 hipLaunchKernelStruct_,
bool* result_d8) {
__global__ void hipLaunchKernelStructFunc8(hipLaunchKernelStruct_t8 hipLaunchKernelStruct_,
bool* result_d8) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
// get the address of the xth element, struct[x],
// size_t(p)%4 will be 0 if aligned to 4Byte address space
int *p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
result_d8[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (hipLaunchKernelStruct_.si == 1)
&& ((size_t(p))%4 ==0)
&& (sizeof(hipLaunchKernelStruct_) == 4));
int* p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
result_d8[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
((size_t(p)) % 4 == 0) && (sizeof(hipLaunchKernelStruct_) == 4));
}
// Passing struct which is packed only, as Sam suggested, should be 4Bytes
// set the result_d8A to true if condition met
__global__ void hipLaunchKernelStructFunc8A(
hipLaunchKernelStruct_t8A hipLaunchKernelStruct_,
bool* result_d8A) {
__global__ void hipLaunchKernelStructFunc8A(hipLaunchKernelStruct_t8A hipLaunchKernelStruct_,
bool* result_d8A) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
// this is packed struct
// the address will not be aglined in this case hence condition removed
// only sizeof(hipLaunchKernelStruct_) will be valided
result_d8A[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (hipLaunchKernelStruct_.si == 1)
&& (sizeof(hipLaunchKernelStruct_) == 4));
result_d8A[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
(sizeof(hipLaunchKernelStruct_) == 4));
}
// Passing struct which is aligned(4) only, as Sam suggested
// , size should be 8Bytes, set the result_d8B to true if condition met
__global__ void hipLaunchKernelStructFunc8B(
hipLaunchKernelStruct_t8B hipLaunchKernelStruct_,
bool* result_d8B) {
__global__ void hipLaunchKernelStructFunc8B(hipLaunchKernelStruct_t8B hipLaunchKernelStruct_,
bool* result_d8B) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
// get the address of the xth element, struct[x],
// size_t(p)%4 will be 0 if aligned to 4Byte address space
int *p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
result_d8B[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (hipLaunchKernelStruct_.si == 1)
&& ((size_t(p))%8 == 0)
&& (sizeof(hipLaunchKernelStruct_) == 8));
int* p = reinterpret_cast<int*>(&hipLaunchKernelStruct_);
result_d8B[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.si == 1) &&
((size_t(p)) % 8 == 0) && (sizeof(hipLaunchKernelStruct_) == 8));
}
// Passing struct with uint pointer object to a hipLaunchKernelGGL()
__global__ void hipLaunchKernelStructFunc9(
const hipLaunchKernelStruct_t9 hipLaunchKernelStruct_,
bool* result_d9) {
__global__ void hipLaunchKernelStructFunc9(const hipLaunchKernelStruct_t9 hipLaunchKernelStruct_,
bool* result_d9) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d9[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (*hipLaunchKernelStruct_.ip == 1));
result_d9[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (*hipLaunchKernelStruct_.ip == 1));
}
// Passing struct with stdint types object, uintN_t, to a hipLaunchKernelGGL()
__global__ void hipLaunchKernelStructFunc10(
hipLaunchKernelStruct_t10 hipLaunchKernelStruct_,
bool* result_d10) {
__global__ void hipLaunchKernelStructFunc10(hipLaunchKernelStruct_t10 hipLaunchKernelStruct_,
bool* result_d10) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d10[x] = ((hipLaunchKernelStruct_.u64 == UINT64_MAX)
&& (hipLaunchKernelStruct_.u32 == 1)
&& (hipLaunchKernelStruct_.u8 == UINT8_MAX));
result_d10[x] = ((hipLaunchKernelStruct_.u64 == UINT64_MAX) &&
(hipLaunchKernelStruct_.u32 == 1) && (hipLaunchKernelStruct_.u8 == UINT8_MAX));
}
// Passing struct with volatile member, to a hipLaunchKernelGGL()
__global__ void hipLaunchKernelStructFunc11(
hipLaunchKernelStruct_t11 hipLaunchKernelStruct_,
bool* result_d11) {
__global__ void hipLaunchKernelStructFunc11(hipLaunchKernelStruct_t11 hipLaunchKernelStruct_,
bool* result_d11) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d11[x] = ((hipLaunchKernelStruct_.i1 == 1)
&& (hipLaunchKernelStruct_.vint == 0));
result_d11[x] = ((hipLaunchKernelStruct_.i1 == 1) && (hipLaunchKernelStruct_.vint == 0));
}
// Passing struct with simple class obj, to a hipLaunchKernelGGL()
__global__ void hipLaunchKernelStructFunc12(
hipLaunchKernelStruct_t12 hipLaunchKernelStruct_,
bool* result_d12) {
__global__ void hipLaunchKernelStructFunc12(hipLaunchKernelStruct_t12 hipLaunchKernelStruct_,
bool* result_d12) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d12[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (hipLaunchKernelStruct_.b.i == 0));
result_d12[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.b.i == 0));
}
// Passing struct with simple __device__ func(), to a hipLaunchKernelGGL()
__global__ void hipLaunchKernelStructFunc13(
hipLaunchKernelStruct_t13 hipLaunchKernelStruct_,
bool* result_d13) {
__global__ void hipLaunchKernelStructFunc13(hipLaunchKernelStruct_t13 hipLaunchKernelStruct_,
bool* result_d13) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d13[x] = ((hipLaunchKernelStruct_.i1 == 1)
&& (hipLaunchKernelStruct_.getvalue() == 1));
result_d13[x] = ((hipLaunchKernelStruct_.i1 == 1) && (hipLaunchKernelStruct_.getvalue() == 1));
}
// Passing struct with array variable, write to from device
__global__ void hipLaunchKernelStructFunc14(
hipLaunchKernelStruct_t14 hipLaunchKernelStruct_,
bool* result_d14) {
__global__ void hipLaunchKernelStructFunc14(hipLaunchKernelStruct_t14 hipLaunchKernelStruct_,
bool* result_d14) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
hipLaunchKernelStruct_.writeint[x] = 1;
// set the result to true if the condition met
result_d14[x] = ((hipLaunchKernelStruct_.readint == 1)
&& (hipLaunchKernelStruct_.writeint[x] == 1));
result_d14[x] =
((hipLaunchKernelStruct_.readint == 1) && (hipLaunchKernelStruct_.writeint[x] == 1));
}
// Passing struct with struct with dynamic memory, new int
// the heap memory will be accessed from device
__global__ void hipLaunchKernelStructFunc15(
hipLaunchKernelStruct_t15 hipLaunchKernelStruct_,
bool* result_d15) {
__global__ void hipLaunchKernelStructFunc15(hipLaunchKernelStruct_t15 hipLaunchKernelStruct_,
bool* result_d15) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d15[x] = ((hipLaunchKernelStruct_.c1 == 'c')
&& (hipLaunchKernelStruct_.heapmem[x] == 1));
result_d15[x] = ((hipLaunchKernelStruct_.c1 == 'c') && (hipLaunchKernelStruct_.heapmem[x] == 1));
}
// Passing simple template struct
__global__ void hipLaunchKernelStructFunc16(
hipLaunchKernelStruct_t16<char> hipLaunchKernelStruct_,
bool* result_d16) {
__global__ void hipLaunchKernelStructFunc16(hipLaunchKernelStruct_t16<char> hipLaunchKernelStruct_,
bool* result_d16) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d16[x] = (hipLaunchKernelStruct_.t1 == 'c');
result_d16[x] = (hipLaunchKernelStruct_.t1 == 'c');
}
// Passing simple explicit template struct
__global__ void hipLaunchKernelStructFunc17(
hipLaunchKernelStruct_t17<int> hipLaunchKernelStruct_,
bool* result_d17) {
__global__ void hipLaunchKernelStructFunc17(hipLaunchKernelStruct_t17<int> hipLaunchKernelStruct_,
bool* result_d17) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// set the result to true if the condition met
result_d17[x] = (hipLaunchKernelStruct_.t1 == 1);
result_d17[x] = (hipLaunchKernelStruct_.t1 == 1);
}
// Passing struct and write to struct memory using __device__ func()
__global__ void hipLaunchKernelStructFunc18(
hipLaunchKernelStruct_t18 hipLaunchKernelStruct_,
bool* result_d18) {
__global__ void hipLaunchKernelStructFunc18(hipLaunchKernelStruct_t18 hipLaunchKernelStruct_,
bool* result_d18) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
hipLaunchKernelStruct_.setChar('c');
// set the result to true if the condition met
result_d18[x] = (hipLaunchKernelStruct_.getChar() == 'c');
result_d18[x] = (hipLaunchKernelStruct_.getChar() == 'c');
}
// Passing out of order initalized struct, access in-order
__global__ void hipLaunchKernelStructFunc20(
hipLaunchKernelStruct_t20 hipLaunchKernelStruct_,
bool* result_d20) {
__global__ void hipLaunchKernelStructFunc20(hipLaunchKernelStruct_t20 hipLaunchKernelStruct_,
bool* result_d20) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// accessing struct members in order
result_d20[x] = (hipLaunchKernelStruct_.name == 'A'
// strcmp(hipLaunchKernelStruct_.name, "AMD") -> strcmp is not broken
&& hipLaunchKernelStruct_.age == 42
&& hipLaunchKernelStruct_.rank == 2);
// strcmp(hipLaunchKernelStruct_.name, "AMD") -> strcmp is not broken
&& hipLaunchKernelStruct_.age == 42 && hipLaunchKernelStruct_.rank == 2);
}
// Passing struct with bit fields
__global__ void hipLaunchKernelStructFunc21(
hipLaunchKernelStruct_t21 hipLaunchKernelStruct_,
bool* result_d21) {
__global__ void hipLaunchKernelStructFunc21(hipLaunchKernelStruct_t21 hipLaunchKernelStruct_,
bool* result_d21) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
// accessing struct members in order
result_d21[x] = (hipLaunchKernelStruct_.i == 2
&& hipLaunchKernelStruct_.j == 0
&& (sizeof(hipLaunchKernelStruct_) == 1));
result_d21[x] = (hipLaunchKernelStruct_.i == 2 && hipLaunchKernelStruct_.j == 0 &&
(sizeof(hipLaunchKernelStruct_) == 1));
}
__global__ void vAdd(float* a) {}
template<class T1, class T2>
__global__ void myKernel(T1 a, T2 b) {}
template <class T1, class T2> __global__ void myKernel(T1 a, T2 b) {}
//---
// Some wrapper macro for testing:
#define WRAP(...) __VA_ARGS__
#define MY_LAUNCH_MACRO(cmd, elapsed, quiet) \
do { \
HIP_CHECK(hipDeviceSynchronize()); \
cmd; \
HIP_CHECK(hipDeviceSynchronize()); \
} while (0);
#define MY_LAUNCH_MACRO(cmd, elapsed, quiet) \
do { \
HIP_CHECK(hipDeviceSynchronize()); \
cmd; \
HIP_CHECK(hipDeviceSynchronize()); \
} while (0);
#define MY_LAUNCH(command, doTrace, msg) \
{ \
if (doTrace) printf("TRACE: %s %s\n", msg, #command); \
command; \
}
#define MY_LAUNCH(command, doTrace, msg) \
{ \
if (doTrace) printf("TRACE: %s %s\n", msg, #command); \
command; \
}
#define MY_LAUNCH_WITH_PAREN(command, doTrace, msg) \
{ \
if (doTrace) printf("TRACE: %s %s\n", msg, #command); \
(command); \
}
#define MY_LAUNCH_WITH_PAREN(command, doTrace, msg) \
{ \
if (doTrace) printf("TRACE: %s %s\n", msg, #command); \
(command); \
}
/**
* @addtogroup hipLaunchKernelGGL hipLaunchKernelGGL
@@ -589,10 +543,9 @@ __global__ void myKernel(T1 a, T2 b) {}
*/
TEST_CASE("Unit_hipLaunchParm") {
hipMallocError = hipMalloc(reinterpret_cast<void**>(&result_d),
BLOCK_DIM_SIZE*sizeof(bool));
hipHostMallocError = hipHostMalloc(reinterpret_cast<void**>(&result_h),
BLOCK_DIM_SIZE*sizeof(bool));
hipMallocError = hipMalloc(reinterpret_cast<void**>(&result_d), BLOCK_DIM_SIZE * sizeof(bool));
hipHostMallocError =
hipHostMalloc(reinterpret_cast<void**>(&result_h), BLOCK_DIM_SIZE * sizeof(bool));
hipMemsetError = hipMemset(result_d, false, BLOCK_DIM_SIZE);
// Validating memory & initial value, for result_d, result_h
@@ -606,10 +559,8 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_h1.li = 1;
hipLaunchKernelStruct_h1.lf = 1.0;
hipLaunchKernelStruct_h1.result = false;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc1),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h1,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc1), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h1, result_d);
ResultValidation();
}
@@ -621,10 +572,8 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_h2.c2 = 'b';
hipLaunchKernelStruct_h2.l2 = 2.0;
hipLaunchKernelStruct_h2.result = false;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc2),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h2,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc2), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h2, result_d);
ResultValidation();
}
@@ -636,22 +585,18 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_h3.l1 = 1.0;
hipLaunchKernelStruct_h3.bf3 = 1;
hipLaunchKernelStruct_h3.result = false;
// initialize to false, will be set to
// true if the struct size is 1Byte, from device size
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc3),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h3,
result_d);
// initialize to false, will be set to
// true if the struct size is 1Byte, from device size
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc3), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h3, result_d);
ResultValidation();
}
SECTION("Empty struct") {
ResetValidationMem();
hipLaunchKernelStruct_t4 hipLaunchKernelStruct_h4;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc4),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h4,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc4), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h4, result_d);
ResultValidation();
}
@@ -664,10 +609,8 @@ TEST_CASE("Unit_hipLaunchParm") {
HIP_CHECK(hipMemset(cp_d5, 'p', sizeof(char)));
hipLaunchKernelStruct_h5.c1 = 'c';
hipLaunchKernelStruct_h5.cp = cp_d5;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc5),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h5,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc5), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h5, result_d);
ResultValidation();
HIP_CHECK(hipFree(reinterpret_cast<void*>(cp_d5)));
}
@@ -677,14 +620,12 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_t6 hipLaunchKernelStruct_h6;
hipLaunchKernelStruct_h6.c1 = 'c';
hipLaunchKernelStruct_h6.si = 1;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc6),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h6,
result_d);
// alignment is broken hence disabled the validation part
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc6), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h6, result_d);
// alignment is broken hence disabled the validation part
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR
ResultValidation();
#endif
#endif
}
SECTION("Passing struct with aligned(16)") {
@@ -692,13 +633,11 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_t7 hipLaunchKernelStruct_h7;
hipLaunchKernelStruct_h7.c1 = 'c';
hipLaunchKernelStruct_h7.si = 1;
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR // This is broken on small bar
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc7),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h7,
result_d);
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR // This is broken on small bar
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc7), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h7, result_d);
ResultValidation();
#endif
#endif
}
SECTION("Passing struct with packed aligned to 4bytes") {
@@ -706,14 +645,12 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_t8 hipLaunchKernelStruct_h8;
hipLaunchKernelStruct_h8.c1 = 'c';
hipLaunchKernelStruct_h8.si = 1;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h8,
result_d);
// packed member broken on large and small bar setup.
#if ENABLE_PACKED_TEST
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h8, result_d);
// packed member broken on large and small bar setup.
#if ENABLE_PACKED_TEST
ResultValidation();
#endif
#endif
}
SECTION("Passing struct with packed to 4Bytes") {
@@ -721,14 +658,12 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_t8A hipLaunchKernelStruct_h8A;
hipLaunchKernelStruct_h8A.c1 = 'c';
hipLaunchKernelStruct_h8A.si = 1;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8A),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h8A,
result_d);
// packed member broken on large and small bar setup.
#if ENABLE_PACKED_TEST
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8A), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h8A, result_d);
// packed member broken on large and small bar setup.
#if ENABLE_PACKED_TEST
ResultValidation();
#endif
#endif
}
SECTION("Passing struct with aligned(4) to 4Bytes") {
@@ -736,14 +671,12 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_t8B hipLaunchKernelStruct_h8B;
hipLaunchKernelStruct_h8B.c1 = 'c';
hipLaunchKernelStruct_h8B.si = 1;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8B),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h8B,
result_d);
// alignment is broken hence disabled the validation part
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc8B), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h8B, result_d);
// alignment is broken hence disabled the validation part
#if ENABLE_ALIGNMENT_TEST_SMALL_BAR
ResultValidation();
#endif
#endif
}
SECTION("Passing const struct object") {
@@ -754,13 +687,11 @@ TEST_CASE("Unit_hipLaunchParm") {
HIP_CHECK(hipMemset(ip_d9, 1, sizeof(uint32_t)));
// ip_d9 passed as pointer to struct member, struct.ip = &ip_d9
const hipLaunchKernelStruct_t9 hipLaunchKernelStruct_h9 = {'c', ip_d9};
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc9),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h9,
result_d);
#if ENABLE_DECLARE_INITIALIZATION_POINTER
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc9), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h9, result_d);
#if ENABLE_DECLARE_INITIALIZATION_POINTER
ResultValidation();
#endif
#endif
HIP_CHECK(hipFree(reinterpret_cast<void*>(ip_d9)));
}
@@ -770,10 +701,8 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_h10.u64 = UINT64_MAX;
hipLaunchKernelStruct_h10.u32 = 1;
hipLaunchKernelStruct_h10.u8 = UINT8_MAX;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc10),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h10,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc10), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h10, result_d);
ResultValidation();
}
@@ -782,10 +711,8 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_t11 hipLaunchKernelStruct_h11;
hipLaunchKernelStruct_h11.i1 = 1;
hipLaunchKernelStruct_h11.vint = 0;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc11),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h11,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc11), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h11, result_d);
ResultValidation();
}
@@ -793,24 +720,20 @@ TEST_CASE("Unit_hipLaunchParm") {
ResetValidationMem();
hipLaunchKernelStruct_t12 hipLaunchKernelStruct_h12;
hipLaunchKernelStruct_h12.c1 = 'c';
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc12),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h12,
result_d);
#if ENABLE_CLASS_OBJ_ACCESS // access class obj from device broken
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc12), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h12, result_d);
#if ENABLE_CLASS_OBJ_ACCESS // access class obj from device broken
// Validation part of the struct, hipLaunchKernelStructFunc12
ResultValidation();
#endif
#endif
}
SECTION("Passing struct with simple __device__ func()") {
ResetValidationMem();
hipLaunchKernelStruct_t13 hipLaunchKernelStruct_h13;
hipLaunchKernelStruct_h13.i1 = 1;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc13),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h13,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc13), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h13, result_d);
ResultValidation();
}
@@ -818,10 +741,8 @@ TEST_CASE("Unit_hipLaunchParm") {
ResetValidationMem();
hipLaunchKernelStruct_t14 hipLaunchKernelStruct_h14;
hipLaunchKernelStruct_h14.readint = 1;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc14),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h14,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc14), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h14, result_d);
ResultValidation();
}
@@ -830,16 +751,12 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_t15 hipLaunchKernelStruct_h15;
hipLaunchKernelStruct_h15.c1 = 'c';
#if ENABLE_HEAP_MEMORY_ACCESS // causing page fault here,
// on small bar set
HIP_CHECK(hipMalloc(&hipLaunchKernelStruct_h15.heapmem,
BLOCK_DIM_SIZE*sizeof(int)));
HIP_CHECK(hipMemset(&hipLaunchKernelStruct_h15.heapmem,
0, BLOCK_DIM_SIZE));
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc15),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h15,
result_d);
#if ENABLE_HEAP_MEMORY_ACCESS // causing page fault here,
// on small bar set
HIP_CHECK(hipMalloc(&hipLaunchKernelStruct_h15.heapmem, BLOCK_DIM_SIZE * sizeof(int)));
HIP_CHECK(hipMemset(&hipLaunchKernelStruct_h15.heapmem, 0, BLOCK_DIM_SIZE));
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc15), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h15, result_d);
ResultValidation();
HIP_CHECK(hipFree(reinterpret_cast<void*>(hipLaunchKernelStruct_h15.heapmem)));
#endif
@@ -849,10 +766,8 @@ TEST_CASE("Unit_hipLaunchParm") {
ResetValidationMem();
hipLaunchKernelStruct_t16<char> hipLaunchKernelStruct_h16;
hipLaunchKernelStruct_h16.t1 = 'c';
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc16),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h16,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc16), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h16, result_d);
ResultValidation();
}
@@ -860,20 +775,16 @@ TEST_CASE("Unit_hipLaunchParm") {
ResetValidationMem();
hipLaunchKernelStruct_t17<int> hipLaunchKernelStruct_h17;
hipLaunchKernelStruct_h17.t1 = 1;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc17),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h17,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc17), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h17, result_d);
ResultValidation();
}
SECTION("Passing struct with simple __device__ func()") {
ResetValidationMem();
hipLaunchKernelStruct_t18 hipLaunchKernelStruct_h18;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc18),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h18,
result_d);
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc18), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h18, result_d);
ResultValidation();
}
@@ -886,26 +797,24 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelStruct_h20.rank = 2;
hipLaunchKernelStruct_h20.age = 42;
bool *result_d20, *result_h20;
#if ENABLE_OUT_OF_ORDER_INITIALIZATION
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc20),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h20, result_d);
#if ENABLE_OUT_OF_ORDER_INITIALIZATION
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc20), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h20, result_d);
ResultValidation();
#endif
#endif
}
SECTION("Passing struct with bit fields operation") {
ResetValidationMem();
hipLaunchKernelStruct_t21 hipLaunchKernelStruct_h21 =
// out of order initalization
{2, 0};
// out of order initalization
{2, 0};
bool *result_d21, *result_h21;
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc21),
dim3(BLOCK_DIM_SIZE),
dim3(1), 0, 0, hipLaunchKernelStruct_h21, result_d);
#if ENABLE_BIT_FIELDS
hipLaunchKernelGGL(HIP_KERNEL_NAME(hipLaunchKernelStructFunc21), dim3(BLOCK_DIM_SIZE), dim3(1),
0, 0, hipLaunchKernelStruct_h21, result_d);
#if ENABLE_BIT_FIELDS
ResultValidation();
#endif
#endif
}
SECTION("Passing the different hipLaunchParm options") {
@@ -917,42 +826,36 @@ TEST_CASE("Unit_hipLaunchParm") {
hipLaunchKernelGGL(HIP_KERNEL_NAME(vAdd), dim3(1024), dim3(1), 0, 0, Ad);
// Test: Passing macro to hipLaunchKernelGGL
#define KERNEL_CONFIG dim3(1024), dim3(1), 0, 0
#define KERNEL_CONFIG dim3(1024), dim3(1), 0, 0
hipLaunchKernelGGL(HIP_KERNEL_NAME(vAdd), KERNEL_CONFIG, Ad);
// Test: Same thing with templates:
int a;
float b;
hipLaunchKernelGGL(HIP_KERNEL_NAME(myKernel<int, float>),
KERNEL_CONFIG, a, b);
hipLaunchKernelGGL(HIP_KERNEL_NAME(myKernel<int, float>), KERNEL_CONFIG, a, b);
#define TYPE_PARAM_CONFIG int, float
hipLaunchKernelGGL(HIP_KERNEL_NAME(myKernel<TYPE_PARAM_CONFIG>),
KERNEL_CONFIG, a, b);
hipLaunchKernelGGL(HIP_KERNEL_NAME(myKernel<TYPE_PARAM_CONFIG>), KERNEL_CONFIG, a, b);
// Test: Passing hipLaunchKernelGGL inside another macro:
float e0;
MY_LAUNCH_MACRO(hipLaunchKernelGGL(vAdd, dim3(1024),
dim3(1), 0, 0, Ad), e0, j);
MY_LAUNCH_MACRO(WRAP(hipLaunchKernelGGL(vAdd, dim3(1024),
dim3(1), 0, 0, Ad)), e0, j);
MY_LAUNCH_MACRO(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad), e0, j);
MY_LAUNCH_MACRO(WRAP(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad)), e0, j);
#ifdef EXTRA_PARENS_1
// Don't wrap hipLaunchKernelGGL in extra set of parens:
MY_LAUNCH_MACRO((hipLaunchKernelGGL(vAdd, dim3(1024),
dim3(1), 0, 0, Ad)), e0, j);
MY_LAUNCH_MACRO((hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad)), e0, j);
#endif
MY_LAUNCH(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1),
0, 0, Ad), true, "firstCall");
MY_LAUNCH(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad), true, "firstCall");
float* A;
float e1;
MY_LAUNCH_WITH_PAREN(static_cast<void>(hipMalloc(&A, 100)), true, "launch2");
#ifdef EXTRA_PARENS_2
// MY_LAUNCH_WITH_PAREN wraps cmd in () which can cause issues.
MY_LAUNCH_WITH_PAREN(hipLaunchKernelGGL(vAdd, dim3(1024),
dim3(1), 0, 0, Ad), true, "firstCall");
MY_LAUNCH_WITH_PAREN(hipLaunchKernelGGL(vAdd, dim3(1024), dim3(1), 0, 0, Ad), true,
"firstCall");
#endif
HIP_CHECK(hipFree(reinterpret_cast<void*>(A)));
HIP_CHECK(hipFree(reinterpret_cast<void*>(Ad)));
@@ -962,6 +865,6 @@ TEST_CASE("Unit_hipLaunchParm") {
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+128 -167
Dosyayı Görüntüle
@@ -20,34 +20,34 @@ THE SOFTWARE.
#include <hip_test_kernels.hh>
#include <hip_test_checkers.hh>
#include <hip_test_common.hh>
class HipFunctorTests {
public:
// Test that a class functor can be passed to hiplaunchparam
// and can be used in kernel
void TestForSimpleClassFunctor(void);
// Test that a templated class functor can be passed to hiplaunchparam
// and can be used in kernel
void TestForClassTemplateFunctor(void);
// Test that a class functor object ptr can be passed to hiplaunchparam
// and can be used in kernel
void TestForClassObjPtrFunctor(void);
// Test that a class object containing functor can be passed
// to hiplaunchparam and can be used in kernel
void TestForFunctorContainInClassObj(void);
// Test that a stuct functor can be passed to hiplaunchparam
// and can be used in kernel
void TestForSimpleStructFunctor(void);
// Test that a stuct functor object ptr can be passed to hiplaunchparam
// and can be used in kernel
void TestForStructObjPtrFunctor(void);
// Test that a templated struct functor can be passed to hiplaunchparam
// and can be used in kernel
void TestForStructTemplateFunctor(void);
// Test that a struct object containing functor can be
// passed to hiplaunchparam and can be used in kernel
void TestForFunctorContainInStructObj(void);
// Test that a class functor can be passed to hiplaunchparam
// and can be used in kernel
void TestForSimpleClassFunctor(void);
// Test that a templated class functor can be passed to hiplaunchparam
// and can be used in kernel
void TestForClassTemplateFunctor(void);
// Test that a class functor object ptr can be passed to hiplaunchparam
// and can be used in kernel
void TestForClassObjPtrFunctor(void);
// Test that a class object containing functor can be passed
// to hiplaunchparam and can be used in kernel
void TestForFunctorContainInClassObj(void);
// Test that a stuct functor can be passed to hiplaunchparam
// and can be used in kernel
void TestForSimpleStructFunctor(void);
// Test that a stuct functor object ptr can be passed to hiplaunchparam
// and can be used in kernel
void TestForStructObjPtrFunctor(void);
// Test that a templated struct functor can be passed to hiplaunchparam
// and can be used in kernel
void TestForStructTemplateFunctor(void);
// Test that a struct object containing functor can be
// passed to hiplaunchparam and can be used in kernel
void TestForFunctorContainInStructObj(void);
};
static const int BLOCK_DIM_SIZE = 1024;
@@ -56,15 +56,13 @@ static const int THREADS_PER_BLOCK = 1;
// class functor tests
// Simple doubler Functor
class DoublerFunctor{
class DoublerFunctor {
public:
__device__ int operator()(int x) { return x * 2;}
__device__ int operator()(int x) { return x * 2; }
};
// simple doubler functor passed to kernel
__global__ void DoublerFunctorKernel(
DoublerFunctor doubler_,
bool* deviceResult) {
__global__ void DoublerFunctorKernel(DoublerFunctor doubler_, bool* deviceResult) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
int result = doubler_(5);
deviceResult[x] = (result == 10);
@@ -73,32 +71,29 @@ __global__ void DoublerFunctorKernel(
void HipFunctorTests::TestForSimpleClassFunctor(void) {
DoublerFunctor doubler;
bool *deviceResults, *hostResults;
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
// initialize to false, will be set to
// true if the functor is called in device code
hostResults[k] = false;
}
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyHostToDevice));
hipLaunchKernelGGL(DoublerFunctorKernel, dim3(BLOCK_DIM_SIZE),
dim3(THREADS_PER_BLOCK), 0, 0, doubler, deviceResults);
HIP_CHECK(
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
hipLaunchKernelGGL(DoublerFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0, 0,
doubler, deviceResults);
// Validation part of TestForSimpleClassFunctor
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
REQUIRE(hostResults[k] == true);
HIP_CHECK(
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
HIP_CHECK(hipHostFree(hostResults));
HIP_CHECK(hipFree(deviceResults));
}
// pointer functor passed to kernel
__global__ void PtrDoublerFunctorKernel(
DoublerFunctor *doubler_,
bool* deviceResult) {
__global__ void PtrDoublerFunctorKernel(DoublerFunctor* doubler_, bool* deviceResult) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
int result = (*doubler_)(5);
deviceResult[x] = (result == 10);
@@ -107,24 +102,23 @@ __global__ void PtrDoublerFunctorKernel(
void HipFunctorTests::TestForClassObjPtrFunctor(void) {
DoublerFunctor* ptrdoubler = new DoublerFunctor[sizeof(int)];
bool *deviceResults, *hostResults;
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
// initialize to false, will be set to
// true if the functor is called in device code
hostResults[k] = false;
}
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyHostToDevice));
hipLaunchKernelGGL(PtrDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE),
dim3(THREADS_PER_BLOCK), 0, 0, ptrdoubler, deviceResults);
HIP_CHECK(
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
hipLaunchKernelGGL(PtrDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0, 0,
ptrdoubler, deviceResults);
// Validation part of TestForClassObjPtrFunctor
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
REQUIRE(hostResults[k] == true);
HIP_CHECK(
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
HIP_CHECK(hipHostFree(hostResults));
HIP_CHECK(hipFree(deviceResults));
delete[] ptrdoubler;
@@ -132,16 +126,13 @@ void HipFunctorTests::TestForClassObjPtrFunctor(void) {
class compare {
public:
template<typename T1, typename T2>
__device__ bool operator()(const T1& v1, const T2& v2) {
return v1 > v2;
}
template <typename T1, typename T2> __device__ bool operator()(const T1& v1, const T2& v2) {
return v1 > v2;
}
};
// template functor passed to kernel
__global__ void TemplateFunctorKernel(
compare compare_,
bool* deviceResult) {
__global__ void TemplateFunctorKernel(compare compare_, bool* deviceResult) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
deviceResult[x] = compare_(2.2, 2.1);
deviceResult[x] = compare_(2, 1);
@@ -151,24 +142,23 @@ __global__ void TemplateFunctorKernel(
void HipFunctorTests::TestForClassTemplateFunctor(void) {
compare comparefunctor;
bool *deviceResults, *hostResults;
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
// initialize to false, will be set to
// true if the functor is called in device code
hostResults[k] = false;
}
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyHostToDevice));
hipLaunchKernelGGL(TemplateFunctorKernel, dim3(BLOCK_DIM_SIZE),
dim3(THREADS_PER_BLOCK), 0, 0, comparefunctor, deviceResults);
HIP_CHECK(
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
hipLaunchKernelGGL(TemplateFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0, 0,
comparefunctor, deviceResults);
// Validation part of TestForClassTemplateFunctor
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
REQUIRE(hostResults[k] == true);
HIP_CHECK(
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
HIP_CHECK(hipHostFree(hostResults));
HIP_CHECK(hipFree(deviceResults));
}
@@ -177,15 +167,13 @@ void HipFunctorTests::TestForClassTemplateFunctor(void) {
// Doubler calculator
class DoublerCalculator {
public:
int a, result;
// fucntor contained in class object
DoublerFunctor doubler;
int a, result;
// fucntor contained in class object
DoublerFunctor doubler;
};
// doubler functor conatined in class obj passed to kernel
__global__ void DoublerCalculatorFunctorKernel(
DoublerCalculator doubler_,
bool* deviceResult) {
__global__ void DoublerCalculatorFunctorKernel(DoublerCalculator doubler_, bool* deviceResult) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
int result = doubler_.doubler(doubler_.a);
deviceResult[x] = (doubler_.result == result);
@@ -194,8 +182,8 @@ __global__ void DoublerCalculatorFunctorKernel(
void HipFunctorTests::TestForFunctorContainInClassObj(void) {
DoublerCalculator Doubler;
bool *deviceResults, *hostResults;
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
// initialize to false, will be set to
// true if the functor is called in device code
@@ -206,16 +194,15 @@ void HipFunctorTests::TestForFunctorContainInClassObj(void) {
Doubler.result = 10;
// pass comparefunctor to hipLaunchParm
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyHostToDevice));
hipLaunchKernelGGL(DoublerCalculatorFunctorKernel, dim3(BLOCK_DIM_SIZE),
dim3(THREADS_PER_BLOCK), 0, 0, Doubler, deviceResults);
HIP_CHECK(
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
hipLaunchKernelGGL(DoublerCalculatorFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK),
0, 0, Doubler, deviceResults);
// Validation part of TestForStructTemplateFunctor
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
REQUIRE(hostResults[k] == true);
HIP_CHECK(
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
HIP_CHECK(hipHostFree(hostResults));
HIP_CHECK(hipFree(deviceResults));
}
@@ -225,14 +212,12 @@ void HipFunctorTests::TestForFunctorContainInClassObj(void) {
// Simple doubler Functor
struct sDoublerFunctor {
public:
__device__ int operator()(int x) { return x * 2;}
__device__ int operator()(int x) { return x * 2; }
};
// simple sturct doubler functor passed to kernel
__global__ void structDoublerFunctorKernel(
sDoublerFunctor doubler_,
bool* deviceResult) {
__global__ void structDoublerFunctorKernel(sDoublerFunctor doubler_, bool* deviceResult) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
int result = doubler_(5);
deviceResult[x] = (result == 10);
@@ -241,32 +226,29 @@ __global__ void structDoublerFunctorKernel(
void HipFunctorTests::TestForSimpleStructFunctor(void) {
sDoublerFunctor doubler;
bool *deviceResults, *hostResults;
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
// initialize to false, will be set to
// true if the functor is called in device code
hostResults[k] = false;
}
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyHostToDevice));
hipLaunchKernelGGL(structDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE),
dim3(THREADS_PER_BLOCK), 0, 0, doubler, deviceResults);
HIP_CHECK(
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
hipLaunchKernelGGL(structDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0,
0, doubler, deviceResults);
// Validation part of TestForSimpleStructFunctor
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
REQUIRE(hostResults[k] == true);
HIP_CHECK(
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
HIP_CHECK(hipHostFree(hostResults));
HIP_CHECK(hipFree(deviceResults));
}
// ptr functor passed to kernel
__global__ void structPtrDoublerFunctorKernel(
sDoublerFunctor *doubler_,
bool* deviceResult) {
__global__ void structPtrDoublerFunctorKernel(sDoublerFunctor* doubler_, bool* deviceResult) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
int result = (*doubler_)(5);
deviceResult[x] = (result == 10);
@@ -275,24 +257,23 @@ __global__ void structPtrDoublerFunctorKernel(
void HipFunctorTests::TestForStructObjPtrFunctor(void) {
sDoublerFunctor* ptrdoubler = new sDoublerFunctor[sizeof(int)];
bool *deviceResults, *hostResults;
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
// initialize to false, will be set to
// true if the functor is called in device code
hostResults[k] = false;
}
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyHostToDevice));
hipLaunchKernelGGL(structPtrDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE),
dim3(THREADS_PER_BLOCK), 0, 0, ptrdoubler, deviceResults);
HIP_CHECK(
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
hipLaunchKernelGGL(structPtrDoublerFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK),
0, 0, ptrdoubler, deviceResults);
// Validation part of TestForStructObjPtrFunctor
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
REQUIRE(hostResults[k] == true);
HIP_CHECK(
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
HIP_CHECK(hipHostFree(hostResults));
HIP_CHECK(hipFree(deviceResults));
delete[] ptrdoubler;
@@ -300,16 +281,13 @@ void HipFunctorTests::TestForStructObjPtrFunctor(void) {
struct sCompare {
public:
template< typename T1, typename T2 >
__device__ bool operator()(const T1& v1, const T2& v2) {
template <typename T1, typename T2> __device__ bool operator()(const T1& v1, const T2& v2) {
return v1 > v2;
}
}
};
// template functor passed to kernel
__global__ void structTemplateFunctorKernel(
sCompare compare_,
bool* deviceResult) {
__global__ void structTemplateFunctorKernel(sCompare compare_, bool* deviceResult) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
deviceResult[x] = compare_(2.2, 2.1);
deviceResult[x] = compare_(2, 1);
@@ -319,26 +297,25 @@ __global__ void structTemplateFunctorKernel(
void HipFunctorTests::TestForStructTemplateFunctor(void) {
sCompare comparefunctor;
bool *deviceResults, *hostResults;
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
// initialize to false, will be set to
// true if the functor is called in device code
hostResults[k] = false;
}
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyHostToDevice));
HIP_CHECK(
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
// pass comparefunctor to hipLaunchKernelGGL
hipLaunchKernelGGL(structTemplateFunctorKernel, dim3(BLOCK_DIM_SIZE),
dim3(THREADS_PER_BLOCK), 0, 0, comparefunctor, deviceResults);
hipLaunchKernelGGL(structTemplateFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK), 0,
0, comparefunctor, deviceResults);
// Validation part of TestForStructTemplateFunctor
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
REQUIRE(hostResults[k] == true);
HIP_CHECK(
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
HIP_CHECK(hipHostFree(hostResults));
HIP_CHECK(hipFree(deviceResults));
}
@@ -346,17 +323,14 @@ void HipFunctorTests::TestForStructTemplateFunctor(void) {
// Doubler calculator struct
struct sDoublerCalculator {
public:
int a, result;
// fucntor contained in class object
DoublerFunctor doubler;
int a, result;
// fucntor contained in class object
DoublerFunctor doubler;
};
// doubler functor contained in struct passed to kernel
__global__ void DoublerCalculatorFunctorKernel(
sDoublerCalculator doubler_,
bool* deviceResult) {
__global__ void DoublerCalculatorFunctorKernel(sDoublerCalculator doubler_, bool* deviceResult) {
int x = blockIdx.x * blockDim.x + threadIdx.x;
int result = doubler_.doubler(doubler_.a);
deviceResult[x] = (doubler_.result == result);
@@ -365,8 +339,8 @@ __global__ void DoublerCalculatorFunctorKernel(
void HipFunctorTests::TestForFunctorContainInStructObj(void) {
sDoublerCalculator Doubler;
bool *deviceResults, *hostResults;
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE*sizeof(bool)));
HIP_CHECK(hipMalloc(&deviceResults, BLOCK_DIM_SIZE * sizeof(bool)));
HIP_CHECK(hipHostMalloc(&hostResults, BLOCK_DIM_SIZE * sizeof(bool)));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) {
// initialize to false, will be set to
// true if the functor is called in device code
@@ -375,19 +349,18 @@ void HipFunctorTests::TestForFunctorContainInStructObj(void) {
Doubler.a = 5;
Doubler.result = 10;
HIP_CHECK(hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyHostToDevice));
HIP_CHECK(
hipMemcpy(deviceResults, hostResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyHostToDevice));
// pass comparefunctor to hipLaunchKernelGGL
hipLaunchKernelGGL(DoublerCalculatorFunctorKernel, dim3(BLOCK_DIM_SIZE),
dim3(THREADS_PER_BLOCK), 0, 0, Doubler, deviceResults);
hipLaunchKernelGGL(DoublerCalculatorFunctorKernel, dim3(BLOCK_DIM_SIZE), dim3(THREADS_PER_BLOCK),
0, 0, Doubler, deviceResults);
// Validation part of TestForStructTemplateFunctor
HIP_CHECK(hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE*sizeof(bool),
hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k)
REQUIRE(hostResults[k] == true);
HIP_CHECK(
hipMemcpy(hostResults, deviceResults, BLOCK_DIM_SIZE * sizeof(bool), hipMemcpyDeviceToHost));
for (int k = 0; k < BLOCK_DIM_SIZE; ++k) REQUIRE(hostResults[k] == true);
HIP_CHECK(hipHostFree(hostResults));
HIP_CHECK(hipFree(deviceResults));
}
@@ -432,24 +405,12 @@ void HipFunctorTests::TestForFunctorContainInStructObj(void) {
TEST_CASE("Unit_hipLaunchParmFunctor") {
HipFunctorTests FunctorTests;
SECTION("test for simple class functor") {
FunctorTests.TestForSimpleClassFunctor();
}
SECTION("test for class objptr functor") {
FunctorTests.TestForClassObjPtrFunctor();
}
SECTION("test for class templete functor") {
FunctorTests.TestForClassTemplateFunctor();
}
SECTION("test for simple struct functor") {
FunctorTests.TestForSimpleStructFunctor();
}
SECTION("test for struct objptr functor") {
FunctorTests.TestForStructObjPtrFunctor();
}
SECTION("test for struct templete functor") {
FunctorTests.TestForStructTemplateFunctor();
}
SECTION("test for simple class functor") { FunctorTests.TestForSimpleClassFunctor(); }
SECTION("test for class objptr functor") { FunctorTests.TestForClassObjPtrFunctor(); }
SECTION("test for class templete functor") { FunctorTests.TestForClassTemplateFunctor(); }
SECTION("test for simple struct functor") { FunctorTests.TestForSimpleStructFunctor(); }
SECTION("test for struct objptr functor") { FunctorTests.TestForStructObjPtrFunctor(); }
SECTION("test for struct templete functor") { FunctorTests.TestForStructTemplateFunctor(); }
SECTION("test for functor contain in classobj") {
FunctorTests.TestForFunctorContainInClassObj();
}
@@ -459,6 +420,6 @@ TEST_CASE("Unit_hipLaunchParmFunctor") {
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+7 -10
Dosyayı Görüntüle
@@ -37,7 +37,7 @@ __global__ void MyKernelConstSize(int* C_d, const int* A_d) {
}
for (size_t i = 0; i < N; ++i) {
C_d[i] = A_d[i] + A1[i%A1size];
C_d[i] = A_d[i] + A1[i % A1size];
}
}
@@ -50,13 +50,13 @@ __global__ void MyKernelVariableSize(int* C_d, const int* A_d) {
}
for (size_t i = 0; i < N; ++i) {
C_d[i] = A_d[i] + A1[i%A1size];
C_d[i] = A_d[i] + A1[i % A1size];
}
}
static bool verify(const int* C_d, const int* A_d) {
for (size_t i = 0; i < N; i++) {
if (C_d[i] != A_d[i] + i%1024) {
if (C_d[i] != A_d[i] + i % 1024) {
return false;
}
}
@@ -68,7 +68,7 @@ TEST_CASE("Unit_hipMemFaultStackAllocation_Check") {
int *A_d, *C_d;
const size_t Nbytes = N * sizeof(int);
const unsigned threadsPerBlock = 256;
const unsigned blocks = (N + threadsPerBlock - 1)/threadsPerBlock;
const unsigned blocks = (N + threadsPerBlock - 1) / threadsPerBlock;
HIP_CHECK(hipMallocManaged(&A_d, Nbytes));
REQUIRE(A_d != nullptr);
@@ -76,20 +76,18 @@ TEST_CASE("Unit_hipMemFaultStackAllocation_Check") {
REQUIRE(C_d != nullptr);
for (size_t i = 0; i < N; i++) {
A_d[i] = i%1024;
A_d[i] = i % 1024;
}
SECTION("Calling Kernel which allocate ConstSize to local array") {
hipLaunchKernelGGL(MyKernelConstSize, dim3(blocks),
dim3(threadsPerBlock), 0, 0, C_d, A_d);
hipLaunchKernelGGL(MyKernelConstSize, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d);
ret = hipGetLastError();
REQUIRE(hipSuccess == ret);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(true == verify(C_d, A_d));
}
SECTION("Calling Kernel which allocate VariableSize to local array") {
hipLaunchKernelGGL(MyKernelVariableSize, dim3(blocks),
dim3(threadsPerBlock), 0, 0, C_d, A_d);
hipLaunchKernelGGL(MyKernelVariableSize, dim3(blocks), dim3(threadsPerBlock), 0, 0, C_d, A_d);
ret = hipGetLastError();
REQUIRE(hipSuccess == ret);
HIP_CHECK(hipDeviceSynchronize());
@@ -99,4 +97,3 @@ TEST_CASE("Unit_hipMemFaultStackAllocation_Check") {
HIP_CHECK(hipFree(C_d));
HIP_CHECK(hipFree(A_d));
}
+10 -11
Dosyayı Görüntüle
@@ -17,15 +17,13 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <cstring>
#include "../kernel/printf_common.h"
#define HIP_ENABLE_PRINTF
__global__ void run_printf() {
printf("Hello World\n");
}
__global__ void run_printf() { printf("Hello World\n"); }
/**
* @addtogroup hipLaunchKernelGGL
* @{
@@ -52,21 +50,22 @@ TEST_CASE("Unit_kernel_ChkPrintf") {
CaptureStream capture(stdout);
HIP_CHECK(hipGetDeviceCount(&device_count));
std::string st = "Hello World";
const char * check = st.c_str();
const char* check = st.c_str();
for (int i = 0; i < device_count; ++i) {
HIP_CHECK(hipSetDevice(i));
hipLaunchKernelGGL(run_printf, dim3(1), dim3(1), 0, 0);
HIP_CHECK(hipDeviceSynchronize());
char* data = new char[st.size()];;
char* data = new char[st.size()];
;
std::ifstream CapturedData = capture.getCapturedData();
CapturedData.getline(data, st.size()+1);
CapturedData.getline(data, st.size() + 1);
int result = strcmp(data, check);
REQUIRE(result == 0);
delete [] data;
delete[] data;
}
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+2 -2
Dosyayı Görüntüle
@@ -41,8 +41,8 @@ TEST_CASE("Unit_hipSetupArgument_Simple") {
* Test Description
* ------------------------
* - Verifies that arguments sent to the kernel with hipSetupArgument are correct by executing
* kernel that calculates sum of two vectors, doing the same calculation on CPU and checking if the
* results are the same, which proves that the arguments used in kernel are the proper ones
* kernel that calculates sum of two vectors, doing the same calculation on CPU and checking if
* the results are the same, which proves that the arguments used in kernel are the proper ones
*
* Test source
* ------------------------
+7 -8
Dosyayı Görüntüle
@@ -17,7 +17,7 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#define LEN 512
#define SIZE 2048
@@ -61,21 +61,20 @@ TEST_CASE("Unit_kernel_chkConstantViaKernel") {
HIP_CHECK(hipMalloc(reinterpret_cast<void**>(&Ad), SIZE));
HIP_CHECK(hipMemcpyToSymbol(HIP_SYMBOL(Value), A, SIZE, 0,
hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpyToSymbol(HIP_SYMBOL(Value), A, SIZE, 0, hipMemcpyHostToDevice));
hipLaunchKernelGGL(Get, dim3(1, 1, 1), dim3(LEN, 1, 1), 0, 0, Ad);
HIP_CHECK(hipMemcpy(B, Ad, SIZE, hipMemcpyDeviceToHost));
for (unsigned i = 0; i < LEN; i++) {
REQUIRE(A[i] == B[i]);
}
delete [] A;
delete [] B;
delete[] A;
delete[] B;
HIP_CHECK(hipFree(Ad));
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+6 -6
Dosyayı Görüntüle
@@ -17,7 +17,7 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#define LEN 512
#define SIZE 2048
@@ -62,7 +62,7 @@ void runTestConstantGlobalVar() {
for (unsigned i = 0; i < LEN; i++) {
REQUIRE(123 == A[i]);
}
delete [] A;
delete[] A;
HIP_CHECK(hipFree(Ad));
}
@@ -91,7 +91,7 @@ void runTestGlobalArray() {
for (unsigned i = 0; i < LEN; i++) {
REQUIRE(i == A[i]);
}
delete [] A;
delete[] A;
HIP_CHECK(hipFree(Ad));
}
@@ -101,6 +101,6 @@ TEST_CASE("Unit_kernel_chkGlobalArrAndGlobalVaribleViaKernelFn") {
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+27 -27
Dosyayı Görüntüle
@@ -17,7 +17,7 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#define LEN8 8 * 4
#define LEN9 9 * 4
@@ -26,53 +26,53 @@ THE SOFTWARE.
#define LEN12 12 * 4
__global__ void MemCpy8(uint8_t* In, uint8_t* Out) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 8, In + tid * 8, 8);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 8, In + tid * 8, 8);
}
__global__ void MemCpy9(uint8_t* In, uint8_t* Out) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 9, In + tid * 9, 9);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 9, In + tid * 9, 9);
}
__global__ void MemCpy10(uint8_t* In, uint8_t* Out) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 10, In + tid * 10, 10);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 10, In + tid * 10, 10);
}
__global__ void MemCpy11(uint8_t* In, uint8_t* Out) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 11, In + tid * 11, 11);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 11, In + tid * 11, 11);
}
__global__ void MemCpy12(uint8_t* In, uint8_t* Out) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 12, In + tid * 12, 12);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memcpy(Out + tid * 12, In + tid * 12, 12);
}
__global__ void MemSet8(uint8_t* In) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 8, 1, 8);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 8, 1, 8);
}
__global__ void MemSet9(uint8_t* In) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 9, 1, 9);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 9, 1, 9);
}
__global__ void MemSet10(uint8_t* In) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 10, 1, 10);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 10, 1, 10);
}
__global__ void MemSet11(uint8_t* In) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 11, 1, 11);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 11, 1, 11);
}
__global__ void MemSet12(uint8_t* In) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 12, 1, 12);
int tid = threadIdx.x + blockIdx.x * blockDim.x;
memset(In + tid * 12, 1, 12);
}
/**
* @addtogroup hipLaunchKernelGGL
@@ -161,9 +161,9 @@ TEST_CASE("Unit_kernel_MemoryOperationsViaKernels") {
B = new uint8_t[LEN10];
C = new uint8_t[LEN10];
for (uint32_t i = 0; i < LEN10; i++) {
A[i] = i;
B[i] = 0;
C[i] = 0;
A[i] = i;
B[i] = 0;
C[i] = 0;
}
HIP_CHECK(hipMalloc(&Ad, LEN10));
HIP_CHECK(hipMalloc(&Bd, LEN10));
@@ -248,6 +248,6 @@ TEST_CASE("Unit_kernel_MemoryOperationsViaKernels") {
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+10 -13
Dosyayı Görüntüle
@@ -17,19 +17,17 @@ OUT OF OR INN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
constexpr size_t N = 1024;
int p_blockSize = 256;
__global__ void
__launch_bounds__(256, 2)
myKern(int* C, const int* A, int N) {
int tid = (blockIdx.x * blockDim.x + threadIdx.x);
__global__ void __launch_bounds__(256, 2) myKern(int* C, const int* A, int N) {
int tid = (blockIdx.x * blockDim.x + threadIdx.x);
if (tid < N) {
C[tid] = A[tid];
}
if (tid < N) {
C[tid] = A[tid];
}
}
/**
* @addtogroup hipLaunchKernelGGL
@@ -70,8 +68,7 @@ TEST_CASE("Unit_kernel_LaunchBounds_Functional") {
HIPCHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
HIPCHECK(hipGetLastError());
hipLaunchKernelGGL(myKern, dim3(blocks), dim3(p_blockSize), 0,
0, C_d, A_d, N);
hipLaunchKernelGGL(myKern, dim3(blocks), dim3(p_blockSize), 0, 0, C_d, A_d, N);
#ifdef __HIP_PLATFORM_NVIDIA__
cudaFuncAttributes attrib;
@@ -102,6 +99,6 @@ TEST_CASE("Unit_kernel_LaunchBounds_Functional") {
}
/**
* End doxygen group KernelTest.
* @}
*/
* End doxygen group KernelTest.
* @}
*/
+3 -6
Dosyayı Görüntüle
@@ -42,7 +42,7 @@ struct CaptureStream {
char tempname[13] = "mytestXXXXXX";
explicit CaptureStream(FILE *original) {
explicit CaptureStream(FILE* original) {
orig_fd = fileno(original);
saved_fd = dup(orig_fd);
@@ -63,8 +63,7 @@ struct CaptureStream {
}
void restoreStream() {
if (saved_fd == -1)
return;
if (saved_fd == -1) return;
fflush(nullptr);
if (dup2(saved_fd, orig_fd) == -1) {
error(0, errno, "Error");
@@ -77,9 +76,7 @@ struct CaptureStream {
saved_fd = -1;
}
const char *getTempFilename() {
return (const char*)tempname;
}
const char* getTempFilename() { return (const char*)tempname; }
std::ifstream getCapturedData() {
restoreStream();