SWDEV-286322 - clean up trailing space (#2361)
Change-Id: I03c07e67a8d1fa1a874718ffba43eb396c2aa05c
Αυτή η υποβολή περιλαμβάνεται σε:
@@ -55,17 +55,17 @@ __host__ __device__ void testOperations(float &fa, float &fb) {
|
||||
hip_bfloat16 bf_a(fa);
|
||||
hip_bfloat16 bf_b(fb);
|
||||
float fc = float(bf_a);
|
||||
float fd = float(bf_b);
|
||||
float fd = float(bf_b);
|
||||
|
||||
assert(testRelativeAccuracy(fa, bf_a));
|
||||
assert(testRelativeAccuracy(fb, bf_b));
|
||||
|
||||
assert(testRelativeAccuracy(fc + fd, bf_a + bf_b));
|
||||
//when checked as above for add, operation sub fails on GPU
|
||||
//when checked as above for add, operation sub fails on GPU
|
||||
assert(hip_bfloat16(fc - fd) == (bf_a - bf_b));
|
||||
assert(testRelativeAccuracy(fc * fd, bf_a * bf_b));
|
||||
assert(testRelativeAccuracy(fc / fd, bf_a / bf_b));
|
||||
|
||||
|
||||
hip_bfloat16 bf_opNegate = -bf_a;
|
||||
assert(bf_opNegate == -bf_a);
|
||||
|
||||
@@ -75,7 +75,7 @@ __host__ __device__ void testOperations(float &fa, float &fb) {
|
||||
bf_x--;
|
||||
++bf_x;
|
||||
--bf_x;
|
||||
//hip_bfloat16 is converted to float and then inc/decremented, hence check with reduced precision
|
||||
//hip_bfloat16 is converted to float and then inc/decremented, hence check with reduced precision
|
||||
assert(testRelativeAccuracy(bf_x,bf_a));
|
||||
|
||||
bf_x = bf_a;
|
||||
@@ -95,7 +95,7 @@ __host__ __device__ void testOperations(float &fa, float &fb) {
|
||||
if (isnan(bf_rounded)) {
|
||||
assert(isnan(bf_rounded) || isinf(bf_rounded));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void testOperationsGPU(float* d_a, float* d_b)
|
||||
{
|
||||
@@ -126,7 +126,7 @@ int main(){
|
||||
|
||||
hipLaunchKernelGGL(testOperationsGPU, 1, SIZE, 0, 0, d_fa, d_fb);
|
||||
hipDeviceSynchronize();
|
||||
cout<<"Device bfloat16 Operations Successful!!"<<endl;
|
||||
cout<<"Device bfloat16 Operations Successful!!"<<endl;
|
||||
|
||||
delete[] h_fa;
|
||||
delete[] h_fb;
|
||||
|
||||
@@ -56,21 +56,21 @@ __global__ void kernel_lgamma_double(double *input, double *output) {
|
||||
void check_lgamma_double() {
|
||||
|
||||
using datatype_t = double;
|
||||
|
||||
|
||||
const int NUM_INPUTS = 8;
|
||||
auto memsize = NUM_INPUTS * sizeof(datatype_t);
|
||||
|
||||
|
||||
// allocate memories
|
||||
datatype_t *inputCPU = (datatype_t *) malloc(memsize);
|
||||
datatype_t *outputCPU = (datatype_t *) malloc(memsize);
|
||||
datatype_t *inputGPU = nullptr; hipMalloc((void**)&inputGPU, memsize);
|
||||
datatype_t *outputGPU = nullptr; hipMalloc((void**)&outputGPU, memsize);
|
||||
|
||||
|
||||
// populate input
|
||||
for (int i=0; i<NUM_INPUTS; i++) {
|
||||
inputCPU[i] = -3.5 + i;
|
||||
}
|
||||
|
||||
|
||||
// copy inputs to device
|
||||
hipMemcpy(inputGPU, inputCPU, memsize, hipMemcpyHostToDevice);
|
||||
|
||||
@@ -84,13 +84,13 @@ void check_lgamma_double() {
|
||||
for (int i=0; i<NUM_INPUTS; i++) {
|
||||
CHECK_LGAMMA_DOUBLE(inputCPU[i], outputCPU[i], lgamma(inputCPU[i]));
|
||||
}
|
||||
|
||||
|
||||
// free memories
|
||||
hipFree(inputGPU);
|
||||
hipFree(outputGPU);
|
||||
free(inputCPU);
|
||||
free(outputCPU);
|
||||
|
||||
|
||||
// done
|
||||
return;
|
||||
}
|
||||
@@ -102,15 +102,15 @@ void check_abs_int64() {
|
||||
|
||||
const int NUM_INPUTS = 8;
|
||||
auto memsize = NUM_INPUTS * sizeof(datatype_t);
|
||||
|
||||
|
||||
// allocate memories
|
||||
datatype_t *inputCPU = (datatype_t *) malloc(memsize);
|
||||
datatype_t *outputCPU = (datatype_t *) malloc(memsize);
|
||||
datatype_t *inputGPU = nullptr; hipMalloc((void**)&inputGPU, memsize);
|
||||
datatype_t *outputGPU = nullptr; hipMalloc((void**)&outputGPU, memsize);
|
||||
|
||||
|
||||
// populate input
|
||||
inputCPU[0] = -81985529216486895ll;
|
||||
inputCPU[0] = -81985529216486895ll;
|
||||
inputCPU[1] = 81985529216486895ll;
|
||||
inputCPU[2] = -1250999896491ll;
|
||||
inputCPU[3] = 1250999896491ll;
|
||||
@@ -118,7 +118,7 @@ void check_abs_int64() {
|
||||
inputCPU[5] = 19088743ll;
|
||||
inputCPU[6] = -291ll;
|
||||
inputCPU[7] = 291ll;
|
||||
|
||||
|
||||
// copy inputs to device
|
||||
hipMemcpy(inputGPU, inputCPU, memsize, hipMemcpyHostToDevice);
|
||||
|
||||
@@ -137,17 +137,17 @@ void check_abs_int64() {
|
||||
CHECK_ABS_INT64(inputCPU[5], outputCPU[5], outputCPU[5]);
|
||||
CHECK_ABS_INT64(inputCPU[6], outputCPU[6], outputCPU[7]);
|
||||
CHECK_ABS_INT64(inputCPU[7], outputCPU[7], outputCPU[7]);
|
||||
|
||||
|
||||
// free memories
|
||||
hipFree(inputGPU);
|
||||
hipFree(outputGPU);
|
||||
free(inputCPU);
|
||||
free(outputCPU);
|
||||
|
||||
|
||||
// done
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
|
||||
template<class T, class F>
|
||||
__global__ void kernel_simple(F f, T *out) {
|
||||
@@ -191,7 +191,7 @@ int main(int argc, char* argv[]) {
|
||||
check_abs_int64();
|
||||
|
||||
// check_lgamma_double();
|
||||
|
||||
|
||||
test_fp16();
|
||||
|
||||
test_pown();
|
||||
|
||||
@@ -82,7 +82,7 @@ __device__ __host__ complex<FloatT> calc(complex<FloatT> A,
|
||||
return A * B;
|
||||
case CK_div:
|
||||
return A / B;
|
||||
|
||||
|
||||
ONE_ARG(abs)
|
||||
ONE_ARG(arg)
|
||||
ONE_ARG(sin)
|
||||
@@ -111,7 +111,7 @@ void test() {
|
||||
hipMalloc((void**)&Ad, sizeof(ComplexT)*LEN);
|
||||
hipMalloc((void**)&Bd, sizeof(ComplexT)*LEN);
|
||||
hipMalloc((void**)&Cd, sizeof(ComplexT)*LEN);
|
||||
|
||||
|
||||
for (uint32_t i = 0; i < LEN; i++) {
|
||||
A[i] = ComplexT((i + 1) * 1.0f, (i + 2) * 1.0f);
|
||||
B[i] = A[i];
|
||||
@@ -119,7 +119,7 @@ void test() {
|
||||
}
|
||||
hipMemcpy(Ad, A, sizeof(ComplexT)*LEN, hipMemcpyHostToDevice);
|
||||
hipMemcpy(Bd, B, sizeof(ComplexT)*LEN, hipMemcpyHostToDevice);
|
||||
|
||||
|
||||
// Run kernel for a calculation kind and verify by comparing with host
|
||||
// calculation result. Returns false if fails.
|
||||
auto test_fun = [&](enum CalcKind CK) {
|
||||
@@ -145,7 +145,7 @@ void test() {
|
||||
}
|
||||
return true;
|
||||
};
|
||||
|
||||
|
||||
#define OP(x) assert(test_fun(CK_##x));
|
||||
ALL_FUN
|
||||
#undef OP
|
||||
|
||||
@@ -84,7 +84,7 @@ void kernel_hisnan(__half* input, int* output) {
|
||||
}
|
||||
|
||||
__global__
|
||||
void kernel_hisinf(__half* input, int* output) {
|
||||
void kernel_hisinf(__half* input, int* output) {
|
||||
int tx = threadIdx.x;
|
||||
output[tx] = __hisinf(input[tx]);
|
||||
}
|
||||
|
||||
@@ -41,7 +41,7 @@ THE SOFTWARE.
|
||||
private:
|
||||
int a;
|
||||
};
|
||||
|
||||
|
||||
static __global__ void kernel(int* Ad) {
|
||||
int tid = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
new(Ad+tid) A();
|
||||
|
||||
Αναφορά σε νέο ζήτημα
Block a user