SWDEV-286322 - clean up trailing space (#2361)

Change-Id: I03c07e67a8d1fa1a874718ffba43eb396c2aa05c
Αυτή η υποβολή περιλαμβάνεται σε:
Julia Jiang
2021-09-24 06:57:51 -04:00
υποβλήθηκε από GitHub
γονέας abe851ad75
υποβολή 44581b4d3c
14 αρχεία άλλαξαν με 85 προσθήκες και 85 διαγραφές
@@ -55,17 +55,17 @@ __host__ __device__ void testOperations(float &fa, float &fb) {
hip_bfloat16 bf_a(fa);
hip_bfloat16 bf_b(fb);
float fc = float(bf_a);
float fd = float(bf_b);
float fd = float(bf_b);
assert(testRelativeAccuracy(fa, bf_a));
assert(testRelativeAccuracy(fb, bf_b));
assert(testRelativeAccuracy(fc + fd, bf_a + bf_b));
//when checked as above for add, operation sub fails on GPU
//when checked as above for add, operation sub fails on GPU
assert(hip_bfloat16(fc - fd) == (bf_a - bf_b));
assert(testRelativeAccuracy(fc * fd, bf_a * bf_b));
assert(testRelativeAccuracy(fc / fd, bf_a / bf_b));
hip_bfloat16 bf_opNegate = -bf_a;
assert(bf_opNegate == -bf_a);
@@ -75,7 +75,7 @@ __host__ __device__ void testOperations(float &fa, float &fb) {
bf_x--;
++bf_x;
--bf_x;
//hip_bfloat16 is converted to float and then inc/decremented, hence check with reduced precision
//hip_bfloat16 is converted to float and then inc/decremented, hence check with reduced precision
assert(testRelativeAccuracy(bf_x,bf_a));
bf_x = bf_a;
@@ -95,7 +95,7 @@ __host__ __device__ void testOperations(float &fa, float &fb) {
if (isnan(bf_rounded)) {
assert(isnan(bf_rounded) || isinf(bf_rounded));
}
}
}
__global__ void testOperationsGPU(float* d_a, float* d_b)
{
@@ -126,7 +126,7 @@ int main(){
hipLaunchKernelGGL(testOperationsGPU, 1, SIZE, 0, 0, d_fa, d_fb);
hipDeviceSynchronize();
cout<<"Device bfloat16 Operations Successful!!"<<endl;
cout<<"Device bfloat16 Operations Successful!!"<<endl;
delete[] h_fa;
delete[] h_fb;
@@ -56,21 +56,21 @@ __global__ void kernel_lgamma_double(double *input, double *output) {
void check_lgamma_double() {
using datatype_t = double;
const int NUM_INPUTS = 8;
auto memsize = NUM_INPUTS * sizeof(datatype_t);
// allocate memories
datatype_t *inputCPU = (datatype_t *) malloc(memsize);
datatype_t *outputCPU = (datatype_t *) malloc(memsize);
datatype_t *inputGPU = nullptr; hipMalloc((void**)&inputGPU, memsize);
datatype_t *outputGPU = nullptr; hipMalloc((void**)&outputGPU, memsize);
// populate input
for (int i=0; i<NUM_INPUTS; i++) {
inputCPU[i] = -3.5 + i;
}
// copy inputs to device
hipMemcpy(inputGPU, inputCPU, memsize, hipMemcpyHostToDevice);
@@ -84,13 +84,13 @@ void check_lgamma_double() {
for (int i=0; i<NUM_INPUTS; i++) {
CHECK_LGAMMA_DOUBLE(inputCPU[i], outputCPU[i], lgamma(inputCPU[i]));
}
// free memories
hipFree(inputGPU);
hipFree(outputGPU);
free(inputCPU);
free(outputCPU);
// done
return;
}
@@ -102,15 +102,15 @@ void check_abs_int64() {
const int NUM_INPUTS = 8;
auto memsize = NUM_INPUTS * sizeof(datatype_t);
// allocate memories
datatype_t *inputCPU = (datatype_t *) malloc(memsize);
datatype_t *outputCPU = (datatype_t *) malloc(memsize);
datatype_t *inputGPU = nullptr; hipMalloc((void**)&inputGPU, memsize);
datatype_t *outputGPU = nullptr; hipMalloc((void**)&outputGPU, memsize);
// populate input
inputCPU[0] = -81985529216486895ll;
inputCPU[0] = -81985529216486895ll;
inputCPU[1] = 81985529216486895ll;
inputCPU[2] = -1250999896491ll;
inputCPU[3] = 1250999896491ll;
@@ -118,7 +118,7 @@ void check_abs_int64() {
inputCPU[5] = 19088743ll;
inputCPU[6] = -291ll;
inputCPU[7] = 291ll;
// copy inputs to device
hipMemcpy(inputGPU, inputCPU, memsize, hipMemcpyHostToDevice);
@@ -137,17 +137,17 @@ void check_abs_int64() {
CHECK_ABS_INT64(inputCPU[5], outputCPU[5], outputCPU[5]);
CHECK_ABS_INT64(inputCPU[6], outputCPU[6], outputCPU[7]);
CHECK_ABS_INT64(inputCPU[7], outputCPU[7], outputCPU[7]);
// free memories
hipFree(inputGPU);
hipFree(outputGPU);
free(inputCPU);
free(outputCPU);
// done
return;
}
template<class T, class F>
__global__ void kernel_simple(F f, T *out) {
@@ -191,7 +191,7 @@ int main(int argc, char* argv[]) {
check_abs_int64();
// check_lgamma_double();
test_fp16();
test_pown();
@@ -82,7 +82,7 @@ __device__ __host__ complex<FloatT> calc(complex<FloatT> A,
return A * B;
case CK_div:
return A / B;
ONE_ARG(abs)
ONE_ARG(arg)
ONE_ARG(sin)
@@ -111,7 +111,7 @@ void test() {
hipMalloc((void**)&Ad, sizeof(ComplexT)*LEN);
hipMalloc((void**)&Bd, sizeof(ComplexT)*LEN);
hipMalloc((void**)&Cd, sizeof(ComplexT)*LEN);
for (uint32_t i = 0; i < LEN; i++) {
A[i] = ComplexT((i + 1) * 1.0f, (i + 2) * 1.0f);
B[i] = A[i];
@@ -119,7 +119,7 @@ void test() {
}
hipMemcpy(Ad, A, sizeof(ComplexT)*LEN, hipMemcpyHostToDevice);
hipMemcpy(Bd, B, sizeof(ComplexT)*LEN, hipMemcpyHostToDevice);
// Run kernel for a calculation kind and verify by comparing with host
// calculation result. Returns false if fails.
auto test_fun = [&](enum CalcKind CK) {
@@ -145,7 +145,7 @@ void test() {
}
return true;
};
#define OP(x) assert(test_fun(CK_##x));
ALL_FUN
#undef OP
@@ -84,7 +84,7 @@ void kernel_hisnan(__half* input, int* output) {
}
__global__
void kernel_hisinf(__half* input, int* output) {
void kernel_hisinf(__half* input, int* output) {
int tx = threadIdx.x;
output[tx] = __hisinf(input[tx]);
}
@@ -41,7 +41,7 @@ THE SOFTWARE.
private:
int a;
};
static __global__ void kernel(int* Ad) {
int tid = threadIdx.x + blockIdx.x * blockDim.x;
new(Ad+tid) A();