SWDEV-1 - Merge github PRs to amd-staging
Change-Id: I2944a63ddc2eec8dc1403d9790ffffbaec343385
This commit is contained in:
@@ -0,0 +1,164 @@
|
||||
# Copyright (c) 2023 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
#
|
||||
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
# of this software and associated documentation files (the "Software"), to deal
|
||||
# in the Software without restriction, including without limitation the rights
|
||||
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
# copies of the Software, and to permit persons to whom the Software is
|
||||
# furnished to do so, subject to the following conditions:
|
||||
#
|
||||
# The above copyright notice and this permission notice shall be included in
|
||||
# all copies or substantial portions of the Software.
|
||||
#
|
||||
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
# THE SOFTWARE.
|
||||
|
||||
set(TEST_SRC
|
||||
trig_funcs.cc
|
||||
misc_funcs.cc
|
||||
remainder_and_rounding_funcs.cc
|
||||
single_precision_intrinsics.cc
|
||||
double_precision_intrinsics.cc
|
||||
integer_intrinsics.cc
|
||||
root_funcs.cc
|
||||
log_funcs.cc
|
||||
special_funcs.cc
|
||||
casting_double_funcs.cc
|
||||
casting_float_funcs.cc
|
||||
casting_int_funcs.cc
|
||||
casting_half2int_funcs.cc
|
||||
casting_int2half_funcs.cc
|
||||
casting_half_float_funcs.cc
|
||||
)
|
||||
|
||||
if(HIP_PLATFORM MATCHES "nvidia")
|
||||
set(LINKER_LIBS nvrtc)
|
||||
elseif(HIP_PLATFORM MATCHES "amd")
|
||||
set(TEST_SRC ${TEST_SRC}
|
||||
pow_funcs.cc
|
||||
casting_half2_funcs.cc
|
||||
half_precision_math.cc
|
||||
half_precision_arithmetic.cc
|
||||
half_precision_comparison.cc
|
||||
)
|
||||
set(LINKER_LIBS hiprtc)
|
||||
endif()
|
||||
|
||||
find_package(Boost 1.70.0)
|
||||
message(STATUS "Boost_FOUND: ${Boost_FOUND}")
|
||||
if(Boost_FOUND)
|
||||
hip_add_exe_to_target(NAME MathsTest
|
||||
TEST_SRC ${TEST_SRC}
|
||||
TEST_TARGET_NAME build_tests COMMON_SHARED_SRC ${COMMON_SHARED_SRC}
|
||||
LINKER_LIBS ${LINKER_LIBS})
|
||||
target_include_directories(MathsTest PRIVATE ${Boost_INCLUDE_DIRS})
|
||||
else()
|
||||
message(STATUS "Boost not found. Dependent math tests not enabled.")
|
||||
endif()
|
||||
|
||||
# Below tests fail in PSDB
|
||||
#add_test(NAME Unit_Device_Single_Precision_Trig_Functions_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# trig_single_precision_negative_kernels.cc 66)
|
||||
#
|
||||
#add_test(NAME Unit_Device_Double_Precision_Trig_Functions_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# trig_double_precision_negative_kernels.cc 66)
|
||||
#add_test(NAME Unit_Device_Misc_Functions_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# misc_negative_kernels.cc 76)
|
||||
#
|
||||
#add_test(NAME Unit_Device_remainder_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# math_remainder_negative_kernels.cc 68)
|
||||
#
|
||||
#add_test(NAME Unit_Device_rounding_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# math_rounding_negative_kernels.cc 40)
|
||||
#
|
||||
#add_test(NAME Unit_Single_Precision_Intrinsics_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# single_precision_intrinsics_negative_kernels.cc 42)
|
||||
#
|
||||
#add_test(NAME Unit_Double_Precision_Intrinsics_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# double_precision_intrinsics_negative_kernels.cc 18)
|
||||
#
|
||||
#add_test(NAME Unit_Integer_Intrinsics_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# integer_intrinsics_negative_kernels.cc 20)
|
||||
#add_test(NAME Unit_Device_root_1Dand2D_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# math_root_negative_kernels_1Dand2D.cc 68)
|
||||
#
|
||||
#add_test(NAME Unit_Device_root_3Dand4D_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# math_root_negative_kernels_3Dand4D.cc 56)
|
||||
#add_test(NAME Unit_Device_pow_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# math_pow_negative_kernels.cc 76)
|
||||
#add_test(NAME Unit_Device_log_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# math_log_negative_kernels.cc 24)
|
||||
#add_test(NAME Unit_Device_special_funcs_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# math_special_func_kernels.cc 76)
|
||||
#add_test(NAME Unit_Device_casting_double_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# casting_double_negative_kernels.cc 69)
|
||||
#add_test(NAME Unit_Device_casting_float_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# casting_float_negative_kernels.cc 54)
|
||||
#add_test(NAME Unit_Device_casting_int_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# casting_int_negative_kernels.cc 92)
|
||||
#
|
||||
#add_test(NAME Unit_Device_casting_half2_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# casting_half2_negative_kernels.cc 53)
|
||||
#add_test(NAME Unit_Half_Precision_Math_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# half_precision_math_negative_kernels.cc 60)
|
||||
#add_test(NAME Unit_Half_Precision_Arithmetic_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# half_precision_arithmetic_negative_kernels.cc 88)
|
||||
#add_test(NAME Unit_Half_Precision_Comparison_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# half_precision_comparison_negative_kernels.cc 168)
|
||||
#add_test(NAME Unit_Device_casting_half2int_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# casting_half2int_negative_kernels.cc 78)
|
||||
#add_test(NAME Unit_Device_casting_int2half_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# casting_int2half_negative_kernels.cc 78)
|
||||
#add_test(NAME Unit_Device_casting_half_float_Negative
|
||||
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
|
||||
# casting_half_float_negative_kernels.cc 18)
|
||||
@@ -0,0 +1,55 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <hip/hip_fp16.h>
|
||||
|
||||
#define FLOAT16_MAX 65504.0f
|
||||
|
||||
class Float16 {
|
||||
public:
|
||||
__host__ __device__ Float16() = default;
|
||||
__host__ __device__ Float16(__half x) : x_{x} {}
|
||||
__host__ __device__ Float16(__half2 x) : x_{__low2half(x)} {}
|
||||
__host__ __device__ Float16(float x) : x_{__float2half(x)} {}
|
||||
|
||||
// __heq doesn't have a __host__ version
|
||||
__host__ __device__ bool operator==(Float16 other) const { return (static_cast<__half_raw>(x_).x == static_cast<__half_raw>(other.x_).x); }
|
||||
__host__ __device__ bool operator!=(Float16 other) const { return !(*this == other); }
|
||||
|
||||
__host__ __device__ operator __half() const { return x_; }
|
||||
__host__ __device__ operator __half2() const { return __half2half2(x_); }
|
||||
__host__ __device__ operator float() const { return __half2float(x_); }
|
||||
|
||||
private:
|
||||
__half x_;
|
||||
};
|
||||
|
||||
namespace {
|
||||
|
||||
inline std::ostream& operator<<(std::ostream& o, Float16 x) {
|
||||
o << static_cast<float>(x);
|
||||
return o;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -0,0 +1,141 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "math_common.hh"
|
||||
#include "math_special_values.hh"
|
||||
|
||||
#include <hip/hip_cooperative_groups.h>
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
#define MATH_BINARY_KERNEL_DEF(func_name) \
|
||||
template <typename T, typename RT = T> \
|
||||
__global__ void func_name##_kernel(RT* const ys, const size_t num_xs, T* const x1s, \
|
||||
T* const x2s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
if constexpr (std::is_same_v<float, T>) { \
|
||||
ys[i] = func_name##f(x1s[i], x2s[i]); \
|
||||
} else if constexpr (std::is_same_v<double, T>) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i]); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void BinaryFloatingPointBruteForceTest(kernel_sig<T, TArg, TArg> kernel,
|
||||
ref_sig<RT, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder,
|
||||
const TArg a = std::numeric_limits<TArg>::lowest(),
|
||||
const TArg b = std::numeric_limits<TArg>::max()) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const uint64_t num_iterations = GetTestIterationCount();
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(TArg) * 2 + sizeof(T)), num_iterations);
|
||||
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
thread_pool.Post([=, &x1s, &x2s] {
|
||||
const auto generator = [=] {
|
||||
static thread_local std::mt19937 rng(std::random_device{}());
|
||||
if constexpr (std::is_same_v<TArg, Float16>) {
|
||||
std::uniform_real_distribution<RefType_t<Float16>> unif_dist(-FLOAT16_MAX, FLOAT16_MAX);
|
||||
return static_cast<Float16>(unif_dist(rng));
|
||||
} else {
|
||||
std::uniform_real_distribution<RefType_t<TArg>> unif_dist(a, b);
|
||||
return static_cast<TArg>(unif_dist(rng));
|
||||
}
|
||||
};
|
||||
std::generate(x1s.ptr() + base_idx, x1s.ptr() + base_idx + sub_batch_size, generator);
|
||||
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator);
|
||||
});
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
|
||||
x2s.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void BinaryFloatingPointSpecialValuesTest(kernel_sig<T, TArg, TArg> kernel,
|
||||
ref_sig<RT, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
using SpecialValsType = std::conditional_t<std::is_same_v<TArg, Float16>, float, TArg>;
|
||||
const auto values = std::get<SpecialVals<SpecialValsType>>(kSpecialValRegistry);
|
||||
|
||||
const auto size = values.size * values.size;
|
||||
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
|
||||
for (auto i = 0u; i < values.size; ++i) {
|
||||
for (auto j = 0u; j < values.size; ++j) {
|
||||
x1s.ptr()[i * values.size + j] = values.data[i];
|
||||
x2s.ptr()[i * values.size + j] = values.data[j];
|
||||
}
|
||||
}
|
||||
|
||||
MathTest math_test(kernel, size);
|
||||
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, size, x1s.ptr(),
|
||||
x2s.ptr());
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void BinaryFloatingPointTest(kernel_sig<T, TArg, TArg> kernel, ref_sig<RT, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Special values") {
|
||||
BinaryFloatingPointSpecialValuesTest(kernel, ref_func, validator_builder);
|
||||
}
|
||||
|
||||
SECTION("Brute force") { BinaryFloatingPointBruteForceTest(kernel, ref_func, validator_builder); }
|
||||
}
|
||||
|
||||
#define MATH_BINARY_WITHIN_ULP_TEST_DEF(kern_name, ref_func, sp_ulp, dp_ulp) \
|
||||
MATH_BINARY_KERNEL_DEF(kern_name) \
|
||||
\
|
||||
TEMPLATE_TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive", "", float, double) { \
|
||||
using RT = RefType_t<TestType>; \
|
||||
RT (*ref)(RT, RT) = ref_func; \
|
||||
const auto ulp = std::is_same_v<float, TestType> ? sp_ulp : dp_ulp; \
|
||||
\
|
||||
BinaryFloatingPointTest(kern_name##_kernel<TestType>, ref, \
|
||||
ULPValidatorBuilderFactory<TestType>(ulp)); \
|
||||
}
|
||||
@@ -0,0 +1,259 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include <fenv.h>
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
#define CAST_KERNEL_DEF(func_name, T1, T2) \
|
||||
__global__ void func_name##_kernel(T1* const ys, const size_t num_xs, T2* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(xs[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define CAST_BINARY_KERNEL_DEF(func_name, T1, T2) \
|
||||
__global__ void func_name##_kernel(T1* const ys, const size_t num_xs, T2* const x1s, \
|
||||
T2* const x2s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define CAST_F2I_REF_DEF(func_name, T1, T2, ref_func) \
|
||||
T1 func_name##_ref(T2 arg) { \
|
||||
if (arg >= static_cast<T2>(std::numeric_limits<T1>::max())) \
|
||||
return std::numeric_limits<T1>::max(); \
|
||||
else if (arg <= static_cast<T2>(std::numeric_limits<T1>::min())) \
|
||||
return std::numeric_limits<T1>::min(); \
|
||||
T2 result = ref_func(arg); \
|
||||
return result; \
|
||||
}
|
||||
|
||||
#define CAST_F2I_RZ_REF_DEF(func_name, T1, T2) \
|
||||
T1 func_name##_ref(T2 arg) { \
|
||||
if (arg >= static_cast<double>(std::numeric_limits<T1>::max())) \
|
||||
return std::numeric_limits<T1>::max(); \
|
||||
else if (arg <= static_cast<double>(std::numeric_limits<T1>::min())) \
|
||||
return std::numeric_limits<T1>::min(); \
|
||||
T1 result = static_cast<T1>(arg); \
|
||||
return result; \
|
||||
}
|
||||
|
||||
#define CAST_RND_REF_DEF(func_name, T1, T2, round_dir) \
|
||||
T1 func_name##_ref(T2 arg) { \
|
||||
int curr_direction = fegetround(); \
|
||||
fesetround(round_dir); \
|
||||
T1 result = static_cast<T1>(arg); \
|
||||
fesetround(curr_direction); \
|
||||
return result; \
|
||||
}
|
||||
|
||||
#define CAST_REF_DEF(func_name, T1, T2) \
|
||||
T1 func_name##_ref(T2 arg) { \
|
||||
T1 result = static_cast<T1>(arg); \
|
||||
return result; \
|
||||
}
|
||||
|
||||
template <typename T1, typename T2> T1 type2_as_type1_ref(T2 arg) {
|
||||
T1 tmp;
|
||||
memcpy(&tmp, &arg, sizeof(tmp));
|
||||
return tmp;
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void CastUnaryHalfPrecisionBruteForceTest(kernel_sig<T, Float16> kernel,
|
||||
ref_sig<RT, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
uint64_t stop = std::numeric_limits<uint16_t>::max() + 1ul;
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(Float16) + sizeof(T)), stop);
|
||||
LinearAllocGuard<Float16> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(Float16)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
|
||||
for (uint64_t v = 0u; v < stop;) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, stop - v);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
|
||||
thread_pool.Post([=, &values] {
|
||||
auto t = v;
|
||||
uint16_t val;
|
||||
for (auto j = 0u; j < sub_batch_size; ++j) {
|
||||
val = static_cast<uint16_t>(t++);
|
||||
values.ptr()[base_idx + j] = *reinterpret_cast<Float16*>(&val);
|
||||
if (std::isnan(values.ptr()[base_idx + j]) || std::isinf(values.ptr()[base_idx + j])) {
|
||||
values.ptr()[base_idx + j] = 0;
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
v += sub_batch_size;
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void CastUnaryHalfPrecisionTest(kernel_sig<T, Float16> kernel, ref_sig<RT, RTArg> ref,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Brute force") { CastUnaryHalfPrecisionBruteForceTest(kernel, ref, validator_builder); }
|
||||
}
|
||||
|
||||
|
||||
template <typename T, typename ValidatorBuilder>
|
||||
void CastDoublePrecisionSpecialValuesTest(kernel_sig<T, double> kernel, ref_sig<T, double> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const auto values = std::get<SpecialVals<double>>(kSpecialValRegistry);
|
||||
std::vector<double> spec_values;
|
||||
|
||||
if (!std::is_same_v<float, T> && !std::is_same_v<double, T> && !std::is_same_v<long double, T>) {
|
||||
for (int i = 0; i < values.size; i++) {
|
||||
if (!std::isnan(values.data[i]) && !std::isinf(values.data[i])) {
|
||||
spec_values.push_back(values.data[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MathTest math_test(kernel, spec_values.size());
|
||||
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func,
|
||||
spec_values.size(), spec_values.data());
|
||||
}
|
||||
|
||||
template <typename T, typename ValidatorBuilder>
|
||||
void CastDoublePrecisionTest(kernel_sig<T, double> kernel, ref_sig<T, double> ref,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Special values") {
|
||||
CastDoublePrecisionSpecialValuesTest(kernel, ref, validator_builder);
|
||||
}
|
||||
|
||||
SECTION("Brute force") { UnaryDoublePrecisionBruteForceTest(kernel, ref, validator_builder); }
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void CastIntRangeTest(kernel_sig<T, TArg> kernel, ref_sig<RT, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder,
|
||||
const TArg a = std::numeric_limits<TArg>::lowest(),
|
||||
const TArg b = std::numeric_limits<TArg>::max()) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const auto max_batch_size = GetMaxAllowedDeviceMemoryUsage() / (sizeof(T) + sizeof(TArg));
|
||||
LinearAllocGuard<TArg> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
size_t inserted = 0u;
|
||||
for (TArg v = a; v <= b; v++) {
|
||||
values.ptr()[inserted++] = v;
|
||||
if (inserted < max_batch_size) continue;
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, inserted, values.ptr());
|
||||
inserted = 0u;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void CastIntBruteForceTest(kernel_sig<T, TArg> kernel, ref_sig<RT, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder,
|
||||
const TArg a = std::numeric_limits<TArg>::lowest(),
|
||||
const TArg b = std::numeric_limits<TArg>::max()) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const uint64_t num_iterations = GetTestIterationCount();
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(T) + sizeof(TArg)), num_iterations);
|
||||
LinearAllocGuard<TArg> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
thread_pool.Post([=, &values] {
|
||||
const auto generator = [=] {
|
||||
static thread_local std::mt19937 rng(std::random_device{}());
|
||||
std::uniform_int_distribution<TArg> unif_dist(a, b);
|
||||
return static_cast<TArg>(unif_dist(rng));
|
||||
};
|
||||
std::generate(values.ptr() + base_idx, values.ptr() + base_idx + sub_batch_size, generator);
|
||||
});
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T1, typename T2, typename ValidatorBuilder>
|
||||
void CastBinaryIntRangeTest(kernel_sig<T1, T2, T2> kernel, ref_sig<T1, T2, T2> ref_func,
|
||||
const ValidatorBuilder& validator_builder,
|
||||
const T2 a = std::numeric_limits<T2>::lowest(),
|
||||
const T2 b = std::numeric_limits<T2>::max()) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const auto max_batch_size = GetMaxAllowedDeviceMemoryUsage() / (sizeof(T1) + 2 * sizeof(T2));
|
||||
LinearAllocGuard<T2> values1{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(T2)};
|
||||
LinearAllocGuard<T2> values2{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(T2)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
size_t inserted = 0u;
|
||||
for (T2 v = a; v <= b; v++) {
|
||||
values1.ptr()[inserted] = v;
|
||||
values2.ptr()[inserted++] = b - v;
|
||||
if (inserted < max_batch_size) continue;
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, inserted, values1.ptr(),
|
||||
values2.ptr());
|
||||
inserted = 0u;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,597 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "casting_common.hh"
|
||||
#include "casting_double_negative_kernels_rtc.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup CastingDoubleType CastingDoubleType
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
#define CAST_DOUBLE2INT_TEST_DEF(kern_name, T, ref_func) \
|
||||
CAST_KERNEL_DEF(kern_name, T, double) \
|
||||
CAST_F2I_REF_DEF(kern_name, T, double, ref_func) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T (*ref)(double) = kern_name##_ref; \
|
||||
CastDoublePrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>()); \
|
||||
}
|
||||
|
||||
#define CAST_DOUBLE2INT_RZ_TEST_DEF(kern_name, T) \
|
||||
CAST_KERNEL_DEF(kern_name, T, double) \
|
||||
CAST_F2I_RZ_REF_DEF(kern_name, T, double) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T (*ref)(double) = kern_name##_ref; \
|
||||
CastDoublePrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2int_rd` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::floor`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2INT_TEST_DEF(__double2int_rd, int, std::floor)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2int_rn` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::rint`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2INT_TEST_DEF(__double2int_rn, int, std::rint)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2int_ru` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::ceil`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2INT_TEST_DEF(__double2int_ru, int, std::ceil)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2int_rz` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* performs cast to int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2INT_RZ_TEST_DEF(__double2int_rz, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __double2int_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2int_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2Int); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2uint_rd` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* `std::floor`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2INT_TEST_DEF(__double2uint_rd, unsigned int, std::floor)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2uint_rn` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* `std::rint`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2INT_TEST_DEF(__double2uint_rn, unsigned int, std::rint)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2uint_ru` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* `std::ceil`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2INT_TEST_DEF(__double2uint_ru, unsigned int, std::ceil)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2uint_rz` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* which performs cast to unsigned int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2INT_RZ_TEST_DEF(__double2uint_rz, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __double2uint_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2uint_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2Uint); }
|
||||
|
||||
#define CAST_DOUBLE2LL_TEST_DEF(kern_name, T, ref_func) \
|
||||
CAST_KERNEL_DEF(kern_name, T, double) \
|
||||
CAST_F2I_REF_DEF(kern_name, T, double, ref_func) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T (*ref)(double) = kern_name##_ref; \
|
||||
UnaryDoublePrecisionBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
|
||||
static_cast<double>(std::numeric_limits<T>::min()), \
|
||||
static_cast<double>(std::numeric_limits<T>::max())); \
|
||||
}
|
||||
|
||||
#define CAST_DOUBLE2LL_RZ_TEST_DEF(kern_name, T) \
|
||||
CAST_KERNEL_DEF(kern_name, T, double) \
|
||||
CAST_F2I_RZ_REF_DEF(kern_name, T, double) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T (*ref)(double) = kern_name##_ref; \
|
||||
UnaryDoublePrecisionBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
|
||||
static_cast<double>(std::numeric_limits<T>::min()), \
|
||||
static_cast<double>(std::numeric_limits<T>::max())); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2ll_rd` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::floor`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2LL_TEST_DEF(__double2ll_rd, long long int, std::floor)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2ll_rn` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::rint`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2LL_TEST_DEF(__double2ll_rn, long long int, std::rint)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2ll_ru` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::ceil`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2LL_TEST_DEF(__double2ll_ru, long long int, std::ceil)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2ll_rz` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* performs cast to long long int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2LL_RZ_TEST_DEF(__double2ll_rz, long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __double2ll_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2ll_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2LL); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2ull_rd` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::floor`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2LL_TEST_DEF(__double2ull_rd, unsigned long long int, std::floor)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2ull_rn` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::rint`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2LL_TEST_DEF(__double2ull_rn, unsigned long long int, std::rint)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2ull_ru` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function
|
||||
* `std::ceil`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2LL_TEST_DEF(__double2ull_ru, unsigned long long int, std::ceil)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2ull_rz` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* performs cast to unsigned long long int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2LL_RZ_TEST_DEF(__double2ull_rz, unsigned long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __double2ull_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2ull_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2ULL); }
|
||||
|
||||
#define CAST_DOUBLE2FLOAT_TEST_DEF(kern_name, round_dir) \
|
||||
CAST_KERNEL_DEF(kern_name, float, double) \
|
||||
CAST_RND_REF_DEF(kern_name, float, double, round_dir) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
float (*ref)(double) = kern_name##_ref; \
|
||||
CastDoublePrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<float>()); \
|
||||
}
|
||||
|
||||
#define CAST_DOUBLE2FLOAT_RN_TEST_DEF(kern_name) \
|
||||
CAST_KERNEL_DEF(kern_name, float, double) \
|
||||
CAST_REF_DEF(kern_name, float, double) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
float (*ref)(double) = kern_name##_ref; \
|
||||
CastDoublePrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<float>()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2float_rd` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* which performs cast to float with rounding mode FE_DOWNWARD.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2FLOAT_TEST_DEF(__double2float_rd, FE_DOWNWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2float_rn` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* which performs cast to float.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2FLOAT_RN_TEST_DEF(__double2float_rn)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2float_ru` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* which performs cast to float with rounding mode FE_UPWARD.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2FLOAT_TEST_DEF(__double2float_ru, FE_UPWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2float_rz` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* which performs cast to float with rounding mode FE_TOWARDZERO.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_DOUBLE2FLOAT_TEST_DEF(__double2float_rz, FE_TOWARDZERO)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __double2float_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2float_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2Float); }
|
||||
|
||||
CAST_KERNEL_DEF(__double2hiint, int, double)
|
||||
|
||||
int __double2hiint_ref(double arg) {
|
||||
int tmp[2];
|
||||
memcpy(tmp, &arg, sizeof(tmp));
|
||||
return tmp[1];
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2hiint` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* performs copy of higher part of double value to int variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2hiint_Positive") {
|
||||
int (*ref)(double) = __double2hiint_ref;
|
||||
CastDoublePrecisionTest(__double2hiint_kernel, ref, EqValidatorBuilderFactory<int>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __double2hiint.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2hiint_Negative_RTC") { NegativeTestRTCWrapper<3>(kDouble2Hiint); }
|
||||
|
||||
CAST_KERNEL_DEF(__double2loint, int, double)
|
||||
|
||||
int __double2loint_ref(double arg) {
|
||||
int tmp[2];
|
||||
memcpy(tmp, &arg, sizeof(tmp));
|
||||
return tmp[0];
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double2loint` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* performs copy of lower part of double value to int variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2loint_Positive") {
|
||||
int (*ref)(double) = __double2loint_ref;
|
||||
CastDoublePrecisionTest(__double2loint_kernel, ref, EqValidatorBuilderFactory<int>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __double2loint.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double2loint_Negative_RTC") { NegativeTestRTCWrapper<3>(kDouble2Loint); }
|
||||
|
||||
CAST_KERNEL_DEF(__double_as_longlong, long long int, double)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__double_as_longlong` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* which performs copy of double value to long long int variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double_as_longlong_Positive") {
|
||||
long long int (*ref)(double) = type2_as_type1_ref<long long int, double>;
|
||||
CastDoublePrecisionTest(__double_as_longlong_kernel, ref,
|
||||
EqValidatorBuilderFactory<long long int>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __double_as_longlong.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_double_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___double_as_longlong_Negative_RTC") {
|
||||
NegativeTestRTCWrapper<3>(kDoubleAsLonglong);
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL(func_name, T) \
|
||||
__global__ void func_name##_kernel_v1(T* result, double* x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(T* result, Dummy x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy* result, double x) { *result = func_name(x); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL(__double2int_rd, int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2int_rn, int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2int_ru, int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2int_rz, int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2uint_rd, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2uint_rn, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2uint_ru, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2uint_rz, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2ll_rd, long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2ll_rn, long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2ll_ru, long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2ll_rz, long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2ull_rd, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2ull_rn, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2ull_ru, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2ull_rz, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2float_rd, float)
|
||||
NEGATIVE_KERNELS_SHELL(__double2float_rn, float)
|
||||
NEGATIVE_KERNELS_SHELL(__double2float_ru, float)
|
||||
NEGATIVE_KERNELS_SHELL(__double2float_rz, float)
|
||||
NEGATIVE_KERNELS_SHELL(__double2hiint, int)
|
||||
NEGATIVE_KERNELS_SHELL(__double2loint, int)
|
||||
NEGATIVE_KERNELS_SHELL(__double_as_longlong, long long int)
|
||||
@@ -0,0 +1,157 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
/*
|
||||
Negative kernels used for the double type casting negative Test Cases that are using RTC.
|
||||
*/
|
||||
|
||||
static constexpr auto kDouble2Int{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void double2int_rd_kernel_v1(int* result, double* x) { *result = __double2int_rd(x); }
|
||||
__global__ void double2int_rd_kernel_v2(int* result, Dummy x) { *result = __double2int_rd(x); }
|
||||
__global__ void double2int_rd_kernel_v3(Dummy* result, double x) { *result = __double2int_rd(x); }
|
||||
__global__ void double2int_rn_kernel_v1(int* result, double* x) { *result = __double2int_rn(x); }
|
||||
__global__ void double2int_rn_kernel_v2(int* result, Dummy x) { *result = __double2int_rn(x); }
|
||||
__global__ void double2int_rn_kernel_v3(Dummy* result, double x) { *result = __double2int_rn(x); }
|
||||
__global__ void double2int_ru_kernel_v1(int* result, double* x) { *result = __double2int_ru(x); }
|
||||
__global__ void double2int_ru_kernel_v2(int* result, Dummy x) { *result = __double2int_ru(x); }
|
||||
__global__ void double2int_ru_kernel_v3(Dummy* result, double x) { *result = __double2int_ru(x); }
|
||||
__global__ void double2int_rz_kernel_v1(int* result, double* x) { *result = __double2int_rz(x); }
|
||||
__global__ void double2int_rz_kernel_v2(int* result, Dummy x) { *result = __double2int_rz(x); }
|
||||
__global__ void double2int_rz_kernel_v3(Dummy* result, double x) { *result = __double2int_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kDouble2Uint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void double2uint_rd_kernel_v1(unsigned int* result, double* x) { *result = __double2uint_rd(x); }
|
||||
__global__ void double2uint_rd_kernel_v2(unsigned int* result, Dummy x) { *result = __double2uint_rd(x); }
|
||||
__global__ void double2uint_rd_kernel_v3(Dummy* result, double x) { *result = __double2uint_rd(x); }
|
||||
__global__ void double2uint_rn_kernel_v1(unsigned int* result, double* x) { *result = __double2uint_rn(x); }
|
||||
__global__ void double2uint_rn_kernel_v2(unsigned int* result, Dummy x) { *result = __double2uint_rn(x); }
|
||||
__global__ void double2uint_rn_kernel_v3(Dummy* result, double x) { *result = __double2uint_rn(x); }
|
||||
__global__ void double2uint_ru_kernel_v1(unsigned int* result, double* x) { *result = __double2uint_ru(x); }
|
||||
__global__ void double2uint_ru_kernel_v2(unsigned int* result, Dummy x) { *result = __double2uint_ru(x); }
|
||||
__global__ void double2uint_ru_kernel_v3(Dummy* result, double x) { *result = __double2uint_ru(x); }
|
||||
__global__ void double2uint_rz_kernel_v1(unsigned int* result, double* x) { *result = __double2uint_rz(x); }
|
||||
__global__ void double2uint_rz_kernel_v2(unsigned int* result, Dummy x) { *result = __double2uint_rz(x); }
|
||||
__global__ void double2uint_rz_kernel_v3(Dummy* result, double x) { *result = __double2uint_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kDouble2LL{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void double2ll_rd_kernel_v1(long long int* result, double* x) { *result = __double2ll_rd(x); }
|
||||
__global__ void double2ll_rd_kernel_v2(long long int* result, Dummy x) { *result = __double2ll_rd(x); }
|
||||
__global__ void double2ll_rd_kernel_v3(Dummy* result, double x) { *result = __double2ll_rd(x); }
|
||||
__global__ void double2ll_rn_kernel_v1(long long int* result, double* x) { *result = __double2ll_rn(x); }
|
||||
__global__ void double2ll_rn_kernel_v2(long long int* result, Dummy x) { *result = __double2ll_rn(x); }
|
||||
__global__ void double2ll_rn_kernel_v3(Dummy* result, double x) { *result = __double2ll_rn(x); }
|
||||
__global__ void double2ll_ru_kernel_v1(long long int* result, double* x) { *result = __double2ll_ru(x); }
|
||||
__global__ void double2ll_ru_kernel_v2(long long int* result, Dummy x) { *result = __double2ll_ru(x); }
|
||||
__global__ void double2ll_ru_kernel_v3(Dummy* result, double x) { *result = __double2ll_ru(x); }
|
||||
__global__ void double2ll_rz_kernel_v1(long long int* result, double* x) { *result = __double2ll_rz(x); }
|
||||
__global__ void double2ll_rz_kernel_v2(long long int* result, Dummy x) { *result = __double2ll_rz(x); }
|
||||
__global__ void double2ll_rz_kernel_v3(Dummy* result, double x) { *result = __double2ll_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kDouble2ULL{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void double2ull_rd_kernel_v1(unsigned long long int* result, double* x) { *result = __double2ull_rd(x); }
|
||||
__global__ void double2ull_rd_kernel_v2(unsigned long long int* result, Dummy x) { *result = __double2ull_rd(x); }
|
||||
__global__ void double2ull_rd_kernel_v3(Dummy* result, double x) { *result = __double2ull_rd(x); }
|
||||
__global__ void double2ull_rn_kernel_v1(unsigned long long int* result, double* x) { *result = __double2ull_rn(x); }
|
||||
__global__ void double2ull_rn_kernel_v2(unsigned long long int* result, Dummy x) { *result = __double2ull_rn(x); }
|
||||
__global__ void double2ull_rn_kernel_v3(Dummy* result, double x) { *result = __double2ull_rn(x); }
|
||||
__global__ void double2ull_ru_kernel_v1(unsigned long long int* result, double* x) { *result = __double2ull_ru(x); }
|
||||
__global__ void double2ull_ru_kernel_v2(unsigned long long int* result, Dummy x) { *result = __double2ull_ru(x); }
|
||||
__global__ void double2ull_ru_kernel_v3(Dummy* result, double x) { *result = __double2ull_ru(x); }
|
||||
__global__ void double2ull_rz_kernel_v1(unsigned long long int* result, double* x) { *result = __double2ull_rz(x); }
|
||||
__global__ void double2ull_rz_kernel_v2(unsigned long long int* result, Dummy x) { *result = __double2ull_rz(x); }
|
||||
__global__ void double2ull_rz_kernel_v3(Dummy* result, double x) { *result = __double2ull_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kDouble2Float{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void double2float_rd_kernel_v1(float* result, double* x) { *result = __double2float_rd(x); }
|
||||
__global__ void double2float_rd_kernel_v2(float* result, Dummy x) { *result = __double2float_rd(x); }
|
||||
__global__ void double2float_rd_kernel_v3(Dummy* result, double x) { *result = __double2float_rd(x); }
|
||||
__global__ void double2float_rn_kernel_v1(float* result, double* x) { *result = __double2float_rn(x); }
|
||||
__global__ void double2float_rn_kernel_v2(float* result, Dummy x) { *result = __double2float_rn(x); }
|
||||
__global__ void double2float_rn_kernel_v3(Dummy* result, double x) { *result = __double2float_rn(x); }
|
||||
__global__ void double2float_ru_kernel_v1(float* result, double* x) { *result = __double2float_ru(x); }
|
||||
__global__ void double2float_ru_kernel_v2(float* result, Dummy x) { *result = __double2float_ru(x); }
|
||||
__global__ void double2float_ru_kernel_v3(Dummy* result, double x) { *result = __double2float_ru(x); }
|
||||
__global__ void double2float_rz_kernel_v1(float* result, double* x) { *result = __double2float_rz(x); }
|
||||
__global__ void double2float_rz_kernel_v2(float* result, Dummy x) { *result = __double2float_rz(x); }
|
||||
__global__ void double2float_rz_kernel_v3(Dummy* result, double x) { *result = __double2float_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kDouble2Hiint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void double2hiint_kernel_v1(int* result, double* x) { *result = __double2hiint(x); }
|
||||
__global__ void double2hiint_kernel_v2(int* result, Dummy x) { *result = __double2hiint(x); }
|
||||
__global__ void double2hiint_kernel_v3(Dummy* result, double x) { *result = __double2hiint(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kDouble2Loint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void double2loint_kernel_v1(int* result, double* x) { *result = __double2loint(x); }
|
||||
__global__ void double2loint_kernel_v2(int* result, Dummy x) { *result = __double2loint(x); }
|
||||
__global__ void double2loint_kernel_v3(Dummy* result, double x) { *result = __double2loint(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kDoubleAsLonglong{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void double_as_longlong_kernel_v1(long long int* result, double* x) { *result = __double_as_longlong(x); }
|
||||
__global__ void double_as_longlong_kernel_v2(long long int* result, Dummy x) { *result = __double_as_longlong(x); }
|
||||
__global__ void double_as_longlong_kernel_v3(Dummy* result, double x) { *result = __double_as_longlong(x); }
|
||||
)"};
|
||||
@@ -0,0 +1,440 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "casting_common.hh"
|
||||
#include "casting_float_negative_kernels_rtc.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup CastingFloatType CastingFloatType
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
#define CAST_FLOAT2INT_TEST_DEF(kern_name, T, ref_func) \
|
||||
CAST_KERNEL_DEF(kern_name, T, float) \
|
||||
CAST_F2I_REF_DEF(kern_name, T, float, ref_func) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T (*ref)(float) = kern_name##_ref; \
|
||||
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
|
||||
std::numeric_limits<float>::lowest(), \
|
||||
std::numeric_limits<float>::max()); \
|
||||
}
|
||||
|
||||
#define CAST_FLOAT2INT_RZ_TEST_DEF(kern_name, T) \
|
||||
CAST_KERNEL_DEF(kern_name, T, float) \
|
||||
CAST_F2I_RZ_REF_DEF(kern_name, T, float) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T (*ref)(float) = kern_name##_ref; \
|
||||
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
|
||||
std::numeric_limits<float>::lowest(), \
|
||||
std::numeric_limits<float>::max()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2int_rd` for all possible inputs. The results are compared against
|
||||
* reference function `std::floor`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2INT_TEST_DEF(__float2int_rd, int, std::floor)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2int_rn` for all possible inputs. The results are compared against
|
||||
* reference function `std::rint`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2INT_TEST_DEF(__float2int_rn, int, std::rint)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2int_ru` for all possible inputs. The results are compared against
|
||||
* reference function `std::ceil`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2INT_TEST_DEF(__float2int_ru, int, std::ceil)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2int_rz` for all possible inputs. The results are compared against
|
||||
* reference function `std::trunc`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2INT_TEST_DEF(__float2int_rz, int, std::trunc)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __float2int_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float2int_Negative_RTC") { NegativeTestRTCWrapper<12>(kFloat2Int); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2uint_rd` for all possible inputs. The results are compared
|
||||
* against reference function `std::floor`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2INT_TEST_DEF(__float2uint_rd, unsigned int, std::floor)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2uint_rn` for all possible inputs. The results are compared
|
||||
* against reference function `std::rint`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2INT_TEST_DEF(__float2uint_rn, unsigned int, std::rint)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2uint_ru` for all possible inputs. The results are compared
|
||||
* against reference function `std::ceil`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2INT_TEST_DEF(__float2uint_ru, unsigned int, std::ceil)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2uint_rz` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* performs cast to unsigned int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2INT_RZ_TEST_DEF(__float2uint_rz, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __float2uint_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float2uint_Negative_RTC") { NegativeTestRTCWrapper<12>(kFloat2Uint); }
|
||||
|
||||
#define CAST_FLOAT2LL_TEST_DEF(kern_name, T, ref_func) \
|
||||
CAST_KERNEL_DEF(kern_name, T, float) \
|
||||
CAST_F2I_REF_DEF(kern_name, T, float, ref_func) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T (*ref)(float) = kern_name##_ref; \
|
||||
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
|
||||
static_cast<float>(std::numeric_limits<T>::min()), \
|
||||
static_cast<float>(std::numeric_limits<T>::max())); \
|
||||
}
|
||||
|
||||
#define CAST_FLOAT2LL_RZ_TEST_DEF(kern_name, T) \
|
||||
CAST_KERNEL_DEF(kern_name, T, float) \
|
||||
CAST_F2I_RZ_REF_DEF(kern_name, T, float) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T (*ref)(float) = kern_name##_ref; \
|
||||
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
|
||||
static_cast<float>(std::numeric_limits<T>::min()), \
|
||||
static_cast<float>(std::numeric_limits<T>::max())); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2ll_rd` for all possible inputs. The results are compared against
|
||||
* reference function `std::floor`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2LL_TEST_DEF(__float2ll_rd, long long int, std::floor)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2ll_rn` for all possible inputs between lowest and maximal long
|
||||
* long int value. The results are compared against reference function `std::rint`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2LL_TEST_DEF(__float2ll_rn, long long int, std::rint)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2ll_ru` for all possible inputs between lowest and maximal long
|
||||
* long int value. The results are compared against reference function `std::ceil`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2LL_TEST_DEF(__float2ll_ru, long long int, std::ceil)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2ll_rz` for all possible inputs between lowest and maximal long
|
||||
* long int value. The results are compared against reference function which performs cast to long
|
||||
* long int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2LL_RZ_TEST_DEF(__float2ll_rz, long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __float2ll_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float2ll_Negative_RTC") { NegativeTestRTCWrapper<12>(kFloat2LL); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2ull_rd` for all possible inputs between lowest and maximal
|
||||
* unsigned long long int value. The results are compared against reference function `std::floor`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2LL_TEST_DEF(__float2ull_rd, unsigned long long int, std::floor)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2ull_rn` for all possible inputs between lowest and maximal
|
||||
* unsigned long long int value. The results are compared against reference function `std::rint`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2LL_TEST_DEF(__float2ull_rn, unsigned long long int, std::rint)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2ull_ru` for all possible inputs between lowest and maximal
|
||||
* unsigned long long int value. The results are compared against reference function `std::ceil`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2LL_TEST_DEF(__float2ull_ru, unsigned long long int, std::ceil)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2ll_rz` for all possible inputs between lowest and maximal
|
||||
* unsigned long long int value. The results are compared against reference function which performs
|
||||
* cast to unsigned long long int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2LL_RZ_TEST_DEF(__float2ull_rz, unsigned long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __float2ull_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float2ull_Negative_RTC") { NegativeTestRTCWrapper<12>(kFloat2ULL); }
|
||||
|
||||
CAST_KERNEL_DEF(__float_as_int, int, float)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float_as_int` for all possible inputs. The results are compared against
|
||||
* reference function which performs copy of float value to int variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float_as_int_Positive") {
|
||||
int (*ref)(float) = type2_as_type1_ref<int, float>;
|
||||
UnarySinglePrecisionTest(__float_as_int_kernel, ref, EqValidatorBuilderFactory<int>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __float_as_int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float_as_int_Negative_RTC") { NegativeTestRTCWrapper<3>(kFloatAsInt); }
|
||||
|
||||
CAST_KERNEL_DEF(__float_as_uint, unsigned int, float)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float_as_uint` for all possible inputs. The results are compared
|
||||
* against reference function which performs copy of float value to unsigned int variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float_as_uint_Positive") {
|
||||
unsigned int (*ref)(float) = type2_as_type1_ref<unsigned int, float>;
|
||||
UnarySinglePrecisionTest(__float_as_uint_kernel, ref, EqValidatorBuilderFactory<unsigned int>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __float_as_uint.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float_as_uint_Negative_RTC") { NegativeTestRTCWrapper<3>(kFloatAsUint); }
|
||||
@@ -0,0 +1,50 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL(func_name, T) \
|
||||
__global__ void func_name##_kernel_v1(T* result, float* x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(T* result, Dummy x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy* result, float x) { *result = func_name(x); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL(__float2int_rd, int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2int_rn, int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2int_ru, int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2int_rz, int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2uint_rd, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2uint_rn, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2uint_ru, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2uint_rz, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2ll_rd, long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2ll_rn, long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2ll_ru, long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2ll_rz, long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2ull_rd, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2ull_rn, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2ull_ru, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__float2ull_rz, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL(__float_as_int, int)
|
||||
NEGATIVE_KERNELS_SHELL(__float_as_uint, unsigned int)
|
||||
@@ -0,0 +1,126 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
/*
|
||||
Negative kernels used for the float type casting negative Test Cases that are using RTC.
|
||||
*/
|
||||
|
||||
static constexpr auto kFloat2Int{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void float2int_rd_kernel_v1(int* result, float* x) { *result = __float2int_rd(x); }
|
||||
__global__ void float2int_rd_kernel_v2(int* result, Dummy x) { *result = __float2int_rd(x); }
|
||||
__global__ void float2int_rd_kernel_v3(Dummy* result, float x) { *result = __float2int_rd(x); }
|
||||
__global__ void float2int_rn_kernel_v1(int* result, float* x) { *result = __float2int_rn(x); }
|
||||
__global__ void float2int_rn_kernel_v2(int* result, Dummy x) { *result = __float2int_rn(x); }
|
||||
__global__ void float2int_rn_kernel_v3(Dummy* result, float x) { *result = __float2int_rn(x); }
|
||||
__global__ void float2int_ru_kernel_v1(int* result, float* x) { *result = __float2int_ru(x); }
|
||||
__global__ void float2int_ru_kernel_v2(int* result, Dummy x) { *result = __float2int_ru(x); }
|
||||
__global__ void float2int_ru_kernel_v3(Dummy* result, float x) { *result = __float2int_ru(x); }
|
||||
__global__ void float2int_rz_kernel_v1(int* result, float* x) { *result = __float2int_rz(x); }
|
||||
__global__ void float2int_rz_kernel_v2(int* result, Dummy x) { *result = __float2int_rz(x); }
|
||||
__global__ void float2int_rz_kernel_v3(Dummy* result, float x) { *result = __float2int_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFloat2Uint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void float2uint_rd_kernel_v1(unsigned int* result, float* x) { *result = __float2uint_rd(x); }
|
||||
__global__ void float2uint_rd_kernel_v2(unsigned int* result, Dummy x) { *result = __float2uint_rd(x); }
|
||||
__global__ void float2uint_rd_kernel_v3(Dummy* result, float x) { *result = __float2uint_rd(x); }
|
||||
__global__ void float2uint_rn_kernel_v1(unsigned int* result, float* x) { *result = __float2uint_rn(x); }
|
||||
__global__ void float2uint_rn_kernel_v2(unsigned int* result, Dummy x) { *result = __float2uint_rn(x); }
|
||||
__global__ void float2uint_rn_kernel_v3(Dummy* result, float x) { *result = __float2uint_rn(x); }
|
||||
__global__ void float2uint_ru_kernel_v1(unsigned int* result, float* x) { *result = __float2uint_ru(x); }
|
||||
__global__ void float2uint_ru_kernel_v2(unsigned int* result, Dummy x) { *result = __float2uint_ru(x); }
|
||||
__global__ void float2uint_ru_kernel_v3(Dummy* result, float x) { *result = __float2uint_ru(x); }
|
||||
__global__ void float2uint_rz_kernel_v1(unsigned int* result, float* x) { *result = __float2uint_rz(x); }
|
||||
__global__ void float2uint_rz_kernel_v2(unsigned int* result, Dummy x) { *result = __float2uint_rz(x); }
|
||||
__global__ void float2uint_rz_kernel_v3(Dummy* result, float x) { *result = __float2uint_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFloat2LL{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void float2ll_rd_kernel_v1(long long int* result, float* x) { *result = __float2ll_rd(x); }
|
||||
__global__ void float2ll_rd_kernel_v2(long long int* result, Dummy x) { *result = __float2ll_rd(x); }
|
||||
__global__ void float2ll_rd_kernel_v3(Dummy* result, float x) { *result = __float2ll_rd(x); }
|
||||
__global__ void float2ll_rn_kernel_v1(long long int* result, float* x) { *result = __float2ll_rn(x); }
|
||||
__global__ void float2ll_rn_kernel_v2(long long int* result, Dummy x) { *result = __float2ll_rn(x); }
|
||||
__global__ void float2ll_rn_kernel_v3(Dummy* result, float x) { *result = __float2ll_rn(x); }
|
||||
__global__ void float2ll_ru_kernel_v1(long long int* result, float* x) { *result = __float2ll_ru(x); }
|
||||
__global__ void float2ll_ru_kernel_v2(long long int* result, Dummy x) { *result = __float2ll_ru(x); }
|
||||
__global__ void float2ll_ru_kernel_v3(Dummy* result, float x) { *result = __float2ll_ru(x); }
|
||||
__global__ void float2ll_rz_kernel_v1(long long int* result, float* x) { *result = __float2ll_rz(x); }
|
||||
__global__ void float2ll_rz_kernel_v2(long long int* result, Dummy x) { *result = __float2ll_rz(x); }
|
||||
__global__ void float2ll_rz_kernel_v3(Dummy* result, float x) { *result = __float2ll_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFloat2ULL{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void float2ull_rd_kernel_v1(unsigned long long int* result, float* x) { *result = __float2ull_rd(x); }
|
||||
__global__ void float2ull_rd_kernel_v2(unsigned long long int* result, Dummy x) { *result = __float2ull_rd(x); }
|
||||
__global__ void float2ull_rd_kernel_v3(Dummy* result, float x) { *result = __float2ull_rd(x); }
|
||||
__global__ void float2ull_rn_kernel_v1(unsigned long long int* result, float* x) { *result = __float2ull_rn(x); }
|
||||
__global__ void float2ull_rn_kernel_v2(unsigned long long int* result, Dummy x) { *result = __float2ull_rn(x); }
|
||||
__global__ void float2ull_rn_kernel_v3(Dummy* result, float x) { *result = __float2ull_rn(x); }
|
||||
__global__ void float2ull_ru_kernel_v1(unsigned long long int* result, float* x) { *result = __float2ull_ru(x); }
|
||||
__global__ void float2ull_ru_kernel_v2(unsigned long long int* result, Dummy x) { *result = __float2ull_ru(x); }
|
||||
__global__ void float2ull_ru_kernel_v3(Dummy* result, float x) { *result = __float2ull_ru(x); }
|
||||
__global__ void float2ull_rz_kernel_v1(unsigned long long int* result, float* x) { *result = __float2ull_rz(x); }
|
||||
__global__ void float2ull_rz_kernel_v2(unsigned long long int* result, Dummy x) { *result = __float2ull_rz(x); }
|
||||
__global__ void float2ull_rz_kernel_v3(Dummy* result, float x) { *result = __float2ull_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFloatAsInt{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void float_as_int_kernel_v1(int* result, float* x) { *result = __float_as_int(x); }
|
||||
__global__ void float_as_int_kernel_v2(int* result, Dummy x) { *result = __float_as_int(x); }
|
||||
__global__ void float_as_int_kernel_v3(Dummy* result, float x) { *result = __float_as_int(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFloatAsUint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void float_as_uint_kernel_v1(unsigned int* result, float* x) { *result = __float_as_uint(x); }
|
||||
__global__ void float_as_uint_kernel_v2(unsigned int* result, Dummy x) { *result = __float_as_uint(x); }
|
||||
__global__ void float_as_uint_kernel_v3(Dummy* result, float x) { *result = __float_as_uint(x); }
|
||||
)"};
|
||||
@@ -0,0 +1,97 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "math_common.hh"
|
||||
#include "validators.hh"
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
#define CAST_HALF2_KERNEL_DEF(func_name, T) \
|
||||
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, Float16* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(__half2{xs[i], -xs[i]}); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define CAST_BINARY_HALF2_KERNEL_DEF(func_name, T) \
|
||||
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, Float16* const x1s, \
|
||||
Float16* const x2s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(__half2{x1s[i], -x1s[i]}, __half2{x2s[i], -x2s[i]}); \
|
||||
} \
|
||||
}
|
||||
|
||||
template <typename VB> class Float2Validator : public MatcherBase<float2> {
|
||||
public:
|
||||
Float2Validator(const float2& target, const VB& vb)
|
||||
: first_matcher_{vb(target.x)}, second_matcher_{vb(target.y)} {}
|
||||
|
||||
bool match(const float2& val) const override {
|
||||
return first_matcher_->match(val.x) && second_matcher_->match(val.y);
|
||||
}
|
||||
|
||||
std::string describe() const override {
|
||||
return "<" + first_matcher_->describe() + ", " + second_matcher_->describe() + ">";
|
||||
}
|
||||
|
||||
private:
|
||||
decltype(std::declval<VB>()(float())) first_matcher_;
|
||||
decltype(std::declval<VB>()(float())) second_matcher_;
|
||||
};
|
||||
|
||||
template <typename ValidatorBuilder>
|
||||
auto Float2ValidatorBuilderFactory(const ValidatorBuilder& vb) {
|
||||
return [=](const float2& t, auto&&...) {
|
||||
return std::make_unique<Float2Validator<ValidatorBuilder>>(t, vb);
|
||||
};
|
||||
}
|
||||
|
||||
template <typename VB> class Half2Validator : public MatcherBase<__half2> {
|
||||
public:
|
||||
Half2Validator(const __half2& target, const VB& vb)
|
||||
: first_matcher_{vb(target.data.x)}, second_matcher_{vb(target.data.y)} {}
|
||||
|
||||
bool match(const __half2& val) const override {
|
||||
return first_matcher_->match(val.data.x) && second_matcher_->match(val.data.y);
|
||||
}
|
||||
|
||||
std::string describe() const override {
|
||||
return "<" + first_matcher_->describe() + ", " + second_matcher_->describe() + ">";
|
||||
}
|
||||
|
||||
private:
|
||||
decltype(std::declval<VB>()(Float16())) first_matcher_;
|
||||
decltype(std::declval<VB>()(Float16())) second_matcher_;
|
||||
};
|
||||
|
||||
template <typename ValidatorBuilder> auto Half2ValidatorBuilderFactory(const ValidatorBuilder& vb) {
|
||||
return [=](const __half2& t, auto&&...) {
|
||||
return std::make_unique<Half2Validator<ValidatorBuilder>>(t, vb);
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,419 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "half_precision_common.hh"
|
||||
#include "casting_common.hh"
|
||||
#include "casting_half2_common.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup HalfPrecisionCastingHalf2 HalfPrecisionCastingHalf2
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
/********** half -> half2 **********/
|
||||
|
||||
CAST_KERNEL_DEF(__half2half2, __half2, Float16)
|
||||
|
||||
static __half2 __half2half2_ref(Float16 x) { return __half2{x, x}; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2half2` for all possible inputs. The results are compared against
|
||||
* reference function which returns __half2 value created from one __half value.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___half2half2_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__half2half2_kernel, __half2half2_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
CAST_BINARY_KERNEL_DEF(make_half2, __half2, Float16)
|
||||
|
||||
static __half2 make_half2_ref(Float16 x, Float16 y) { return __half2{x, y}; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `make_half2` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* returns __half2 value created from two __half values.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_make_half2_Accuracy_Positive") {
|
||||
BinaryFloatingPointTest(make_half2_kernel, make_half2_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
CAST_BINARY_KERNEL_DEF(__halves2half2, __half2, Float16)
|
||||
|
||||
static __half2 __halves2half2_ref(Float16 x, Float16 y) { return __half2{x, y}; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__halves2half2` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* returns __half2 value created from two __half values.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___halves2half2_Accuracy_Positive") {
|
||||
BinaryFloatingPointTest(__halves2half2_kernel, __halves2half2_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
/********** half2 -> half **********/
|
||||
|
||||
|
||||
CAST_HALF2_KERNEL_DEF(__low2half, Float16)
|
||||
|
||||
static Float16 __low2half_ref(Float16 x) { return x; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__low2half` for all possible inputs. The results are compared against
|
||||
* reference function which returns __half value created from lower __half2 element.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___low2half_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__low2half_kernel, __low2half_ref, EqValidatorBuilderFactory<Float16>());
|
||||
}
|
||||
|
||||
CAST_HALF2_KERNEL_DEF(__high2half, Float16)
|
||||
|
||||
static Float16 __high2half_ref(Float16 x) { return -x; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__high2half` for all possible inputs. The results are compared against
|
||||
* reference function which returns __half value created from higher __half2 element.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___high2half_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__high2half_kernel, __high2half_ref, EqValidatorBuilderFactory<Float16>());
|
||||
}
|
||||
|
||||
/********** half2 -> half2 **********/
|
||||
|
||||
CAST_HALF2_KERNEL_DEF(__low2half2, __half2)
|
||||
|
||||
static __half2 __low2half2_ref(Float16 x) { return __half2{x, x}; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__low2half2` for all possible inputs. The results are compared against
|
||||
* reference function which returns __half2 value created from two lower __half2 elements.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___low2half2_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__low2half2_kernel, __low2half2_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
CAST_HALF2_KERNEL_DEF(__high2half2, __half2)
|
||||
|
||||
static __half2 __high2half2_ref(Float16 x) { return __half2{-x, -x}; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__high2half2` for all possible inputs. The results are compared against
|
||||
* reference function which returns __half2 value created from two higher __half2 elements.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___high2half2_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__high2half2_kernel, __high2half2_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
CAST_HALF2_KERNEL_DEF(__lowhigh2highlow, __half2)
|
||||
|
||||
static __half2 __lowhigh2highlow_ref(Float16 x) { return __half2{-x, x}; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__lowhigh2highlow` for all possible inputs. The results are compared
|
||||
* against reference function which returns __half2 value created from higher and lower __half2
|
||||
* elements.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___lowhigh2highlow_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__lowhigh2highlow_kernel, __lowhigh2highlow_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
CAST_BINARY_HALF2_KERNEL_DEF(__lows2half2, __half2)
|
||||
|
||||
static __half2 __lows2half2_ref(Float16 x, Float16 y) { return __half2{x, y}; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__lows2half2` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* returns __half2 value created from lower elements of two __half2 values.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___lows2half2_Accuracy_Positive") {
|
||||
BinaryFloatingPointTest(__lows2half2_kernel, __lows2half2_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
CAST_BINARY_HALF2_KERNEL_DEF(__highs2half2, __half2)
|
||||
|
||||
static __half2 __highs2half2_ref(Float16 x, Float16 y) { return __half2{-x, -y}; }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__highs2half2` against a table of difficult values, followed by a large
|
||||
* number of randomly generated values. The results are compared against reference function which
|
||||
* returns __half2 value created from higher elements of two __half2 values.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___highs2half2_Accuracy_Positive") {
|
||||
BinaryFloatingPointTest(__highs2half2_kernel, __highs2half2_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
/********** float -> half2 **********/
|
||||
|
||||
CAST_KERNEL_DEF(__float2half2_rn, __half2, float)
|
||||
|
||||
static __half2 __float2half2_rn_ref(float x) {
|
||||
return __half2{static_cast<Float16>(x), static_cast<Float16>(x)};
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2half2_rn` for all possible inputs. The results are compared
|
||||
* against reference function which returns __half2 value created from one casted float value.
|
||||
* elements.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float2half2_rn_Accuracy_Positive") {
|
||||
UnarySinglePrecisionTest(__float2half2_rn_kernel, __float2half2_rn_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
CAST_BINARY_KERNEL_DEF(__floats2half2_rn, __half2, float)
|
||||
|
||||
static __half2 __floats2half2_rn_ref(float x, float y) {
|
||||
return __half2{static_cast<Float16>(x), static_cast<Float16>(y)};
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__floats2half2_rn` against a table of difficult values, followed by a
|
||||
* large number of randomly generated values. The results are compared against reference function
|
||||
* which returns __half2 value created from two casted float values.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___floats2half2_rn_Accuracy_Positive") {
|
||||
BinaryFloatingPointTest(__floats2half2_rn_kernel, __floats2half2_rn_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
/********** float2 -> half2 **********/
|
||||
|
||||
__global__ void __float22half2_rn_kernel(__half2* const ys, const size_t num_xs, float* const xs) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
ys[i] = __float22half2_rn(make_float2(xs[i], -xs[i]));
|
||||
}
|
||||
}
|
||||
|
||||
static __half2 __float22half2_rn_ref(float x) {
|
||||
return __half2{static_cast<Float16>(x), static_cast<Float16>(-x)};
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float22half2_rn` for all possible inputs. The results are compared
|
||||
* against reference function which returns __half2 value created from two casted float values.
|
||||
* elements.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float22half2_rn_Accuracy_Positive") {
|
||||
UnarySinglePrecisionTest(__float22half2_rn_kernel, __float22half2_rn_ref,
|
||||
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
|
||||
}
|
||||
|
||||
/********** half2 -> float **********/
|
||||
|
||||
CAST_HALF2_KERNEL_DEF(__low2float, float)
|
||||
|
||||
static float __low2float_ref(Float16 x) { return static_cast<float>(x); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__low2float` for all possible inputs. The results are compared
|
||||
* against reference function which returns float value created from lower __half2 element.
|
||||
* elements.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___low2float_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__low2float_kernel, __low2float_ref, EqValidatorBuilderFactory<float>());
|
||||
}
|
||||
|
||||
CAST_HALF2_KERNEL_DEF(__high2float, float)
|
||||
|
||||
static float __high2float_ref(Float16 x) { return static_cast<float>(-x); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__high2float` for all possible inputs. The results are compared
|
||||
* against reference function which returns float value created from higher __half2 element.
|
||||
* elements.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___high2float_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__high2float_kernel, __high2float_ref, EqValidatorBuilderFactory<float>());
|
||||
}
|
||||
|
||||
/********** half2 -> float2 **********/
|
||||
|
||||
CAST_HALF2_KERNEL_DEF(__half22float2, float2)
|
||||
|
||||
static float2 __half22float2_ref(Float16 x) {
|
||||
return make_float2(static_cast<float>(x), static_cast<float>(-x));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half22float2` for all possible inputs. The results are compared against
|
||||
* reference function which returns float2 value created from casted elements of one __half2 value.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___half22float2_Accuracy_Positive") {
|
||||
UnaryHalfPrecisionTest(__half22float2_kernel, __half22float2_ref,
|
||||
Float2ValidatorBuilderFactory(EqValidatorBuilderFactory<float>()));
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip/hip_fp16.h>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_UNARY_KERNELS_SHELL(func_name, T1, T2) \
|
||||
__global__ void func_name##_kernel_v1(T1* result, T2* x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(T1* result, Dummy x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy* result, T2 x) { *result = func_name(x); }
|
||||
|
||||
|
||||
#define NEGATIVE_BINARY_KERNELS_SHELL(func_name, T1, T2) \
|
||||
__global__ void func_name##_kernel_v1(T2* x, T2 y) { T1 result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(T2 x, T2* y) { T1 result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, T2 y) { T1 result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(T2 x, Dummy y) { T1 result = func_name(x, y); }
|
||||
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__half2half2, __half2, __half)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__low2half, __half, __half2)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__high2half, __half, __half2)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__low2half2, __half2, __half2)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__high2half2, __half2, __half2)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__lowhigh2highlow, __half2, __half2)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__float2half2_rn, __half2, float)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__float22half2_rn, __half2, float2)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__low2float, float, __half2)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__high2float, float, __half2)
|
||||
NEGATIVE_UNARY_KERNELS_SHELL(__half22float2, float2, __half2)
|
||||
|
||||
NEGATIVE_BINARY_KERNELS_SHELL(make_half2, __half2, __half)
|
||||
NEGATIVE_BINARY_KERNELS_SHELL(__halves2half2, __half2, __half)
|
||||
NEGATIVE_BINARY_KERNELS_SHELL(__lows2half2, __half2, __half2)
|
||||
NEGATIVE_BINARY_KERNELS_SHELL(__highs2half2, __half2, __half2)
|
||||
NEGATIVE_BINARY_KERNELS_SHELL(__floats2half2_rn, __half2, float)
|
||||
@@ -0,0 +1,440 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "half_precision_common.hh"
|
||||
#include "casting_common.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup HalfPrecisionCastingIntTypes HalfPrecisionCastingIntTypes
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
#define CAST_HALF2INT_RN_TEST_DEF(kern_name, T) \
|
||||
CAST_KERNEL_DEF(kern_name, T, Float16) \
|
||||
CAST_F2I_RZ_REF_DEF(kern_name, T, Float16) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive") { \
|
||||
T (*ref)(Float16) = kern_name##_ref; \
|
||||
CastUnaryHalfPrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2int_rn` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2int_rn, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2int_rz` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2int_rz, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2int_rd` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2int_rd, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2int_ru` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2int_ru, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2uint_rn` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to unsigned int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2uint_rn, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2uint_rz` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to unsigned int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2uint_rz, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2uint_rd` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to unsigned int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2uint_rd, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2uint_ru` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to unsigned int.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2uint_ru, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2short_rn` for all possible inputs. The results are compared
|
||||
* against reference function which performs __half cast to short.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2short_rn, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2short_rz` for all possible inputs. The results are compared
|
||||
* against reference function which performs __half cast to short.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2short_rz, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2short_rd` for all possible inputs. The results are compared
|
||||
* against reference function which performs __half cast to short.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2short_rd, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2short_ru` for all possible inputs. The results are compared
|
||||
* against reference function which performs __half cast to short.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2short_ru, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ushort_rn` for all possible inputs. The results are compared
|
||||
* against reference function which performs __half cast to unsigned short.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ushort_rn, unsigned short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ushort_rz` for all possible inputs. The results are compared
|
||||
* against reference function which performs __half cast to unsigned short.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ushort_rz, unsigned short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ushort_rd` for all possible inputs. The results are compared
|
||||
* against reference function which performs __half cast to unsigned short.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ushort_rd, unsigned short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ushort_ru` for all possible inputs. The results are compared
|
||||
* against reference function which performs __half cast to unsigned short.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ushort_ru, unsigned short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ll_rn` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to long long.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ll_rn, long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ll_rz` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to long long.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ll_rz, long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ll_rd` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to long long.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ll_rd, long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ll_ru` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to long long.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ll_ru, long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ull_rn` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to unsigned long long.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ull_rn, unsigned long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ull_rz` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to unsigned long long.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ull_rz, unsigned long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ull_rd` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to unsigned long long.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ull_rd, unsigned long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2ull_ru` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to unsigned long long.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_HALF2INT_RN_TEST_DEF(__half2ull_ru, unsigned long long)
|
||||
|
||||
CAST_KERNEL_DEF(__half_as_short, short, Float16)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half_as_short` for all possible inputs. The results are compared
|
||||
* against reference function which performs copy of __half value to short variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___half_as_short_Accuracy_Positive") {
|
||||
short (*ref)(Float16) = type2_as_type1_ref<short, Float16>;
|
||||
CastUnaryHalfPrecisionTest(__half_as_short_kernel, ref, EqValidatorBuilderFactory<short>());
|
||||
}
|
||||
|
||||
CAST_KERNEL_DEF(__half_as_ushort, unsigned short, Float16)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half_as_ushort` for all possible inputs. The results are compared
|
||||
* against reference function which performs copy of __half value to unsigned short variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half2int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___half_as_ushort_Accuracy_Positive") {
|
||||
unsigned short (*ref)(Float16) = type2_as_type1_ref<unsigned short, Float16>;
|
||||
CastUnaryHalfPrecisionTest(__half_as_ushort_kernel, ref,
|
||||
EqValidatorBuilderFactory<unsigned short>());
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip/hip_fp16.h>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL(func_name, T) \
|
||||
__global__ void func_name##_kernel_v1(T* result, __half* x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(T* result, Dummy x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy* result, __half x) { *result = unc_name(x); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL(__half2int_rn, int)
|
||||
NEGATIVE_KERNELS_SHELL(__half2int_rz, int)
|
||||
NEGATIVE_KERNELS_SHELL(__half2int_rd, int)
|
||||
NEGATIVE_KERNELS_SHELL(__half2int_ru, int)
|
||||
NEGATIVE_KERNELS_SHELL(__half2uint_rn, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__half2uint_rz, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__half2uint_rd, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__half2uint_ru, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__half2short_rn, short)
|
||||
NEGATIVE_KERNELS_SHELL(__half2short_rz, short)
|
||||
NEGATIVE_KERNELS_SHELL(__half2short_rd, short)
|
||||
NEGATIVE_KERNELS_SHELL(__half2short_ru, short)
|
||||
NEGATIVE_KERNELS_SHELL(__half_as_short, short)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ushort_rn, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ushort_rz, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ushort_rd, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ushort_ru, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__half_as_ushort, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ll_rn, long long)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ll_rz, long long)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ll_rd, long long)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ll_ru, long long)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ull_rn, unsigned long long)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ull_rz, unsigned long long)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ull_rd, unsigned long long)
|
||||
NEGATIVE_KERNELS_SHELL(__half2ull_ru, unsigned long long)
|
||||
@@ -0,0 +1,247 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "half_precision_common.hh"
|
||||
#include "casting_common.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup HalfPrecisionCastingFloat HalfPrecisionCastingFloat
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
#define CAST_FLOAT2HALF_TEST_DEF(kern_name, round_dir) \
|
||||
CAST_KERNEL_DEF(kern_name, Float16, float) \
|
||||
CAST_RND_REF_DEF(kern_name, Float16, float, round_dir) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Limited_Positive") { \
|
||||
Float16 (*ref)(float) = kern_name##_ref; \
|
||||
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>(), \
|
||||
std::numeric_limits<float>::min(), 0.f); \
|
||||
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>(), \
|
||||
0.0001f, std::numeric_limits<float>::max()); \
|
||||
}
|
||||
|
||||
#define CAST_FLOAT2HALF_RN_TEST_DEF(kern_name) \
|
||||
CAST_KERNEL_DEF(kern_name, Float16, float) \
|
||||
CAST_REF_DEF(kern_name, Float16, float) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive") { \
|
||||
Float16 (*ref)(float) = kern_name##_ref; \
|
||||
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>(), \
|
||||
std::numeric_limits<float>::min(), \
|
||||
std::numeric_limits<float>::max()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2half_rd` for all possible inputs apart from very small positive
|
||||
* values. Rounding behaviour is not correct for host functions for this range. The results are
|
||||
* compared against reference function which performs float cast to __half with FE_DOWNWARD rounding
|
||||
* mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2HALF_TEST_DEF(__float2half_rd, FE_DOWNWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2half_rn` for all possible inputs. The results are compared against
|
||||
* reference function which performs float cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2HALF_RN_TEST_DEF(__float2half_rn)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2half` for all possible inputs. The results are compared against
|
||||
* reference function which performs float cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2HALF_RN_TEST_DEF(__float2half)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2half_ru` for all possible inputs apart from very small positive
|
||||
* values. Rounding behaviour is not correct for host functions for this range. The results are
|
||||
* compared against reference function which performs float cast to __half with FE_UPWARD rounding
|
||||
* mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2HALF_TEST_DEF(__float2half_ru, FE_UPWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__float2half_rz` for all possible inputs apart from very small positive
|
||||
* values. Rounding behaviour is not correct for host functions for this range. The results are
|
||||
* compared against reference function which performs float cast to __half with FE_TOWARDZERO rounding
|
||||
* mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_FLOAT2HALF_TEST_DEF(__float2half_rz, FE_TOWARDZERO)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test that checks `__float2half_rd` for very small positive values.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float2half_rd_SmallVals_Sanity_Positive") {
|
||||
const float input[] = {0.8859e-06f, 1.5454e-07f, 6.5955e-08f, 2.7955e-08f,
|
||||
3.7956e-09f, 4.8995e-10f, 5.7997e-15f, 6.2117e-20f,
|
||||
7.4999e-25f, 8.9999e-30f, 9.0001e-35f};
|
||||
const Float16 reference[] = {8.34465e-07, 1.19209e-07, 5.96046e-08, 0, 0, 0, 0, 0, 0, 0, 0};
|
||||
LinearAllocGuard<float> input_dev{LinearAllocs::hipMalloc, sizeof(float)};
|
||||
LinearAllocGuard<Float16> out(LinearAllocs::hipMallocManaged, sizeof(Float16));
|
||||
|
||||
|
||||
for (int i = 0; i < 11; ++i) {
|
||||
HIP_CHECK(hipMemcpy(input_dev.ptr(), input + i, sizeof(float), hipMemcpyHostToDevice));
|
||||
|
||||
__float2half_rd_kernel<<<1, 1>>>(out.ptr(), 1, input_dev.ptr());
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
REQUIRE(out.ptr()[0] == reference[i]);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test that checks `__float2half_ru` for very small positive values.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float2half_ru_SmallVals_Sanity_Positive") {
|
||||
const float input[] = {0.8859e-06f, 1.5454e-07f, 6.5955e-08f, 2.7955e-08f,
|
||||
3.7956e-09f, 4.8995e-10f, 5.7997e-15f, 6.2117e-20f,
|
||||
7.4999e-25f, 8.9999e-30f, 9.0001e-35f};
|
||||
const Float16 reference[] = {8.9407e-07, 1.78814e-07, 1.19209e-07, 5.96046e-08,
|
||||
5.96046e-08, 5.96046e-08, 5.96046e-08, 5.96046e-08,
|
||||
5.96046e-08, 5.96046e-08, 5.96046e-08};
|
||||
LinearAllocGuard<float> input_dev{LinearAllocs::hipMalloc, sizeof(float)};
|
||||
LinearAllocGuard<Float16> out(LinearAllocs::hipMallocManaged, sizeof(Float16));
|
||||
|
||||
|
||||
for (int i = 0; i < 11; ++i) {
|
||||
HIP_CHECK(hipMemcpy(input_dev.ptr(), input + i, sizeof(float), hipMemcpyHostToDevice));
|
||||
|
||||
__float2half_ru_kernel<<<1, 1>>>(out.ptr(), 1, input_dev.ptr());
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
REQUIRE(out.ptr()[0] == reference[i]);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test that checks `__float2half_rz` for very small positive values.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___float2half_rz_SmallVals_Sanity_Positive") {
|
||||
const float input[] = {0.8859e-06f, 1.5454e-07f, 6.5955e-08f, 2.7955e-08f,
|
||||
3.7956e-09f, 4.8995e-10f, 5.7997e-15f, 6.2117e-20f,
|
||||
7.4999e-25f, 8.9999e-30f, 9.0001e-35f};
|
||||
const Float16 reference[] = {8.34465e-07, 1.19209e-07, 5.96046e-08, 0, 0, 0, 0, 0, 0, 0, 0};
|
||||
LinearAllocGuard<float> input_dev{LinearAllocs::hipMalloc, sizeof(float)};
|
||||
LinearAllocGuard<Float16> out(LinearAllocs::hipMallocManaged, sizeof(Float16));
|
||||
|
||||
|
||||
for (int i = 0; i < 11; ++i) {
|
||||
HIP_CHECK(hipMemcpy(input_dev.ptr(), input + i, sizeof(float), hipMemcpyHostToDevice));
|
||||
|
||||
__float2half_rz_kernel<<<1, 1>>>(out.ptr(), 1, input_dev.ptr());
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
REQUIRE(out.ptr()[0] == reference[i]);
|
||||
}
|
||||
}
|
||||
|
||||
CAST_KERNEL_DEF(__half2float, float, Float16)
|
||||
CAST_REF_DEF(__half2float, float, Float16)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__half2float` for all possible inputs. The results are compared against
|
||||
* reference function which performs __half cast to float.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_half_float_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___half2float_Accuracy_Positive") {
|
||||
float (*ref)(Float16) = __half2float_ref;
|
||||
UnaryHalfPrecisionTest(__half2float_kernel, ref, EqValidatorBuilderFactory<float>());
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip/hip_fp16.h>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_F2H_KERNELS_SHELL(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half* result, float* x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(__half* result, Dummy x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy* result, float x) { *result = func_name(x); }
|
||||
|
||||
#define NEGATIVE_H2F_KERNELS_SHELL(func_name) \
|
||||
__global__ void func_name##_kernel_v1(float* result, __half* x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(float* result, Dummy x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy* result, __half x) { *result = func_name(x); }
|
||||
|
||||
NEGATIVE_F2H_KERNELS_SHELL(__float2half_rd)
|
||||
NEGATIVE_F2H_KERNELS_SHELL(__float2half_rn)
|
||||
NEGATIVE_F2H_KERNELS_SHELL(__float2half_ru)
|
||||
NEGATIVE_F2H_KERNELS_SHELL(__float2half_rz)
|
||||
NEGATIVE_F2H_KERNELS_SHELL(__float2half)
|
||||
|
||||
NEGATIVE_H2F_KERNELS_SHELL(__half2float)
|
||||
@@ -0,0 +1,448 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "half_precision_common.hh"
|
||||
#include "casting_common.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup HalfPrecisionCastingIntTypes HalfPrecisionCastingIntTypes
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
#define CAST_INT2HALF_RN_TEST_DEF(kern_name, T) \
|
||||
CAST_KERNEL_DEF(kern_name, Float16, T) \
|
||||
CAST_REF_DEF(kern_name, Float16, T) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive") { \
|
||||
Float16 (*ref)(T) = kern_name##_ref; \
|
||||
CastIntRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2half_rn` for all possible inputs. The results are compared against
|
||||
* reference function which performs int cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__int2half_rn, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2half_rz` for all possible inputs. The results are compared against
|
||||
* reference function which performs int cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__int2half_rz, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2half_rd` for all possible inputs. The results are compared against
|
||||
* reference function which performs int cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__int2half_rd, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2half_ru` for all possible inputs. The results are compared against
|
||||
* reference function which performs int cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__int2half_ru, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2half_rn` for all possible inputs. The results are compared against
|
||||
* reference function which performs unsigned int cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__uint2half_rn, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2half_rz` for all possible inputs. The results are compared against
|
||||
* reference function which performs unsigned int cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__uint2half_rz, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2half_rd` for all possible inputs. The results are compared against
|
||||
* reference function which performs unsigned int cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__uint2half_rd, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2half_ru` for all possible inputs. The results are compared against
|
||||
* reference function which performs unsigned int cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__uint2half_ru, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__short2half_rn` for all possible inputs. The results are compared
|
||||
* against reference function which performs short cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__short2half_rn, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__short2half_rz` for all possible inputs. The results are compared
|
||||
* against reference function which performs short cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__short2half_rz, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__short2half_rd` for all possible inputs. The results are compared
|
||||
* against reference function which performs short cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__short2half_rd, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__short2half_ru` for all possible inputs. The results are compared
|
||||
* against reference function which performs short cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__short2half_ru, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ushort2half_rn` for all possible inputs. The results are compared
|
||||
* against reference function which performs unsigned short cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__ushort2half_rn, unsigned short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ushort2half_rz` for all possible inputs. The results are compared
|
||||
* against reference function which performs unsigned short cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__ushort2half_rz, unsigned short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ushort2half_rd` for all possible inputs. The results are compared
|
||||
* against reference function which performs unsigned short cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__ushort2half_rd, unsigned short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ushort2half_ru` for all possible inputs. The results are compared
|
||||
* against reference function which performs unsigned short cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2HALF_RN_TEST_DEF(__ushort2half_ru, unsigned short)
|
||||
|
||||
#define CAST_LL2HALF_TEST_DEF(kern_name, T) \
|
||||
CAST_KERNEL_DEF(kern_name, Float16, T) \
|
||||
CAST_REF_DEF(kern_name, Float16, T) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive") { \
|
||||
Float16 (*ref)(T) = kern_name##_ref; \
|
||||
CastIntBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2half_rn` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs long long cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2HALF_TEST_DEF(__ll2half_rn, long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2half_rz` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs long long cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2HALF_TEST_DEF(__ll2half_rz, long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2half_rd` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs long long cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2HALF_TEST_DEF(__ll2half_rd, long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2half_ru` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs long long cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2HALF_TEST_DEF(__ll2half_ru, long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2half_rn` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs unsigned long long cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2HALF_TEST_DEF(__ull2half_rn, unsigned long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2half_rz` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs unsigned long long cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2HALF_TEST_DEF(__ull2half_rz, unsigned long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2half_rd` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs unsigned long long cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2HALF_TEST_DEF(__ull2half_rd, unsigned long long)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2half_ru` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs unsigned long long cast to __half.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2HALF_TEST_DEF(__ull2half_ru, unsigned long long)
|
||||
|
||||
CAST_KERNEL_DEF(__short_as_half, Float16, short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__short_as_half` for all possible inputs. The results are compared
|
||||
* against reference function which performs copy of short value to __half variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___short_as_half_Accuracy_Positive") {
|
||||
Float16 (*ref)(short) = type2_as_type1_ref<Float16, short>;
|
||||
CastIntBruteForceTest(__short_as_half_kernel, ref, EqValidatorBuilderFactory<Float16>());
|
||||
}
|
||||
|
||||
CAST_KERNEL_DEF(__ushort_as_half, Float16, unsigned short)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ushort_as_half` for all possible inputs. The results are compared
|
||||
* against reference function which performs copy of unsigned short value to __half variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int2half_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___ushort_as_half_Accuracy_Positive") {
|
||||
Float16 (*ref)(unsigned short) = type2_as_type1_ref<Float16, unsigned short>;
|
||||
CastIntBruteForceTest(__ushort_as_half_kernel, ref, EqValidatorBuilderFactory<Float16>());
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip/hip_fp16.h>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL(func_name, T) \
|
||||
__global__ void func_name##_kernel_v1(__half* result, T* x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(__half* result, Dummy x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy* result, T x) { *result = func_name(x); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL(__int2half_rn, int)
|
||||
NEGATIVE_KERNELS_SHELL(__int2half_rz, int)
|
||||
NEGATIVE_KERNELS_SHELL(__int2half_rd, int)
|
||||
NEGATIVE_KERNELS_SHELL(__int2half_ru, int)
|
||||
NEGATIVE_KERNELS_SHELL(__uint2half_rn, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__uint2half_rz, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__uint2half_rd, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__uint2half_ru, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL(__short2half_rn, short)
|
||||
NEGATIVE_KERNELS_SHELL(__short2half_rz, short)
|
||||
NEGATIVE_KERNELS_SHELL(__short2half_rd, short)
|
||||
NEGATIVE_KERNELS_SHELL(__short2half_ru, short)
|
||||
NEGATIVE_KERNELS_SHELL(__short_as_half, short)
|
||||
NEGATIVE_KERNELS_SHELL(__ushort2half_rn, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__ushort2half_rz, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__ushort2half_rd, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__ushort2half_ru, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__ushort_as_half, unsigned short)
|
||||
NEGATIVE_KERNELS_SHELL(__ll2half_rn, long long)
|
||||
NEGATIVE_KERNELS_SHELL(__ll2half_rz, long long)
|
||||
NEGATIVE_KERNELS_SHELL(__ll2half_rd, long long)
|
||||
NEGATIVE_KERNELS_SHELL(__ll2half_ru, long long)
|
||||
NEGATIVE_KERNELS_SHELL(__ull2half_rn, unsigned long long)
|
||||
NEGATIVE_KERNELS_SHELL(__ull2half_rz, unsigned long long)
|
||||
NEGATIVE_KERNELS_SHELL(__ull2half_rd, unsigned long long)
|
||||
NEGATIVE_KERNELS_SHELL(__ull2half_ru, unsigned long long)
|
||||
@@ -0,0 +1,735 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "casting_common.hh"
|
||||
#include "casting_int_negative_kernels_rtc.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup CastingIntTypes CastingIntTypes
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
#define CAST_INT2FLOAT_TEST_DEF(kern_name, T1, T2, round_dir) \
|
||||
CAST_KERNEL_DEF(kern_name, T1, T2) \
|
||||
CAST_RND_REF_DEF(kern_name, T1, T2, round_dir) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T1 (*ref)(T2) = kern_name##_ref; \
|
||||
CastIntRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T1>()); \
|
||||
}
|
||||
|
||||
#define CAST_INT2FLOAT_RN_TEST_DEF(kern_name, T1, T2) \
|
||||
CAST_KERNEL_DEF(kern_name, T1, T2) \
|
||||
CAST_REF_DEF(kern_name, T1, T2) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T1 (*ref)(T2) = kern_name##_ref; \
|
||||
CastIntRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T1>()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2float_rd` for all possible inputs. The results are compared against
|
||||
* reference function which performs cast to float with FE_DOWNWARD rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_TEST_DEF(__int2float_rd, float, int, FE_DOWNWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2float_rn` for all possible inputs. The results are compared against
|
||||
* reference function which performs cast to float.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_RN_TEST_DEF(__int2float_rn, float, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2float_ru` for all possible inputs. The results are compared against
|
||||
* reference function which performs cast to float with FE_UPWARD rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_TEST_DEF(__int2float_ru, float, int, FE_UPWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2float_rz` for all possible inputs. The results are compared against
|
||||
* reference function which performs cast to float with FE_TOWARDZERO rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_TEST_DEF(__int2float_rz, float, int, FE_TOWARDZERO)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __int2float_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_int2float___Negative_RTC") { NegativeTestRTCWrapper<12>(kInt2Float); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2float_rd` for all possible inputs. The results are compared
|
||||
* against reference function which performs cast to float with FE_DOWNWARD rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_TEST_DEF(__uint2float_rd, float, unsigned int, FE_DOWNWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2float_rn` for all possible inputs. The results are compared
|
||||
* against reference function which performs cast to float.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_RN_TEST_DEF(__uint2float_rn, float, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2float_ru` for all possible inputs. The results are compared
|
||||
* against reference function which performs cast to float with FE_UPWARD rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_TEST_DEF(__uint2float_ru, float, unsigned int, FE_UPWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2float_rz` for all possible inputs. The results are compared
|
||||
* against reference function which performs cast to float with FE_TOWARDZERO rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_TEST_DEF(__uint2float_rz, float, unsigned int, FE_TOWARDZERO)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __uint2float_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___uint2float_Negative_RTC") { NegativeTestRTCWrapper<12>(kUint2Float); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int2double_rn` for all possible inputs. The results are compared
|
||||
* against reference function which performs cast to double.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_RN_TEST_DEF(__int2double_rn, double, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __int2double_rn.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___int2double_Negative_RTC") { NegativeTestRTCWrapper<3>(kInt2Double); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint2double_rn` for all possible inputs. The results are compared
|
||||
* against reference function which performs cast to double.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_INT2FLOAT_RN_TEST_DEF(__uint2double_rn, double, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __uint2double_rn.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___uint2double_Negative_RTC") { NegativeTestRTCWrapper<3>(kUint2Double); }
|
||||
|
||||
#define CAST_LL2FLOAT_TEST_DEF(kern_name, T1, T2, round_dir) \
|
||||
CAST_KERNEL_DEF(kern_name, T1, T2) \
|
||||
CAST_RND_REF_DEF(kern_name, T1, T2, round_dir) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T1 (*ref)(T2) = kern_name##_ref; \
|
||||
CastIntBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T1>()); \
|
||||
}
|
||||
|
||||
#define CAST_LL2FLOAT_RN_TEST_DEF(kern_name, T1, T2) \
|
||||
CAST_KERNEL_DEF(kern_name, T1, T2) \
|
||||
CAST_REF_DEF(kern_name, T1, T2) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
|
||||
T1 (*ref)(T2) = kern_name##_ref; \
|
||||
CastIntBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T1>()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2float_rd` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to float with FE_DOWNWARD
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ll2float_rd, float, long long int, FE_DOWNWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2float_rn` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to float.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_RN_TEST_DEF(__ll2float_rn, float, long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2float_ru` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to float with FE_UPWARD
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ll2float_ru, float, long long int, FE_UPWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2float_rz` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to float with FE_TOWARDZERO
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ll2float_rz, float, long long int, FE_TOWARDZERO)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __ll2float_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___ll2float_Negative_RTC") { NegativeTestRTCWrapper<12>(kLL2Float); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2float_rd` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to float with FE_DOWNWARD
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ull2float_rd, float, unsigned long long int, FE_DOWNWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2float_rn` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to float.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_RN_TEST_DEF(__ull2float_rn, float, unsigned long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2float_ru` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to float with FE_UPWARD
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ull2float_ru, float, unsigned long long int, FE_UPWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2float_rz` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to float with FE_TOWARDZERO
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ull2float_rz, float, unsigned long long int, FE_TOWARDZERO)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __ull2float_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___ull2float_Negative_RTC") { NegativeTestRTCWrapper<12>(kULL2Float); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2double_rd` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to double with FE_DOWNWARD
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ll2double_rd, double, long long int, FE_DOWNWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2double_rn` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to double.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_RN_TEST_DEF(__ll2double_rn, double, long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2double_ru` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to double with FE_UPWARD
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ll2double_ru, double, long long int, FE_UPWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ll2double_rz` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to double with FE_TOWARDZERO
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ll2double_rz, double, long long int, FE_TOWARDZERO)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __ll2double_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___ll2double_Negative_RTC") { NegativeTestRTCWrapper<12>(kLL2Double); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2double_rd` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to double with FE_DOWNWARD
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ull2double_rd, double, unsigned long long int, FE_DOWNWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2double_rn` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to double.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_RN_TEST_DEF(__ull2double_rn, double, unsigned long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2double_ru` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to double with FE_UPWARD
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ull2double_ru, double, unsigned long long int, FE_UPWARD)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__ull2double_rz` against a large number of randomly generated values. The
|
||||
* results are compared against reference function which performs cast to double with FE_TOWARDZERO
|
||||
* rounding mode.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
CAST_LL2FLOAT_TEST_DEF(__ull2double_rz, double, unsigned long long int, FE_TOWARDZERO)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __ull2double_[rd,rn,ru,rz].
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___ull2double_Negative_RTC") { NegativeTestRTCWrapper<12>(kULL2Double); }
|
||||
|
||||
CAST_KERNEL_DEF(__int_as_float, float, int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__int_as_float` for all possible inputs. The results are compared against
|
||||
* reference function which performs copy of int value to float variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___int_as_float_Positive") {
|
||||
float (*ref)(int) = type2_as_type1_ref<float, int>;
|
||||
CastIntRangeTest(__int_as_float_kernel, ref, EqValidatorBuilderFactory<float>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __int_as_float.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___int_as_float_Negative_RTC") { NegativeTestRTCWrapper<3>(kIntAsFloat); }
|
||||
|
||||
CAST_KERNEL_DEF(__uint_as_float, float, unsigned int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__uint_as_float` for all possible inputs. The results are compared
|
||||
* against reference function which performs copy of unsigned int value to float variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___uint_as_float_Positive") {
|
||||
float (*ref)(unsigned int) = type2_as_type1_ref<float, unsigned int>;
|
||||
CastIntRangeTest(__uint_as_float_kernel, ref, EqValidatorBuilderFactory<float>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __uint_as_float.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___uint_as_float_Negative_RTC") { NegativeTestRTCWrapper<3>(kUintAsFloat); }
|
||||
|
||||
CAST_KERNEL_DEF(__longlong_as_double, double, long long int)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__longlong_as_double` against a large number of randomly generated
|
||||
* values. The results are compared against reference function which performs copy of long long int
|
||||
* value to double variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___longlong_as_double_Positive") {
|
||||
double (*ref)(long long int) = type2_as_type1_ref<double, long long int>;
|
||||
CastIntBruteForceTest(__longlong_as_double_kernel, ref, EqValidatorBuilderFactory<double>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __longlong_as_double.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___longlong_as_double_Negative_RTC") {
|
||||
NegativeTestRTCWrapper<3>(kLonglongAsDouble);
|
||||
}
|
||||
|
||||
__global__ void __hiloint2double_kernel(double* const ys, const size_t num_xs, int* const x1s,
|
||||
int* const x2s) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
ys[i] = __hiloint2double(x1s[i], x2s[i]);
|
||||
}
|
||||
}
|
||||
|
||||
double __hiloint2double_ref(int hi, int lo) {
|
||||
uint64_t tmp0 = (static_cast<uint64_t>(hi) << 32ull) | static_cast<uint32_t>(lo);
|
||||
double tmp1;
|
||||
memcpy(&tmp1, &tmp0, sizeof(tmp0));
|
||||
|
||||
return tmp1;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests that checks `__hiloint2double` for all possible inputs for hi value. The results are
|
||||
* compared against reference function which performs copy of hi int value to higher part of double
|
||||
* variable and copy of lo int value to lower part of double variable.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___hiloint2double_Positive") {
|
||||
double (*ref)(int, int) = __hiloint2double_ref;
|
||||
CastBinaryIntRangeTest(__hiloint2double_kernel, ref, EqValidatorBuilderFactory<double>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for __hiloint2double.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/casting_int_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___hiloint2double_Negative_RTC") { NegativeTestRTCWrapper<5>(kHilo2Double); }
|
||||
@@ -0,0 +1,79 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_ONE_ARG(func_name, T1, T2) \
|
||||
__global__ void func_name##_kernel_v1(T1* result, T2* x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(T1* result, Dummy x) { *result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy* result, T2 x) { *result = func_name(x); }
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_TWO_ARGS(func_name, T1, T2) \
|
||||
__global__ void func_name##_kernel_v1(T1* result, T2* x, T2 y) { \
|
||||
*result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(T1* result, T2 x, T2* y) { \
|
||||
*result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(T1* result, Dummy x, T2 y) { \
|
||||
*result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v4(T1* result, T2 x, Dummy y) { \
|
||||
*result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v5(Dummy* result, T2 x, T2 y) { \
|
||||
*result = func_name(x, y); \
|
||||
}
|
||||
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2float_rd, float, int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2float_rn, float, int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2float_ru, float, int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2float_rz, float, int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2float_rd, float, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2float_rn, float, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2float_ru, float, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2float_rz, float, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2float_rd, float, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2float_rn, float, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2float_ru, float, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2float_rz, float, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2float_rd, float, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2float_rn, float, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2float_ru, float, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2float_rz, float, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2double_rn, double, int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2double_rn, double, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2double_rd, double, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2double_rn, double, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2double_ru, double, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2double_rz, double, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2double_rd, double, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2double_rn, double, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2double_ru, double, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2double_rz, double, unsigned long long int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int_as_float, float, int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint_as_float, float, unsigned int)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(__longlong_as_double, double, long long int)
|
||||
NEGATIVE_KERNELS_SHELL_TWO_ARGS(__hiloint2double, double, int)
|
||||
@@ -0,0 +1,215 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
/*
|
||||
Negative kernels used for the <unsigned> int/long long type casting negative Test Cases that are using RTC.
|
||||
*/
|
||||
|
||||
static constexpr auto kInt2Float{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void int2float_rd_kernel_v1(float* result, int* x) { *result = __int2float_rd(x); }
|
||||
__global__ void int2float_rd_kernel_v2(float* result, Dummy x) { *result = __int2float_rd(x); }
|
||||
__global__ void int2float_rd_kernel_v3(Dummy* result, int x) { *result = __int2float_rd(x); }
|
||||
__global__ void int2float_rn_kernel_v1(float* result, int* x) { *result = __int2float_rn(x); }
|
||||
__global__ void int2float_rn_kernel_v2(float* result, Dummy x) { *result = __int2float_rn(x); }
|
||||
__global__ void int2float_rn_kernel_v3(Dummy* result, int x) { *result = __int2float_rn(x); }
|
||||
__global__ void int2float_ru_kernel_v1(float* result, int* x) { *result = __int2float_ru(x); }
|
||||
__global__ void int2float_ru_kernel_v2(float* result, Dummy x) { *result = __int2float_ru(x); }
|
||||
__global__ void int2float_ru_kernel_v3(Dummy* result, int x) { *result = __int2float_ru(x); }
|
||||
__global__ void int2float_rz_kernel_v1(float* result, int* x) { *result = __int2float_rz(x); }
|
||||
__global__ void int2float_rz_kernel_v2(float* result, Dummy x) { *result = __int2float_rz(x); }
|
||||
__global__ void int2float_rz_kernel_v3(Dummy* result, int x) { *result = __int2float_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kUint2Float{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void uint2float_rd_kernel_v1(float* result, unsigned int* x) { *result = __uint2float_rd(x); }
|
||||
__global__ void uint2float_rd_kernel_v2(float* result, Dummy x) { *result = __uint2float_rd(x); }
|
||||
__global__ void uint2float_rd_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2float_rd(x); }
|
||||
__global__ void uint2float_rn_kernel_v1(float* result, unsigned int* x) { *result = __uint2float_rn(x); }
|
||||
__global__ void uint2float_rn_kernel_v2(float* result, Dummy x) { *result = __uint2float_rn(x); }
|
||||
__global__ void uint2float_rn_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2float_rn(x); }
|
||||
__global__ void uint2float_ru_kernel_v1(float* result, unsigned int* x) { *result = __uint2float_ru(x); }
|
||||
__global__ void uint2float_ru_kernel_v2(float* result, Dummy x) { *result = __uint2float_ru(x); }
|
||||
__global__ void uint2float_ru_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2float_ru(x); }
|
||||
__global__ void uint2float_rz_kernel_v1(float* result, unsigned int* x) { *result = __uint2float_rz(x); }
|
||||
__global__ void uint2float_rz_kernel_v2(float* result, Dummy x) { *result = __uint2float_rz(x); }
|
||||
__global__ void uint2float_rz_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2float_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLL2Float{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void ll2float_rd_kernel_v1(float* result, long long int* x) { *result = __ll2float_rd(x); }
|
||||
__global__ void ll2float_rd_kernel_v2(float* result, Dummy x) { *result = __ll2float_rd(x); }
|
||||
__global__ void ll2float_rd_kernel_v3(Dummy* result, long long int x) { *result = __ll2float_rd(x); }
|
||||
__global__ void ll2float_rn_kernel_v1(float* result, long long int* x) { *result = __ll2float_rn(x); }
|
||||
__global__ void ll2float_rn_kernel_v2(float* result, Dummy x) { *result = __ll2float_rn(x); }
|
||||
__global__ void ll2float_rn_kernel_v3(Dummy* result, long long int x) { *result = __ll2float_rn(x); }
|
||||
__global__ void ll2float_ru_kernel_v1(float* result, long long int* x) { *result = __ll2float_ru(x); }
|
||||
__global__ void ll2float_ru_kernel_v2(float* result, Dummy x) { *result = __ll2float_ru(x); }
|
||||
__global__ void ll2float_ru_kernel_v3(Dummy* result, long long int x) { *result = __ll2float_ru(x); }
|
||||
__global__ void ll2float_rz_kernel_v1(float* result, long long int* x) { *result = __ll2float_rz(x); }
|
||||
__global__ void ll2float_rz_kernel_v2(float* result, Dummy x) { *result = __ll2float_rz(x); }
|
||||
__global__ void ll2float_rz_kernel_v3(Dummy* result, long long int x) { *result = __ll2float_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kULL2Float{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void ull2float_rd_kernel_v1(float* result, unsigned long long int* x) { *result = __ull2float_rd(x); }
|
||||
__global__ void ull2float_rd_kernel_v2(float* result, Dummy x) { *result = __ull2float_rd(x); }
|
||||
__global__ void ull2float_rd_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2float_rd(x); }
|
||||
__global__ void ull2float_rn_kernel_v1(float* result, unsigned long long int* x) { *result = __ull2float_rn(x); }
|
||||
__global__ void ull2float_rn_kernel_v2(float* result, Dummy x) { *result = __ull2float_rn(x); }
|
||||
__global__ void ull2float_rn_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2float_rn(x); }
|
||||
__global__ void ull2float_ru_kernel_v1(float* result, unsigned long long int* x) { *result = __ull2float_ru(x); }
|
||||
__global__ void ull2float_ru_kernel_v2(float* result, Dummy x) { *result = __ull2float_ru(x); }
|
||||
__global__ void ull2float_ru_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2float_ru(x); }
|
||||
__global__ void ull2float_rz_kernel_v1(float* result, unsigned long long int* x) { *result = __ull2float_rz(x); }
|
||||
__global__ void ull2float_rz_kernel_v2(float* result, Dummy x) { *result = __ull2float_rz(x); }
|
||||
__global__ void ull2float_rz_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2float_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kIntAsFloat{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void int_as_float_kernel_v1(float* result, int* x) { *result = __int_as_float(x); }
|
||||
__global__ void int_as_float_kernel_v2(float* result, Dummy x) { *result = __int_as_float(x); }
|
||||
__global__ void int_as_float_kernel_v3(Dummy* result, int x) { *result = __int_as_float(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kUintAsFloat{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void uint_as_float_kernel_v1(float* result, unsigned int* x) { *result = __uint_as_float(x); }
|
||||
__global__ void uint_as_float_kernel_v2(float* result, Dummy x) { *result = __uint_as_float(x); }
|
||||
__global__ void uint_as_float_kernel_v3(Dummy* result, unsigned int x) { *result = __uint_as_float(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kInt2Double{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void int2double_rn_kernel_v1(double* result, int* x) { *result = __int2double_rn(x); }
|
||||
__global__ void int2double_rn_kernel_v2(double* result, Dummy x) { *result = __int2double_rn(x); }
|
||||
__global__ void int2double_rn_kernel_v3(Dummy* result, int x) { *result = __int2double_rn(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kUint2Double{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void uint2double_rn_kernel_v1(double* result, unsigned int* x) { *result = __uint2double_rn(x); }
|
||||
__global__ void uint2double_rn_kernel_v2(double* result, Dummy x) { *result = __uint2double_rn(x); }
|
||||
__global__ void uint2double_rn_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2double_rn(x); }
|
||||
)"};
|
||||
|
||||
|
||||
static constexpr auto kLL2Double{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void ll2double_rd_kernel_v1(double* result, long long int* x) { *result = __ll2double_rd(x); }
|
||||
__global__ void ll2double_rd_kernel_v2(double* result, Dummy x) { *result = __ll2double_rd(x); }
|
||||
__global__ void ll2double_rd_kernel_v3(Dummy* result, long long int x) { *result = __ll2double_rd(x); }
|
||||
__global__ void ll2double_rn_kernel_v1(double* result, long long int* x) { *result = __ll2double_rn(x); }
|
||||
__global__ void ll2double_rn_kernel_v2(double* result, Dummy x) { *result = __ll2double_rn(x); }
|
||||
__global__ void ll2double_rn_kernel_v3(Dummy* result, long long int x) { *result = __ll2double_rn(x); }
|
||||
__global__ void ll2double_ru_kernel_v1(double* result, long long int* x) { *result = __ll2double_ru(x); }
|
||||
__global__ void ll2double_ru_kernel_v2(double* result, Dummy x) { *result = __ll2double_ru(x); }
|
||||
__global__ void ll2double_ru_kernel_v3(Dummy* result, long long int x) { *result = __ll2double_ru(x); }
|
||||
__global__ void ll2double_rz_kernel_v1(double* result, long long int* x) { *result = __ll2double_rz(x); }
|
||||
__global__ void ll2double_rz_kernel_v2(double* result, Dummy x) { *result = __ll2double_rz(x); }
|
||||
__global__ void ll2double_rz_kernel_v3(Dummy* result, long long int x) { *result = __ll2double_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kULL2Double{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void ull2double_rd_kernel_v1(double* result, unsigned long long int* x) { *result = __ull2double_rd(x); }
|
||||
__global__ void ull2double_rd_kernel_v2(double* result, Dummy x) { *result = __ull2double_rd(x); }
|
||||
__global__ void ull2double_rd_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2double_rd(x); }
|
||||
__global__ void ull2double_rn_kernel_v1(double* result, unsigned long long int* x) { *result = __ull2double_rn(x); }
|
||||
__global__ void ull2double_rn_kernel_v2(double* result, Dummy x) { *result = __ull2double_rn(x); }
|
||||
__global__ void ull2double_rn_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2double_rn(x); }
|
||||
__global__ void ull2double_ru_kernel_v1(double* result, unsigned long long int* x) { *result = __ull2double_ru(x); }
|
||||
__global__ void ull2double_ru_kernel_v2(double* result, Dummy x) { *result = __ull2double_ru(x); }
|
||||
__global__ void ull2double_ru_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2double_ru(x); }
|
||||
__global__ void ull2double_rz_kernel_v1(double* result, unsigned long long int* x) { *result = __ull2double_rz(x); }
|
||||
__global__ void ull2double_rz_kernel_v2(double* result, Dummy x) { *result = __ull2double_rz(x); }
|
||||
__global__ void ull2double_rz_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2double_rz(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLonglongAsDouble{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void longlong_as_double_kernel_v1(double* result, long long int* x) { *result = __longlong_as_double(x); }
|
||||
__global__ void longlong_as_double_kernel_v2(double* result, Dummy x) { *result = __longlong_as_double(x); }
|
||||
__global__ void longlong_as_double_kernel_v3(Dummy* result, long long int x) { *result = __longlong_as_double(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kHilo2Double{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void hiloint2double_kernel_v1(double* result, int* x, int y) { *result = __hiloint2double(x, y); }
|
||||
__global__ void hiloint2double_kernel_v2(double* result, int x, int* y) { *result = __hiloint2double(x, y); }
|
||||
__global__ void hiloint2double_kernel_v3(double* result, Dummy x, int y) { *result = __hiloint2double(x, y); }
|
||||
__global__ void hiloint2double_kernel_v4(double* result, int x, Dummy y) { *result = __hiloint2double(x, y); }
|
||||
__global__ void hiloint2double_kernel_v5(Dummy* result, int x, int y) { *result = __hiloint2double(x, y); }
|
||||
)"};
|
||||
|
||||
|
||||
@@ -0,0 +1,243 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "binary_common.hh"
|
||||
#include "ternary_common.hh"
|
||||
|
||||
/********** Unary Functions **********/
|
||||
|
||||
#define MATH_UNARY_DP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(double* const ys, const size_t num_xs, double* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(xs[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_UNARY_DP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
UnaryDoublePrecisionTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_UNARY_DP_TEST_DEF(func_name, ref_func) \
|
||||
MATH_UNARY_DP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_UNARY_DP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<double>> func_name##_validator_builder(double target, double x)
|
||||
|
||||
|
||||
static double __drcp_rn_ref(double x) { return 1.0 / x; }
|
||||
|
||||
MATH_UNARY_DP_KERNEL_DEF(__drcp_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__drcp_rn(x)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are
|
||||
* IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/double_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_DP_TEST_DEF_IMPL(__drcp_rn, __drcp_rn_ref, EqValidatorBuilderFactory<double>());
|
||||
|
||||
|
||||
MATH_UNARY_DP_KERNEL_DEF(__dsqrt_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__dsqrt_rn(x)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are
|
||||
* compared against reference function `double std::sqrt(double)`. The error bounds are
|
||||
* IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/double_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_DP_TEST_DEF_IMPL(__dsqrt_rn, static_cast<double (*)(double)>(std::sqrt),
|
||||
EqValidatorBuilderFactory<double>());
|
||||
|
||||
|
||||
/********** Binary Functions **********/
|
||||
|
||||
#define MATH_BINARY_DP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(double* const ys, const size_t num_xs, double* const x1s, \
|
||||
double* const x2s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_BINARY_DP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
BinaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_BINARY_DP_TEST_DEF(func_name, ref_func) \
|
||||
MATH_BINARY_DP_TEST_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_BINARY_DP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<double>> func_name##_validator_builder(double target, \
|
||||
double x1, double x2)
|
||||
|
||||
|
||||
static double __dadd_rn_ref(double x1, double x2) { return x1 + x2; }
|
||||
|
||||
MATH_BINARY_DP_KERNEL_DEF(__dadd_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__dadd_rn(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/double_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_DP_TEST_DEF_IMPL(__dadd_rn, __dadd_rn_ref, EqValidatorBuilderFactory<double>());
|
||||
|
||||
|
||||
static double __dsub_rn_ref(double x1, double x2) { return x1 - x2; }
|
||||
|
||||
MATH_BINARY_DP_KERNEL_DEF(__dsub_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__dsub_rn(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/double_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_DP_TEST_DEF_IMPL(__dsub_rn, __dsub_rn_ref, EqValidatorBuilderFactory<double>());
|
||||
|
||||
|
||||
static double __dmul_rn_ref(double x1, double x2) { return x1 * x2; }
|
||||
|
||||
MATH_BINARY_DP_KERNEL_DEF(__dmul_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__dmul_rn(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/double_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_DP_TEST_DEF_IMPL(__dmul_rn, __dmul_rn_ref, EqValidatorBuilderFactory<double>());
|
||||
|
||||
|
||||
static double __ddiv_rn_ref(double x1, double x2) { return x1 / x2; }
|
||||
|
||||
MATH_BINARY_DP_KERNEL_DEF(__ddiv_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__ddiv_rn(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/double_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_DP_TEST_DEF_IMPL(__ddiv_rn, __ddiv_rn_ref, EqValidatorBuilderFactory<double>());
|
||||
|
||||
|
||||
/********** Ternary Functions **********/
|
||||
|
||||
#define MATH_TERNARY_DP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(double* const ys, const size_t num_xs, double* const x1s, \
|
||||
double* const x2s, double* const x3s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i], x3s[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_TERNARY_DP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
TernaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_TERNARY_DP_TEST_DEF(func_name, ref_func, validator_builder) \
|
||||
MATH_TERNARY_DP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_TERNARY_DP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<double>> func_name##_validator_builder( \
|
||||
double target, double x1, double x2, double x3)
|
||||
|
||||
|
||||
MATH_TERNARY_DP_KERNEL_DEF(__fma_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__fma(x,y,z)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/double_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_TERNARY_DP_TEST_DEF_IMPL(__fma_rn, static_cast<double (*)(double, double, double)>(std::fma),
|
||||
EqValidatorBuilderFactory<double>());
|
||||
@@ -0,0 +1,46 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define INTRINSIC_UNARY_DOUBLE_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); }
|
||||
|
||||
#define INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x, double y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(double x, double* y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, double y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(double x, Dummy y) { double result = func_name(x, y); }
|
||||
|
||||
|
||||
INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(__dadd_rn)
|
||||
INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(__dsub_rn)
|
||||
INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(__dmul_rn)
|
||||
INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(__ddiv_rn)
|
||||
INTRINSIC_UNARY_DOUBLE_NEGATIVE_KERNELS(__dsqrt_rn)
|
||||
@@ -0,0 +1,441 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "half_precision_common.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup HalfPrecisionArithmetic HalfPrecisionArithmetic
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(__habs);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__habs(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::abs(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(__habs, static_cast<float (*)(float)>(std::abs),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(__habs2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__habs2(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::abs(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(__habs2, static_cast<float (*)(float)>(std::abs),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __hneg_ref(float x) { return -x; }
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(__hneg);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hneg(x)` for all possible inputs. The error bounds are
|
||||
* IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(__hneg, __hneg_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(__hneg2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hneg2(x)` for all possible inputs. The error bounds are
|
||||
* IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(__hneg2, __hneg_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
// Wrapper to avoid ambiguity error with __hadd(int, int)
|
||||
__device__ __half __hadd_wrapper(__half x1, __half x2) { return __hadd(x1, x2); }
|
||||
|
||||
static float __hadd_ref(float x1, float x2) { return x1 + x2; }
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hadd_wrapper);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hadd(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hadd_wrapper, __hadd_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hadd2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hadd2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hadd2, __hadd_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __hadd_sat_ref(float x1, float x2) { return std::clamp(x1 + x2, 0.0f, 1.0f); }
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hadd_sat);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hadd_sat(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hadd_sat, __hadd_sat_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hadd2_sat);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hadd2_sat(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hadd2_sat, __hadd_sat_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __hsub_ref(float x1, float x2) { return x1 - x2; }
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hsub);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hsub(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hsub, __hsub_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hsub2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hsub2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hsub2, __hsub_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __hsub_sat_ref(float x1, float x2) { return std::clamp(x1 - x2, 0.0f, 1.0f); }
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hsub_sat);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hsub_sat(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hsub_sat, __hsub_sat_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hsub2_sat);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hsub2_sat(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hsub2_sat, __hsub_sat_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __hmul_ref(float x1, float x2) { return x1 * x2; }
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hmul);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hmul(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hmul, __hmul_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hmul2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hmul2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hmul2, __hmul_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __hmul_sat_ref(float x1, float x2) { return std::clamp(x1 * x2, 0.0f, 1.0f); }
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hmul_sat);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hmul_sat(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hmul_sat, __hmul_sat_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hmul2_sat);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hmul2_sat(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hmul2_sat, __hmul_sat_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __hdiv_ref(float x1, float x2) { return x1 / x2; }
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hdiv);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hdiv(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hdiv, __hdiv_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__h2div);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__h2div(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__h2div, __hdiv_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
MATH_TERNARY_HP_KERNEL_DEF(__hfma);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hfma(x,y,z)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_TERNARY_HP_TEST_DEF_IMPL(__hfma, static_cast<float (*)(float, float, float)>(std::fma),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_TERNARY_HP_KERNEL_DEF(__hfma2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hfma2(x,y,z)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_TERNARY_HP_TEST_DEF_IMPL(__hfma2, static_cast<float (*)(float, float, float)>(std::fma),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __hfma_sat_ref(float x1, float x2, float x3) {
|
||||
return std::clamp(std::fma(x1, x2, x3), 0.0f, 1.0f);
|
||||
}
|
||||
|
||||
MATH_TERNARY_HP_KERNEL_DEF(__hfma_sat);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hfma_sat(x,y,z)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_TERNARY_HP_TEST_DEF_IMPL(__hfma_sat, __hfma_sat_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_TERNARY_HP_KERNEL_DEF(__hfma2_sat);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hfma2_sat(x,y,z)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_arithmetic.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_TERNARY_HP_TEST_DEF_IMPL(__hfma2_sat, __hfma_sat_ref, EqValidatorBuilderFactory<float>());
|
||||
@@ -0,0 +1,124 @@
|
||||
/*
|
||||
Copyright (c) 2022 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip/hip_fp16.h>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
|
||||
#define UNARY_HALF_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half* x) { __half result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { __half result = func_name(x); }
|
||||
|
||||
#define BINARY_HALF_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half* x, __half y) { __half result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(__half x, __half* y) { __half result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, __half y) { __half result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(__half x, Dummy y) { __half result = func_name(x, y); }
|
||||
|
||||
#define TERNARY_HALF_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half* x, __half y, __half z) { \
|
||||
__half result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(__half x, __half* y, __half z) { \
|
||||
__half result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(__half x, __half y, __half* z) { \
|
||||
__half result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v4(Dummy x, __half y, __half z) { \
|
||||
__half result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v5(__half x, Dummy y, __half z) { \
|
||||
__half result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v6(__half x, __half y, Dummy z) { \
|
||||
__half result = func_name(x, y, z); \
|
||||
}
|
||||
|
||||
UNARY_HALF_NEGATIVE_KERNELS(__habs)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(__hneg)
|
||||
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hadd)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hadd_sat)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hsub)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hsub_sat)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hmul)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hmul_sat)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hdiv)
|
||||
|
||||
TERNARY_HALF_NEGATIVE_KERNELS(__hfma)
|
||||
TERNARY_HALF_NEGATIVE_KERNELS(__hfma_sat)
|
||||
|
||||
|
||||
#define UNARY_HALF2_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half2* x) { __half2 result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { __half2 result = func_name(x); }
|
||||
|
||||
#define BINARY_HALF2_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half2* x, __half2 y) { \
|
||||
__half2 result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(__half2 x, __half2* y) { \
|
||||
__half2 result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, __half2 y) { __half2 result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(__half2 x, Dummy y) { __half2 result = func_name(x, y); }
|
||||
|
||||
#define TERNARY_HALF2_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half2* x, __half2 y, __half2 z) { \
|
||||
__half2 result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(__half2 x, __half2* y, __half2 z) { \
|
||||
__half2 result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(__half2 x, __half2 y, __half2* z) { \
|
||||
__half2 result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v4(Dummy x, __half2 y, __half2 z) { \
|
||||
__half2 result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v5(__half2 x, Dummy y, __half2 z) { \
|
||||
__half2 result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v6(__half2 x, __half2 y, Dummy z) { \
|
||||
__half2 result = func_name(x, y, z); \
|
||||
}
|
||||
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(__habs2)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(__hneg2)
|
||||
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hadd2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hadd2_sat)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hsub2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hsub2_sat)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hmul2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hmul2_sat)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__h2div)
|
||||
|
||||
TERNARY_HALF2_NEGATIVE_KERNELS(__hfma2)
|
||||
TERNARY_HALF2_NEGATIVE_KERNELS(__hfma2_sat)
|
||||
@@ -0,0 +1,103 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "binary_common.hh"
|
||||
#include "ternary_common.hh"
|
||||
|
||||
|
||||
/********** Unary **********/
|
||||
|
||||
#define MATH_UNARY_HP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(Float16* const ys, const size_t num_xs, Float16* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(xs[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_UNARY_HP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
UnaryHalfPrecisionTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_UNARY_HP_TEST_DEF(func_name, ref_func) \
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_UNARY_HP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x)
|
||||
|
||||
|
||||
/********** Binary **********/
|
||||
|
||||
#define MATH_BINARY_HP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(Float16* const ys, const size_t num_xs, Float16* const x1s, \
|
||||
Float16* const x2s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_BINARY_HP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
BinaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_BINARY_HP_TEST_DEF(func_name, ref_func) \
|
||||
MATH_BINARY_HP_TEST_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_BINARY_HP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x1, \
|
||||
float x2)
|
||||
|
||||
|
||||
/********** Ternary **********/
|
||||
|
||||
#define MATH_TERNARY_HP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(Float16* const ys, const size_t num_xs, Float16* const x1s, \
|
||||
Float16* const x2s, Float16* const x3s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i], x3s[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_TERNARY_HP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
TernaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_TERNARY_HP_TEST_DEF(func_name, ref_func, validator_builder) \
|
||||
MATH_TERNARY_HP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_TERNARY_HP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x1, \
|
||||
float x2, float x3)
|
||||
@@ -0,0 +1,847 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "half_precision_common.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup HalfPrecisionComparison HalfPrecisionComparison
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
/********** Unary Functions **********/
|
||||
|
||||
#define MATH_BOOL_UNARY_HP_TEST_DEF(func_name, ref_func) \
|
||||
__global__ void func_name##_kernel(bool* const ys, const size_t num_xs, Float16* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(xs[i]); \
|
||||
} \
|
||||
} \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
UnaryHalfPrecisionTest(func_name##_kernel, ref_func, EqValidatorBuilderFactory<bool>()); \
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hisinf(x)` for all possible inputs. The results are
|
||||
* compared against reference function `bool std::isinf(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BOOL_UNARY_HP_TEST_DEF(__hisinf, static_cast<bool (*)(float)>(std::isinf))
|
||||
|
||||
static float __hisinf2_ref(float x) { return static_cast<float>(std::isinf(x)); }
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(__hisinf2)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hisinf2(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::isinf(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(__hisinf2, __hisinf2_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hisnan(x)` for all possible inputs. The results are
|
||||
* compared against reference function `bool std::isnan(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BOOL_UNARY_HP_TEST_DEF(__hisnan, static_cast<bool (*)(float)>(std::isnan))
|
||||
|
||||
static float __hisnan2_ref(float x) { return static_cast<float>(std::isnan(x)); }
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(__hisnan2)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hisnan2(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::isnan(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(__hisnan2, __hisnan2_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
/********** Binary Functions **********/
|
||||
|
||||
#define MATH_COMPARISON_HP_TEST_DEF(func_name, ref_func, T, RT, nan_value) \
|
||||
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, Float16* const x1s, \
|
||||
Float16* const x2s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i]); \
|
||||
} \
|
||||
} \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
BinaryFloatingPointTest(func_name##_kernel, ref_func<nan_value, RT>, \
|
||||
EqValidatorBuilderFactory<RT>()); \
|
||||
}
|
||||
|
||||
|
||||
template <bool nan_value, typename T> static T __heq_ref(float x1, float x2) {
|
||||
if (std::isnan(x1) || std::isnan(x2)) {
|
||||
return static_cast<T>(nan_value);
|
||||
}
|
||||
return x1 == x2;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__heq(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'equal
|
||||
* to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__heq, __heq_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbeq2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'equal
|
||||
* to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbeq2, __heq_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hequ(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'equal
|
||||
* to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hequ, __heq_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbequ2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are compared against result
|
||||
* of 'equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbequ2, __heq_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__heq2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'equal
|
||||
* to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__heq2, __heq_ref, Float16, float, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hequ2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'equal
|
||||
* to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hequ2, __heq_ref, Float16, float, true)
|
||||
|
||||
|
||||
template <bool nan_value, typename T> static T __hne_ref(float x1, float x2) {
|
||||
if (std::isnan(x1) || std::isnan(x2)) {
|
||||
return static_cast<T>(nan_value);
|
||||
}
|
||||
return x1 != x2;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hne(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'not
|
||||
* equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hne, __hne_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbne2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'not
|
||||
* equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbne2, __hne_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hneu(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'not
|
||||
* equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hneu, __hne_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbneu2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are compared against result
|
||||
* of 'not equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbneu2, __hne_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hne2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'not
|
||||
* equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hne2, __hne_ref, Float16, float, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hneu2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'not
|
||||
* equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hneu2, __hne_ref, Float16, float, true)
|
||||
|
||||
|
||||
template <bool nan_value, typename T> static T __hge_ref(float x1, float x2) {
|
||||
if (std::isnan(x1) || std::isnan(x2)) {
|
||||
return static_cast<T>(nan_value);
|
||||
}
|
||||
return x1 >= x2;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hge(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hge, __hge_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbge2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbge2, __hge_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hgeu(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hgeu, __hge_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbgeu2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are compared against result
|
||||
* of 'greater than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbgeu2, __hge_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hge2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hge2, __hge_ref, Float16, float, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hgeu2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hgeu2, __hge_ref, Float16, float, true)
|
||||
|
||||
|
||||
template <bool nan_value, typename T> static T __hgt_ref(float x1, float x2) {
|
||||
if (std::isnan(x1) || std::isnan(x2)) {
|
||||
return static_cast<T>(nan_value);
|
||||
}
|
||||
return x1 > x2;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hgt(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hgt, __hgt_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbgt2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbgt2, __hgt_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hgtu(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hgtu, __hgt_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbgtu2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are compared against result
|
||||
* of 'greater than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbgtu2, __hgt_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hgt2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hgt2, __hgt_ref, Float16, float, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hgtu2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of
|
||||
* 'greater than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hgtu2, __hgt_ref, Float16, float, true)
|
||||
|
||||
|
||||
template <bool nan_value, typename T> static T __hle_ref(float x1, float x2) {
|
||||
if (std::isnan(x1) || std::isnan(x2)) {
|
||||
return static_cast<T>(nan_value);
|
||||
}
|
||||
return x1 <= x2;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hle(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hle, __hle_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hble2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hble2, __hle_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hleu(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hleu, __hle_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbleu2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are compared against result
|
||||
* of 'less than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbleu2, __hle_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hle2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hle2, __hle_ref, Float16, float, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hleu2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than equal to' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hleu2, __hle_ref, Float16, float, true)
|
||||
|
||||
|
||||
template <bool nan_value, typename T> static T __hlt_ref(float x1, float x2) {
|
||||
if (std::isnan(x1) || std::isnan(x2)) {
|
||||
return static_cast<T>(nan_value);
|
||||
}
|
||||
return x1 < x2;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hlt(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hlt, __hlt_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hblt2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hblt2, __hlt_ref, bool, bool, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hltu(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hltu, __hlt_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hbltu2(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are compared against result
|
||||
* of 'less than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hbltu2, __hlt_ref, bool, bool, true)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hlt2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hlt2, __hlt_ref, Float16, float, false)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hltu2(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against result of 'less
|
||||
* than' relational operator for float operands.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_COMPARISON_HP_TEST_DEF(__hltu2, __hlt_ref, Float16, float, true)
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hmax)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hmax(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against reference
|
||||
* function `float std::fmax(float, float)`
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hmax, static_cast<float (*)(float, float)>(std::fmax),
|
||||
EqValidatorBuilderFactory<float>())
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hmin)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hmin(x,y)` against a table of difficult values, followed
|
||||
* by a large number of randomly generated values. The results are compared against reference
|
||||
* function `float std::fmin(float, float)`
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hmin, static_cast<float (*)(float, float)>(std::fmin),
|
||||
EqValidatorBuilderFactory<float>())
|
||||
|
||||
static float __hmax_nan_ref(float x1, float x2) {
|
||||
if (std::isnan(x1))
|
||||
return x1;
|
||||
else if (std::isnan(x2))
|
||||
return x2;
|
||||
else
|
||||
return std::fmax(x1, x2);
|
||||
}
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hmax_nan)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hmax_nan(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are compared against
|
||||
* reference function `float std::fmax(float, float)` with modified result when an operand is nan.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hmax_nan, __hmax_nan_ref, EqValidatorBuilderFactory<float>())
|
||||
|
||||
static float __hmin_nan_ref(float x1, float x2) {
|
||||
if (std::isnan(x1))
|
||||
return x1;
|
||||
else if (std::isnan(x2))
|
||||
return x2;
|
||||
else
|
||||
return std::fmin(x1, x2);
|
||||
}
|
||||
|
||||
MATH_BINARY_HP_KERNEL_DEF(__hmin_nan)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__hmin_nan(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are compared against
|
||||
* reference function `float std::fmin(float, float)` with modified result when an operand is nan.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_comparison.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_HP_TEST_DEF_IMPL(__hmin_nan, __hmin_nan_ref, EqValidatorBuilderFactory<float>())
|
||||
@@ -0,0 +1,120 @@
|
||||
/*
|
||||
Copyright (c) 2022 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip/hip_fp16.h>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
|
||||
#define UNARY_BOOL_HALF_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half* x) { bool result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { bool result = func_name(x); }
|
||||
|
||||
#define BINARY_BOOL_HALF_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half* x, __half y) { bool result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(__half x, __half* y) { bool result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, __half y) { bool result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(__half x, Dummy y) { bool result = func_name(x, y); }
|
||||
|
||||
|
||||
#define BINARY_HALF_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half* x, __half y) { __half result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(__half x, __half* y) { __half result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, __half y) { __half result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(__half x, Dummy y) { __half result = func_name(x, y); }
|
||||
|
||||
|
||||
UNARY_BOOL_HALF_NEGATIVE_KERNELS(__hisinf)
|
||||
UNARY_BOOL_HALF_NEGATIVE_KERNELS(__hisnan)
|
||||
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__heq)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hequ)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hne)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hneu)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hge)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hgeu)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hgt)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hgtu)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hle)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hleu)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hlt)
|
||||
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hltu)
|
||||
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hmax)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hmax_nan)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hmin)
|
||||
BINARY_HALF_NEGATIVE_KERNELS(__hmin_nan)
|
||||
|
||||
|
||||
#define UNARY_HALF2_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half2* x) { __half2 result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { __half2 result = func_name(x); }
|
||||
|
||||
#define BINARY_HALF2_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half2* x, __half2 y) { \
|
||||
__half2 result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(__half2 x, __half2* y) { \
|
||||
__half2 result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, __half2 y) { __half2 result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(__half2 x, Dummy y) { __half2 result = func_name(x, y); }
|
||||
|
||||
#define BINARY_BOOL_HALF2_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half2* x, __half2 y) { bool result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(__half2 x, __half2* y) { bool result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, __half2 y) { bool result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(__half2 x, Dummy y) { bool result = func_name(x, y); }
|
||||
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(__hisinf2)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(__hisnan2)
|
||||
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__heq2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hequ2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hne2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hneu2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hge2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hgeu2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hgt2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hgtu2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hle2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hleu2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hlt2)
|
||||
BINARY_HALF2_NEGATIVE_KERNELS(__hltu2)
|
||||
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbeq2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbequ2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbne2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbneu2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbge2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbgeu2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbgt2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbgtu2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hble2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbleu2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hblt2)
|
||||
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbltu2)
|
||||
@@ -0,0 +1,580 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "half_precision_common.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup HalfPrecisionMath HalfPrecisionMath
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hcos);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hcos(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::cos(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hcos, static_cast<float (*)(float)>(std::cos),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2cos);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2cos(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::cos(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2cos, static_cast<float (*)(float)>(std::cos),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hsin);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hsin(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::sin(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hsin, static_cast<float (*)(float)>(std::sin),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2sin);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2sin(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::sin(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2sin, static_cast<float (*)(float)>(std::sin),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hexp);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hexp(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::exp(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hexp, static_cast<float (*)(float)>(std::exp),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2exp);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2exp(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::exp(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2exp, static_cast<float (*)(float)>(std::exp),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hexp10);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hexp10(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float exp10(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hexp10, static_cast<float (*)(float)>(exp10f),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2exp10);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2exp10(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float exp10(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2exp10, static_cast<float (*)(float)>(exp10f),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hexp2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hexp2(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::exp2(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hexp2, static_cast<float (*)(float)>(std::exp2),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2exp2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2exp2(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::exp2(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2exp2, static_cast<float (*)(float)>(std::exp2),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hlog);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hlog(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::log(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hlog, static_cast<float (*)(float)>(std::log),
|
||||
ULPValidatorBuilderFactory<float>(1));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2log);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2log(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::log(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2log, static_cast<float (*)(float)>(std::log),
|
||||
ULPValidatorBuilderFactory<float>(1));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hlog10);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hlog10(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::log10(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hlog10, static_cast<float (*)(float)>(std::log10),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2log10);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2log10(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::log10(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2log10, static_cast<float (*)(float)>(std::log10),
|
||||
ULPValidatorBuilderFactory<float>(2));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hlog2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hlog2(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::log2(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hlog2, static_cast<float (*)(float)>(std::log2),
|
||||
ULPValidatorBuilderFactory<float>(1));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2log2);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2log2(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::log2(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2log2, static_cast<float (*)(float)>(std::log2),
|
||||
ULPValidatorBuilderFactory<float>(1));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hsqrt);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hsqrt(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::sqrt(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hsqrt, static_cast<float (*)(float)>(std::sqrt),
|
||||
ULPValidatorBuilderFactory<float>(1));
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2sqrt);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2sqrt(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::sqrt(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2sqrt, static_cast<float (*)(float)>(std::sqrt),
|
||||
ULPValidatorBuilderFactory<float>(1));
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hceil);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hceil(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::ceil(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hceil, static_cast<float (*)(float)>(std::ceil),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2ceil);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2ceil(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::ceil(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2ceil, static_cast<float (*)(float)>(std::ceil),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hfloor);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hfloor(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::floor(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hfloor, static_cast<float (*)(float)>(std::floor),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2floor);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2floor(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::floor(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2floor, static_cast<float (*)(float)>(std::floor),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(htrunc);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `htrunc(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::trunc(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(htrunc, static_cast<float (*)(float)>(std::trunc),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2trunc);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2trunc(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::trunc(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2trunc, static_cast<float (*)(float)>(std::trunc),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float hrcp_ref(float x) { return 1.0f / x; }
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hrcp);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hrcp(x)` for all possible inputs.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hrcp, hrcp_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2rcp);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2rcp(x)` for all possible inputs.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2rcp, hrcp_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float hrsqrt_ref(float x) { return 1.0f / std::sqrt(x); }
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hrsqrt);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hrsqrt(x)` for all possible inputs.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hrsqrt, hrsqrt_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2rsqrt);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2rsqrt(x)` for all possible inputs.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2rsqrt, hrsqrt_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(hrint);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hrint(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::rint(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(hrint, static_cast<float (*)(float)>(std::rint),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
MATH_UNARY_HP_KERNEL_DEF(h2rint);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `h2rint(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::rint(float)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/half_precision_math.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_HP_TEST_DEF_IMPL(h2rint, static_cast<float (*)(float)>(std::rint),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
@@ -0,0 +1,72 @@
|
||||
/*
|
||||
Copyright (c) 2022 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip/hip_fp16.h>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
|
||||
#define UNARY_HALF_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half* x) { __half result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { __half result = func_name(x); }
|
||||
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hcos)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hsin)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hexp)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hexp10)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hexp2)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hlog)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hlog10)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hlog2)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hsqrt)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hceil)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hfloor)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(htrunc)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hrcp)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hrsqrt)
|
||||
UNARY_HALF_NEGATIVE_KERNELS(hrint)
|
||||
|
||||
|
||||
#define UNARY_HALF2_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(__half2* x) { __half2 result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { __half2 result = func_name(x); }
|
||||
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2cos)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2sin)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2exp)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2exp10)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2exp2)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2log)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2log10)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2log2)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2sqrt)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2ceil)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2floor)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2trunc)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2rcp)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2rsqrt)
|
||||
UNARY_HALF2_NEGATIVE_KERNELS(h2rint)
|
||||
@@ -0,0 +1,794 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <resource_guards.hh>
|
||||
|
||||
__global__ void __brev_kernel(unsigned int* y, unsigned int x) { y[0] = __brev(x); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__brev(x)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___brev_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
__brev_kernel<<<1, 1>>>(y.ptr(), 0xAAAAAAAA);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 0x55555555);
|
||||
}
|
||||
|
||||
__global__ void __brevll_kernel(unsigned long long int* y, unsigned long long int x) {
|
||||
y[0] = __brevll(x);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__brevll(x)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___brevll_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned long long int> y(LinearAllocs::hipMallocManaged,
|
||||
sizeof(unsigned long long int));
|
||||
|
||||
__brevll_kernel<<<1, 1>>>(y.ptr(), 0xAAAAAAAAAAAAAAAA);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 0x5555555555555555);
|
||||
}
|
||||
|
||||
template <typename T> __global__ void __clz_kernel(T* y, T x) { y[0] = __clz(x); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__clz(x)`. Run for `int` and `unsigned int` overloads.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device___clz_Sanity_Positive", "", int, unsigned int) {
|
||||
LinearAllocGuard<TestType> y(LinearAllocs::hipMallocManaged, sizeof(TestType));
|
||||
|
||||
__clz_kernel<<<1, 1>>>(y.ptr(), static_cast<TestType>(0));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 32);
|
||||
|
||||
TestType x = 1;
|
||||
for (int i = 0; i < 32; ++i) {
|
||||
__clz_kernel<<<1, 1>>>(y.ptr(), x << i);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 31 - i);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> __global__ void __clzll_kernel(T* y, T x) { y[0] = __clzll(x); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__clzll(x)`. Run for `long long int` and `unsigned long long int`
|
||||
* overloads.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device___clzll_Sanity_Positive", "", long long int,
|
||||
unsigned long long int) {
|
||||
LinearAllocGuard<TestType> y(LinearAllocs::hipMallocManaged, sizeof(TestType));
|
||||
|
||||
__clzll_kernel<<<1, 1>>>(y.ptr(), static_cast<TestType>(0));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 64);
|
||||
|
||||
TestType x = 1;
|
||||
for (int i = 0; i < 64; ++i) {
|
||||
__clzll_kernel<<<1, 1>>>(y.ptr(), x << i);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 63 - i);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> __global__ void __ffs_kernel(T* y, T x) { y[0] = __ffs(x); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__ffs(x)`. Run for `int` and `unsigned int` overloads.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device___ffs_Sanity_Positive", "", int, unsigned int) {
|
||||
LinearAllocGuard<TestType> y(LinearAllocs::hipMallocManaged, sizeof(TestType));
|
||||
|
||||
__ffs_kernel<<<1, 1>>>(y.ptr(), static_cast<TestType>(0));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 0);
|
||||
|
||||
TestType x = 1;
|
||||
for (int i = 0; i < 32; ++i) {
|
||||
__ffs_kernel<<<1, 1>>>(y.ptr(), x << i);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == i + 1);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> __global__ void __ffsll_kernel(T* y, T x) { y[0] = __ffsll(x); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__ffsll(x)`. Run for `long long int` and `unsigned long long int`
|
||||
* overloads.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device___ffsll_Sanity_Positive", "", long long int,
|
||||
unsigned long long int) {
|
||||
LinearAllocGuard<TestType> y(LinearAllocs::hipMallocManaged, sizeof(TestType));
|
||||
|
||||
__ffsll_kernel<<<1, 1>>>(y.ptr(), static_cast<TestType>(0));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 0);
|
||||
|
||||
TestType x = 1;
|
||||
for (int i = 0; i < 64; ++i) {
|
||||
__ffsll_kernel<<<1, 1>>>(y.ptr(), x << i);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == i + 1);
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void __popc_kernel(unsigned int* y, unsigned int x) { y[0] = __popc(x); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__popc(x)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___popc_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
__popc_kernel<<<1, 1>>>(y.ptr(), 0);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 0);
|
||||
|
||||
unsigned int x = 0;
|
||||
for (int i = 0; i < 32; ++i) {
|
||||
__popc_kernel<<<1, 1>>>(y.ptr(), x |= (1u << i));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == i + 1);
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void __popcll_kernel(unsigned long long int* y, unsigned long long int x) {
|
||||
y[0] = __popcll(x);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__popcll(x)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___popcll_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned long long int> y(LinearAllocs::hipMallocManaged,
|
||||
sizeof(unsigned long long int));
|
||||
|
||||
__popcll_kernel<<<1, 1>>>(y.ptr(), 0);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == 0);
|
||||
|
||||
unsigned long long int x = 0;
|
||||
for (int i = 0; i < 64; ++i) {
|
||||
__popcll_kernel<<<1, 1>>>(y.ptr(), x |= (1ull << i));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == i + 1);
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void __mul24_kernel(int* y, int x1, int x2) { y[0] = __mul24(x1, x2); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__mul24(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___mul24_Sanity_Positive") {
|
||||
LinearAllocGuard<int> y(LinearAllocs::hipMallocManaged, sizeof(int));
|
||||
|
||||
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
|
||||
__mul24_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == x1 * x2);
|
||||
}
|
||||
|
||||
__global__ void __umul24_kernel(unsigned int* y, unsigned int x1, unsigned int x2) {
|
||||
y[0] = __umul24(x1, x2);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__umul24(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___umul24_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
unsigned int x1 = GENERATE(0, 42, 0xFFFFFF);
|
||||
unsigned int x2 = GENERATE(0, 42, 0xFFFFFF);
|
||||
|
||||
__umul24_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
REQUIRE(y.ptr()[0] == x1 * x2);
|
||||
}
|
||||
|
||||
__global__ void __funnelshift_l_kernel(unsigned int* y, unsigned int lo, unsigned int hi,
|
||||
unsigned int shift) {
|
||||
y[0] = __funnelshift_l(lo, hi, shift);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__funnelshift_l(lo,hi,shift)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___funnelshift_l_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
const unsigned int lo = 0xAAAAAAAA, hi = 0xBBBBBBBB;
|
||||
const unsigned long long hi_lo = (static_cast<unsigned long long>(hi) << 32) | lo;
|
||||
|
||||
for (unsigned int shift = 0; shift < 64; ++shift) {
|
||||
__funnelshift_l_kernel<<<1, 1>>>(y.ptr(), lo, hi, shift);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("shift: " << shift);
|
||||
REQUIRE(y.ptr()[0] == static_cast<unsigned int>((hi_lo << (shift & 31)) >> 32));
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void __funnelshift_lc_kernel(unsigned int* y, unsigned int lo, unsigned int hi,
|
||||
unsigned int shift) {
|
||||
y[0] = __funnelshift_lc(lo, hi, shift);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__funnelshift_lc(lo,hi,shift)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___funnelshift_lc_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
const unsigned int lo = 0xAAAAAAAA, hi = 0xBBBBBBBB;
|
||||
const unsigned long long hi_lo = (static_cast<unsigned long long>(hi) << 32) | lo;
|
||||
|
||||
for (unsigned int shift = 0; shift < 64; ++shift) {
|
||||
__funnelshift_lc_kernel<<<1, 1>>>(y.ptr(), lo, hi, shift);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("shift: " << shift);
|
||||
REQUIRE(y.ptr()[0] == static_cast<unsigned int>((hi_lo << std::min(shift, 32u)) >> 32));
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void __funnelshift_r_kernel(unsigned int* y, unsigned int lo, unsigned int hi,
|
||||
unsigned int shift) {
|
||||
y[0] = __funnelshift_r(lo, hi, shift);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__funnelshift_r(lo,hi,shift)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___funnelshift_r_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
const unsigned int lo = 0xAAAAAAAA, hi = 0xBBBBBBBB;
|
||||
const unsigned long long hi_lo = (static_cast<unsigned long long>(hi) << 32) | lo;
|
||||
|
||||
for (unsigned int shift = 0; shift < 64; ++shift) {
|
||||
__funnelshift_r_kernel<<<1, 1>>>(y.ptr(), lo, hi, shift);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("shift: " << shift);
|
||||
REQUIRE(y.ptr()[0] == static_cast<unsigned int>(hi_lo >> (shift & 31)));
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void __funnelshift_rc_kernel(unsigned int* y, unsigned int lo, unsigned int hi,
|
||||
unsigned int shift) {
|
||||
y[0] = __funnelshift_rc(lo, hi, shift);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__funnelshift_rc(lo,hi,shift)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___funnelshift_rc_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
const unsigned int lo = 0xAAAAAAAA, hi = 0xBBBBBBBB;
|
||||
const unsigned long long hi_lo = (static_cast<unsigned long long>(hi) << 32) | lo;
|
||||
|
||||
for (unsigned int shift = 0; shift < 64; ++shift) {
|
||||
__funnelshift_rc_kernel<<<1, 1>>>(y.ptr(), lo, hi, shift);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("shift: " << shift);
|
||||
REQUIRE(y.ptr()[0] == static_cast<unsigned int>(hi_lo >> std::min(shift, 32u)));
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void __hadd_kernel(int* y, int x1, int x2) { y[0] = __hadd(x1, x2); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__hadd(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___hadd_Sanity_Positive") {
|
||||
LinearAllocGuard<int> y(LinearAllocs::hipMallocManaged, sizeof(int));
|
||||
|
||||
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
|
||||
__hadd_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] == static_cast<int>((static_cast<long long>(x1) + x2) >> 1));
|
||||
}
|
||||
|
||||
__global__ void __uhadd_kernel(unsigned int* y, unsigned int x1, unsigned int x2) {
|
||||
y[0] = __uhadd(x1, x2);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__uhadd(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___uhadd_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
unsigned int x1 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
unsigned int x2 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
|
||||
__uhadd_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] == static_cast<unsigned int>((static_cast<unsigned long long>(x1) + x2) >> 1));
|
||||
}
|
||||
|
||||
__global__ void __rhadd_kernel(int* y, int x1, int x2) { y[0] = __rhadd(x1, x2); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__rhadd(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___rhadd_Sanity_Positive") {
|
||||
LinearAllocGuard<int> y(LinearAllocs::hipMallocManaged, sizeof(int));
|
||||
|
||||
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
|
||||
__rhadd_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] == static_cast<int>((static_cast<long long>(x1) + x2 + 1) >> 1));
|
||||
}
|
||||
|
||||
__global__ void __urhadd_kernel(unsigned int* y, unsigned int x1, unsigned int x2) {
|
||||
y[0] = __urhadd(x1, x2);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__urhadd(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___urhadd_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
unsigned int x1 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
unsigned int x2 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
|
||||
__urhadd_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] ==
|
||||
static_cast<unsigned int>((static_cast<unsigned long long>(x1) + x2 + 1) >> 1));
|
||||
}
|
||||
|
||||
__global__ void __mulhi_kernel(int* y, int x1, int x2) { y[0] = __mulhi(x1, x2); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__mulhi(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___mulhi_Sanity_Positive") {
|
||||
LinearAllocGuard<int> y(LinearAllocs::hipMallocManaged, sizeof(int));
|
||||
|
||||
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
|
||||
__mulhi_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] ==
|
||||
static_cast<int>((static_cast<long long>(x1) * static_cast<long long>(x2)) >> 32));
|
||||
}
|
||||
|
||||
__global__ void __umulhi_kernel(unsigned int* y, unsigned int x1, unsigned int x2) {
|
||||
y[0] = __umulhi(x1, x2);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__umulhi(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___umulhi_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
unsigned int x1 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
unsigned int x2 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
|
||||
__umulhi_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] ==
|
||||
static_cast<unsigned int>((static_cast<unsigned long long>(x1) * x2) >> 32));
|
||||
}
|
||||
|
||||
__global__ void __mul64hi_kernel(long long* y, long long x1, long long x2) {
|
||||
y[0] = __mul64hi(x1, x2);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__mul64hi(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___mul64hi_Sanity_Positive") {
|
||||
LinearAllocGuard<long long> y(LinearAllocs::hipMallocManaged, sizeof(long long));
|
||||
|
||||
long long x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
long long x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
|
||||
__mul64hi_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(
|
||||
y.ptr()[0] ==
|
||||
static_cast<long long>((static_cast<__int128_t>(x1) * static_cast<__int128_t>(x2)) >> 64));
|
||||
}
|
||||
|
||||
__global__ void __umul64hi_kernel(unsigned long long* y, unsigned long long x1,
|
||||
unsigned long long x2) {
|
||||
y[0] = __umul64hi(x1, x2);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__umul64hi(x,y)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___umul64hi_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned long long> y(LinearAllocs::hipMallocManaged,
|
||||
sizeof(unsigned long long));
|
||||
|
||||
unsigned long long x1 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
unsigned long long x2 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
|
||||
__umul64hi_kernel<<<1, 1>>>(y.ptr(), x1, x2);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] ==
|
||||
static_cast<unsigned long long>(
|
||||
(static_cast<__uint128_t>(x1) * static_cast<__uint128_t>(x2)) >> 64));
|
||||
}
|
||||
|
||||
__global__ void __sad_kernel(unsigned int* y, int x1, int x2, unsigned int x3) {
|
||||
y[0] = __sad(x1, x2, x3);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__sad(x,y,z)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___sad_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
|
||||
unsigned int x3 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
|
||||
__sad_kernel<<<1, 1>>>(y.ptr(), x1, x2, x3);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] == (static_cast<unsigned int>(std::abs(x1 - x2)) + x3));
|
||||
}
|
||||
|
||||
__global__ void __usad_kernel(unsigned int* y, unsigned int x1, unsigned int x2, unsigned int x3) {
|
||||
y[0] = __usad(x1, x2, x3);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__usad(x,y,z)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___usad_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
unsigned int x1 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
unsigned int x2 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
unsigned int x3 = GENERATE(0, 42, 0xFFFFFFFF);
|
||||
|
||||
__usad_kernel<<<1, 1>>>(y.ptr(), x1, x2, x3);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
INFO("x1: " << x1);
|
||||
INFO("x2: " << x2);
|
||||
REQUIRE(y.ptr()[0] ==
|
||||
(static_cast<unsigned int>(
|
||||
std::abs(static_cast<long long>(x1) - static_cast<long long>(x2))) +
|
||||
x3));
|
||||
}
|
||||
|
||||
__global__ void __byte_perm(unsigned int* y, unsigned int x1, unsigned int x2, unsigned int s) {
|
||||
y[0] = __byte_perm(x1, x2, s);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `__byte_perm(x,y,s)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/integer_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device___byte_perm_Sanity_Positive") {
|
||||
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
|
||||
|
||||
unsigned int bytes[] = {0x88, 0x99, 0xAA, 0xBB, 0xCC, 0xDD, 0xEE, 0xFF};
|
||||
|
||||
unsigned int x1 = (bytes[3] << 24) | (bytes[2] << 16) | (bytes[1] << 8) | bytes[0];
|
||||
unsigned int x2 = (bytes[7] << 24) | (bytes[6] << 16) | (bytes[5] << 8) | bytes[4];
|
||||
|
||||
unsigned int s0 = GENERATE(0, 1);
|
||||
unsigned int s1 = GENERATE(2, 3);
|
||||
unsigned int s2 = GENERATE(4, 5);
|
||||
unsigned int s3 = GENERATE(6, 7);
|
||||
|
||||
unsigned int s = (s3 << 12) | (s2 << 8) | (s1 << 4) | s0;
|
||||
|
||||
__byte_perm<<<1, 1>>>(y.ptr(), x1, x2, s);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
unsigned int expected = (bytes[s3] << 24) | (bytes[s2] << 16) | (bytes[s1] << 8) | bytes[s0];
|
||||
REQUIRE(y.ptr()[0] == expected);
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(int* x) { int result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { int result = func_name(x); }
|
||||
|
||||
#define INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(long long int* x) { long long int result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { long long int result = func_name(x); }
|
||||
|
||||
#define INTRINSIC_BINARY_INT_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(int* x, int y) { int result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(int x, int* y) { int result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, int y) { int result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(int x, Dummy y) { int result = func_name(x, y); }
|
||||
|
||||
#define INTRINSIC_BINARY_LONGLONG_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(long long int* x, long long int y) { \
|
||||
long long int result = func_name(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(long long int x, long long int* y) { \
|
||||
long long int result = func_name##(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, long long int y) { \
|
||||
long long int result = func_name##(x, y); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v4(long long int x, Dummy y) { \
|
||||
long long int result = func_name##(x, y); \
|
||||
}
|
||||
|
||||
INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(__brev)
|
||||
INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(__clz)
|
||||
INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(__ffs)
|
||||
INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(__popc)
|
||||
INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(__brevll)
|
||||
INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(__clzll)
|
||||
INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(__ffsll)
|
||||
INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(__popcll)
|
||||
INTRINSIC_BINARY_INT_NEGATIVE_KERNELS(__mul24)
|
||||
@@ -0,0 +1,260 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "math_log_negative_kernels_rtc.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup LogMathFuncs LogMathFuncs
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
/********** Unary Functions **********/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `logf(x)` for all possible inputs and `log(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::log(T)`. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(log, 1, 1)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for logf and log.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_log_logf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLog); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `log2f(x)` for all possible inputs and `log2(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::log2(T)`. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(log2, 1, 1)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for log2f and log2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_log2_log2f_Negative_RTC") { NegativeTestRTCWrapper<4>(kLog2); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `log10f(x)` for all possible inputs and `log10(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::log10(T)`. The maximum ulp error for single
|
||||
* precision is 2 and for double precision is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(log10, 2, 1)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for log10f and log10.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_log10_log10f_Negative_RTC") { NegativeTestRTCWrapper<4>(kLog10); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `log1pf(x)` for all possible inputs and `log1p(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::log1p(T)`. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(log1p, 1, 1)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for log1pf and log1p.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_log1p_log1pf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLog1p); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `logb(x)` for all possible inputs and `logb(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::logb(T)`. The maximum ulp error is 0.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(logb, 0, 0)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for logbf and logb.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_logb_logbf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLogb); }
|
||||
|
||||
|
||||
template <typename T>
|
||||
__global__ void ilogb_kernel(int* const ys, const size_t num_xs, T* const xs) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
if constexpr (std::is_same_v<float, T>) {
|
||||
ys[i] = ilogbf(xs[i]);
|
||||
} else if constexpr (std::is_same_v<double, T>) {
|
||||
ys[i] = ilogb(xs[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> int ilogb_ref(T arg) {
|
||||
if (arg == 0) {
|
||||
return std::numeric_limits<int>::min();
|
||||
} else if (std::isnan(arg)) {
|
||||
return std::numeric_limits<int>::min();
|
||||
} else if (std::isinf(arg)) {
|
||||
return std::numeric_limits<int>::max();
|
||||
} else {
|
||||
return std::ilogb(arg);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `ilogbf(x)` for all possible inputs. The results are
|
||||
* compared against reference function `int std::ilogb(double)`. The maximum ulp error is 0.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_ilogbf_Accuracy_Positive") {
|
||||
UnarySinglePrecisionTest(ilogb_kernel<float>, ilogb_ref<double>,
|
||||
EqValidatorBuilderFactory<int>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `ilogb(x)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are
|
||||
* compared against reference function `int std::ilogb(long double)`. The maximum ulp error is 0.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_ilogb_Accuracy_Positive") {
|
||||
UnaryDoublePrecisionTest(ilogb_kernel<double>, ilogb_ref<long double>,
|
||||
EqValidatorBuilderFactory<int>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for ilogbf and ilogb.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/log_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_ilogb_ilogbf_Negative_RTC") { NegativeTestRTCWrapper<4>(kIlogb); }
|
||||
@@ -0,0 +1,256 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cmd_options.hh>
|
||||
#include <hip_test_common.hh>
|
||||
#include <resource_guards.hh>
|
||||
|
||||
#include <hip/hip_cooperative_groups.h>
|
||||
|
||||
#include "Float16.hh"
|
||||
#include "thread_pool.hh"
|
||||
#include "validators.hh"
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
template <typename T, typename U>
|
||||
std::enable_if_t<std::conjunction_v<std::is_arithmetic<T>, std::is_arithmetic<U>>, std::ostream&>
|
||||
operator<<(std::ostream& os, const std::pair<T, U>& p) {
|
||||
const auto default_prec = os.precision();
|
||||
return os << "<" << std::setprecision(std::numeric_limits<T>::max_digits10 - 1) << p.first << ", "
|
||||
<< std::setprecision(std::numeric_limits<U>::max_digits10 - 1) << p.second << ">"
|
||||
<< std::setprecision(default_prec);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
std::enable_if_t<sizeof(T) / sizeof(decltype(T().x)) == 2 && !std::is_same_v<T, __half2>, std::ostream&>
|
||||
operator<<(std::ostream& os, const T& p) {
|
||||
const auto default_prec = os.precision();
|
||||
return os << "<" << std::setprecision(std::numeric_limits<decltype(T().x)>::max_digits10 - 1) << p.x << ", "
|
||||
<< std::setprecision(std::numeric_limits<decltype(T().x)>::max_digits10 - 1) << p.y << ">"
|
||||
<< std::setprecision(default_prec);
|
||||
}
|
||||
|
||||
// This class represents a generic numerical accuracy math test. Template parameter T is the output
|
||||
// type of the function being tested, and template parameter pack Ts represents the input types. The
|
||||
// constructor takes a kernel with the signature void(T*, const size_t, Ts*...). The first kernel
|
||||
// parameter is the output array, the second parameter is the number of outputs, and the rest of the
|
||||
// parameters are arrays containing input values. The number of input arrays depends on the arity of
|
||||
// the function being tested e.g. one input array for unary functions, two input arrays for binary
|
||||
// functions, etc. The kernel threads take one element from each input array at the index
|
||||
// corresponding to that thread, feed the input elements to the testee function, and store the
|
||||
// result in the output array at the corresponding index.
|
||||
//
|
||||
// E.g. for a binary function the kernel would have the following signature:
|
||||
// void kernel(float* y, const size_t n, float* x1, float* x2)
|
||||
//
|
||||
// The outputs would be calculated in parallel the following way:
|
||||
// y[0] = testee(x1[0], x2[0])
|
||||
// y[1] = testee(x1[1], x2[1])
|
||||
// y[2] = testee(x1[2], x2[2])
|
||||
// ...
|
||||
//
|
||||
// The constructor also takes max_num_args, which represents the maximum number of input values used
|
||||
// for one kernel launch. The device memory for the input and output arrays is allocated based on
|
||||
// that number.
|
||||
template <typename T, typename... Ts> class MathTest {
|
||||
public:
|
||||
MathTest(void (*kernel)(T*, const size_t, Ts*...), const size_t max_num_args)
|
||||
: kernel_{kernel},
|
||||
xss_dev_(LinearAllocGuard<Ts>(LinearAllocs::hipMalloc, max_num_args * sizeof(Ts))...),
|
||||
y_dev_{LinearAllocs::hipMalloc, max_num_args * sizeof(T)},
|
||||
y_{LinearAllocs::hipHostMalloc, max_num_args * sizeof(T)} {}
|
||||
|
||||
// This method runs the test with the following steps:
|
||||
// 1. Copy the values from the input arrays provided in the parameter pack xss to device memory
|
||||
// 2. Launch the kernel using the configuration provided in grid_dims and block_dims
|
||||
// 3. Copy the outputs back to host memory
|
||||
// 4. Generate the reference values using ref_func and compare against the outputs using the
|
||||
// validator provided by validator_builder
|
||||
// 5. If non-type template parameter parallel is true, then step 4 is broken up into chunks of
|
||||
// work that are done in parallel on the host.
|
||||
template <bool parallel = true, typename RT, typename ValidatorBuilder, typename... RTs>
|
||||
void Run(const ValidatorBuilder& validator_builder, const size_t grid_dims,
|
||||
const size_t block_dims, RT (*const ref_func)(RTs...), const size_t num_args,
|
||||
const Ts*... xss) {
|
||||
fail_flag_.store(false);
|
||||
error_info_.clear();
|
||||
RunImpl<parallel>(validator_builder, grid_dims, block_dims, ref_func, num_args,
|
||||
std::index_sequence_for<Ts...>{}, xss...);
|
||||
}
|
||||
|
||||
private:
|
||||
void (*kernel_)(T*, const size_t, Ts*...);
|
||||
std::tuple<LinearAllocGuard<Ts>...> xss_dev_;
|
||||
LinearAllocGuard<T> y_dev_;
|
||||
LinearAllocGuard<T> y_;
|
||||
std::atomic<bool> fail_flag_{false};
|
||||
std::mutex mtx_;
|
||||
std::string error_info_;
|
||||
|
||||
template <bool parallel, typename RT, typename ValidatorBuilder, typename... RTs, size_t... I>
|
||||
void RunImpl(const ValidatorBuilder& validator_builder, const size_t grid_dim,
|
||||
const size_t block_dim, RT (*const ref_func)(RTs...), const size_t num_args,
|
||||
std::index_sequence<I...>, const Ts*... xss) {
|
||||
const auto xss_tup = std::make_tuple(xss...);
|
||||
|
||||
constexpr auto f = [](auto dst, auto src, size_t size) {
|
||||
HIP_CHECK(hipMemcpy(dst, src, size, hipMemcpyHostToDevice))
|
||||
};
|
||||
|
||||
((f(std::get<I>(xss_dev_).ptr(), std::get<I>(xss_tup),
|
||||
num_args * sizeof(*std::get<I>(xss_tup)))),
|
||||
...);
|
||||
|
||||
kernel_<<<grid_dim, block_dim>>>(y_dev_.ptr(), num_args, std::get<I>(xss_dev_).ptr()...);
|
||||
HIP_CHECK(hipGetLastError());
|
||||
|
||||
HIP_CHECK(hipMemcpy(y_.ptr(), y_dev_.ptr(), num_args * sizeof(T), hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipStreamSynchronize(nullptr));
|
||||
|
||||
if constexpr (!parallel) {
|
||||
for (auto i = 0u; i < num_args; ++i) {
|
||||
const auto actual_val = y_.ptr()[i];
|
||||
const auto ref_val = static_cast<T>(ref_func(xss[i]...));
|
||||
const auto validator = validator_builder(ref_val, xss[i]...);
|
||||
|
||||
if (!validator->match(actual_val)) {
|
||||
const auto log = MakeLogMessage(actual_val, xss[i]...) + validator->describe() + "\n";
|
||||
INFO(log);
|
||||
REQUIRE(false);
|
||||
}
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
const auto task = [&, this](size_t iters, size_t base_idx) {
|
||||
for (auto i = 0u; i < iters; ++i) {
|
||||
if (fail_flag_.load(std::memory_order_relaxed)) return;
|
||||
|
||||
const auto actual_val = y_.ptr()[base_idx + i];
|
||||
const auto ref_val = static_cast<T>(ref_func(xss[base_idx + i]...));
|
||||
const auto validator = validator_builder(ref_val, xss[base_idx + i]...);
|
||||
|
||||
if (!validator->match(actual_val)) {
|
||||
fail_flag_.store(true, std::memory_order_relaxed);
|
||||
// Several threads might have passed the first check, but failed validation. On the
|
||||
// chance of this happening, access to the string stream must be serialized.
|
||||
const auto log =
|
||||
MakeLogMessage(actual_val, xss[base_idx + i]...) + validator->describe() + "\n";
|
||||
{
|
||||
std::lock_guard lg{mtx_};
|
||||
error_info_ += log;
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const auto task_count = thread_pool.thread_count();
|
||||
const auto chunk_size = num_args / task_count;
|
||||
const auto tail = num_args % task_count;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < task_count; ++i) {
|
||||
const auto iters = chunk_size + (i < tail);
|
||||
thread_pool.Post([=, &task] { task(iters, base_idx); });
|
||||
base_idx += iters;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
INFO(error_info_);
|
||||
REQUIRE(!fail_flag_);
|
||||
}
|
||||
|
||||
template <typename... Args> std::string MakeLogMessage(T actual_val, Args... args) {
|
||||
std::stringstream ss;
|
||||
ss << "Input value(s): " << std::scientific
|
||||
<< std::setprecision(std::numeric_limits<T>::max_digits10 - 1);
|
||||
((ss << " " << args), ...) << "\n" << actual_val << " ";
|
||||
|
||||
return ss.str();
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T> struct RefType {};
|
||||
|
||||
template <> struct RefType<Float16> { using type = float; };
|
||||
|
||||
template <> struct RefType<float> { using type = double; };
|
||||
|
||||
template <> struct RefType<double> { using type = long double; };
|
||||
|
||||
template <typename T> using RefType_t = typename RefType<T>::type;
|
||||
|
||||
template <typename F> auto GetOccupancyMaxPotentialBlockSize(F kernel) {
|
||||
int grid_size = 0, block_size = 0;
|
||||
HIP_CHECK(hipOccupancyMaxPotentialBlockSize(&grid_size, &block_size, kernel, 0, 0));
|
||||
return std::make_tuple(grid_size, block_size);
|
||||
}
|
||||
|
||||
inline size_t GetMaxAllowedDeviceMemoryUsage() {
|
||||
hipDeviceProp_t props;
|
||||
HIP_CHECK(hipGetDeviceProperties(&props, 0));
|
||||
return props.totalGlobalMem * (cmd_options.accuracy_max_memory * 0.01f);
|
||||
}
|
||||
|
||||
inline uint64_t GetTestIterationCount() { return cmd_options.accuracy_iterations; }
|
||||
|
||||
template <typename T, typename... Ts> using kernel_sig = void (*)(T*, const size_t, Ts*...);
|
||||
|
||||
template <typename T, typename... Ts> using ref_sig = T (*)(Ts...);
|
||||
|
||||
template <int error_num> void NegativeTestRTCWrapper(const char* program_source) {
|
||||
hiprtcProgram program{};
|
||||
|
||||
HIPRTC_CHECK(
|
||||
hiprtcCreateProgram(&program, program_source, "math_test_rtc.cc", 0, nullptr, nullptr));
|
||||
#if HT_AMD
|
||||
std::string args = std::string("-ferror-limit=200");
|
||||
const char* options[] = {args.c_str()};
|
||||
hiprtcResult result{hiprtcCompileProgram(program, 1, options)};
|
||||
#else
|
||||
hiprtcResult result{hiprtcCompileProgram(program, 0, nullptr)};
|
||||
#endif
|
||||
|
||||
// Get the compile log and count compiler error messages
|
||||
size_t log_size{};
|
||||
HIPRTC_CHECK(hiprtcGetProgramLogSize(program, &log_size));
|
||||
std::string log(log_size, ' ');
|
||||
HIPRTC_CHECK(hiprtcGetProgramLog(program, log.data()));
|
||||
int error_count{0};
|
||||
|
||||
int expected_error_count{error_num};
|
||||
std::string error_message{"error:"};
|
||||
|
||||
size_t n_pos = log.find(error_message, 0);
|
||||
while (n_pos != std::string::npos) {
|
||||
++error_count;
|
||||
n_pos = log.find(error_message, n_pos + 1);
|
||||
}
|
||||
|
||||
HIPRTC_CHECK(hiprtcDestroyProgram(&program));
|
||||
HIPRTC_CHECK_ERROR(result, HIPRTC_ERROR_COMPILATION);
|
||||
REQUIRE(error_count == expected_error_count);
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); } \
|
||||
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL(log)
|
||||
NEGATIVE_KERNELS_SHELL(log2)
|
||||
NEGATIVE_KERNELS_SHELL(log10)
|
||||
NEGATIVE_KERNELS_SHELL(log1p)
|
||||
NEGATIVE_KERNELS_SHELL(logb)
|
||||
NEGATIVE_KERNELS_SHELL(ilogb)
|
||||
@@ -0,0 +1,96 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
/*
|
||||
Negative kernels used for the math log negative Test Cases that are using RTC.
|
||||
*/
|
||||
|
||||
static constexpr auto kLog{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void log_kernel_v1(double* x) { double result = log(x); }
|
||||
__global__ void log_kernel_v2(Dummy x) { double result = log(x); }
|
||||
__global__ void logf_kernel_v1(float* x) { float result = logf(x); }
|
||||
__global__ void logf_kernel_v2(Dummy x) { float result = logf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLog2{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void log2_kernel_v1(double* x) { double result = log2(x); }
|
||||
__global__ void log2_kernel_v2(Dummy x) { double result = log2(x); }
|
||||
__global__ void log2f_kernel_v1(float* x) { float result = log2f(x); }
|
||||
__global__ void log2f_kernel_v2(Dummy x) { float result = log2f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLog10{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void log10_kernel_v1(double* x) { double result = log10(x); }
|
||||
__global__ void log10_kernel_v2(Dummy x) { double result = log10(x); }
|
||||
__global__ void log10f_kernel_v1(float* x) { float result = log10f(x); }
|
||||
__global__ void log10f_kernel_v2(Dummy x) { float result = log10f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLog1p{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void log1p_kernel_v1(double* x) { double result = log1p(x); }
|
||||
__global__ void log1p_kernel_v2(Dummy x) { double result = log1p(x); }
|
||||
__global__ void log1pf_kernel_v1(float* x) { float result = log1pf(x); }
|
||||
__global__ void log1pf_kernel_v2(Dummy x) { float result = log1pf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLogb{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void logb_kernel_v1(double* x) { double result = logb(x); }
|
||||
__global__ void logb_kernel_v2(Dummy x) { double result = logb(x); }
|
||||
__global__ void logbf_kernel_v1(float* x) { float result = logbf(x); }
|
||||
__global__ void logbf_kernel_v2(Dummy x) { float result = logbf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kIlogb{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void ilogb_kernel_v1(double* x) { double result = ilogb(x); }
|
||||
__global__ void ilogb_kernel_v2(Dummy x) { double result = ilogb(x); }
|
||||
__global__ void ilogbf_kernel_v1(float* x) { float result = ilogbf(x); }
|
||||
__global__ void ilogbf_kernel_v2(Dummy x) { float result = ilogbf(x); }
|
||||
)"};
|
||||
@@ -0,0 +1,92 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_EXP(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); } \
|
||||
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_INT_2ND(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x, int e) { double result = func_name(x, e); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x, int e) { double result = func_name(x, e); } \
|
||||
__global__ void func_name##_kernel_v3(double x, int* e) { double result = func_name(x, e); } \
|
||||
__global__ void func_name##_kernel_v4(double x, Dummy e) { double result = func_name(x, e); } \
|
||||
__global__ void func_name##f_kernel_v1(float* x, int e) { float result = func_name##f(x, e); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x, int e) { float result = func_name##f(x, e); } \
|
||||
__global__ void func_name##f_kernel_v3(float x, int* e) { float result = func_name##f(x, e); } \
|
||||
__global__ void func_name##f_kernel_v4(float x, Dummy e) { float result = func_name##f(x, e); }
|
||||
|
||||
|
||||
NEGATIVE_KERNELS_SHELL_EXP(exp)
|
||||
NEGATIVE_KERNELS_SHELL_EXP(exp2)
|
||||
NEGATIVE_KERNELS_SHELL_EXP(exp10)
|
||||
NEGATIVE_KERNELS_SHELL_EXP(expm1)
|
||||
|
||||
__global__ void frexp_kernel_v1(double* x, int* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v2(Dummy x, int* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v3(double x, char* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v4(double x, short* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v5(double x, long* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v6(double x, long long* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v7(double x, float* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v8(double x, double* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v9(double x, Dummy* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v10(double x, const int* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexpf_kernel_v1(float* x, int* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v2(Dummy x, int* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v3(float x, char* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v4(float x, short* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v5(float x, long* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v6(float x, long long* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v7(float x, float* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v8(float x, double* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v9(float x, Dummy* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v10(float x, const int* nptr) { float result = frexpf(x, nptr); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL_INT_2ND(ldexp)
|
||||
|
||||
__global__ void pow_kernel_v1(double* x, double e) { double result = pow(x, e); }
|
||||
__global__ void pow_kernel_v2(Dummy x, double e) { double result = pow(x, e); }
|
||||
__global__ void pow_kernel_v3(double x, double* e) { double result = pow(x, e); }
|
||||
__global__ void pow_kernel_v4(double x, Dummy e) { double result = pow(x, e); }
|
||||
__global__ void powf_kernel_v1(float* x, float e) { float result = powf(x, e); }
|
||||
__global__ void powf_kernel_v2(Dummy x, float e) { float result = powf(x, e); }
|
||||
__global__ void powf_kernel_v3(float x, float* e) { float result = powf(x, e); }
|
||||
__global__ void powf_kernel_v4(float x, Dummy e) { float result = powf(x, e); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL_INT_2ND(powi)
|
||||
NEGATIVE_KERNELS_SHELL_INT_2ND(scalbn)
|
||||
|
||||
__global__ void scalbln_kernel_v1(double* x, long int n) { double result = scalbln(x, n); }
|
||||
__global__ void scalbln_kernel_v2(Dummy x, long int n) { double result = scalbln(x, n); }
|
||||
__global__ void scalbln_kernel_v3(double x, long int* n) { double result = scalbln(x, n); }
|
||||
__global__ void scalbln_kernel_v4(double x, Dummy n) { double result = scalbln(x, n); }
|
||||
__global__ void scalblnf_kernel_v1(float* x, long int n) { float result = scalblnf(x, n); }
|
||||
__global__ void scalblnf_kernel_v2(Dummy x, long int n) { float result = scalblnf(x, n); }
|
||||
__global__ void scalblnf_kernel_v3(float x, long int* n) { float result = scalblnf(x, n); }
|
||||
__global__ void scalblnf_kernel_v4(float x, Dummy n) { float result = scalblnf(x, n); }
|
||||
@@ -0,0 +1,150 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
/*
|
||||
Negative kernels used for the math pow negative Test Cases that are using RTC.
|
||||
*/
|
||||
|
||||
static constexpr auto kExp{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void exp_kernel_v1(double* x) { double result = exp(x); }
|
||||
__global__ void exp_kernel_v2(Dummy x) { double result = exp(x); }
|
||||
__global__ void expf_kernel_v1(float* x) { float result = expf(x); }
|
||||
__global__ void expf_kernel_v2(Dummy x) { float result = expf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kExp2{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void exp2_kernel_v1(double* x) { double result = exp2(x); }
|
||||
__global__ void exp2_kernel_v2(Dummy x) { double result = exp2(x); }
|
||||
__global__ void exp2f_kernel_v1(float* x) { float result = exp2f(x); }
|
||||
__global__ void exp2f_kernel_v2(Dummy x) { float result = exp2f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kExp10{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void exp10_kernel_v1(double* x) { double result = exp10(x); }
|
||||
__global__ void exp10_kernel_v2(Dummy x) { double result = exp10(x); }
|
||||
__global__ void exp10f_kernel_v1(float* x) { float result = exp10f(x); }
|
||||
__global__ void exp10f_kernel_v2(Dummy x) { float result = exp10f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kExpm1{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void expm1_kernel_v1(double* x) { double result = expm1(x); }
|
||||
__global__ void expm1_kernel_v2(Dummy x) { double result = expm1(x); }
|
||||
__global__ void expm1f_kernel_v1(float* x) { float result = expm1f(x); }
|
||||
__global__ void expm1f_kernel_v2(Dummy x) { float result = expm1f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFrexp{R"(
|
||||
__global__ void frexp_kernel_v1(double* x, int* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v2(Dummy x, int* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v3(double x, char* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v4(double x, short* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v5(double x, long* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v6(double x, long long* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v7(double x, float* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v8(double x, double* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v9(double x, Dummy* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexp_kernel_v10(double x, const int* nptr) { double result = frexp(x, nptr); }
|
||||
__global__ void frexpf_kernel_v1(float* x, int* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v2(Dummy x, int* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v3(float x, char* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v4(float x, short* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v5(float x, long* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v6(float x, long long* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v7(float x, float* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v8(float x, double* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v9(float x, Dummy* nptr) { float result = frexpf(x, nptr); }
|
||||
__global__ void frexpf_kernel_v10(float x, const int* nptr) { float result = frexpf(x, nptr); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLdexp{R"(
|
||||
__global__ void ldexp_kernel_v1(double* x, int e) { double result = ldexp(x, e); }
|
||||
__global__ void ldexp_kernel_v2(Dummy x, int e) { double result = ldexp(x, e); }
|
||||
__global__ void ldexp_kernel_v3(double x, int* e) { double result = ldexp(x, e); }
|
||||
__global__ void ldexp_kernel_v4(double x, Dummy e) { double result = ldexp(x, e); }
|
||||
__global__ void ldexpf_kernel_v1(float* x, int e) { float result = ldexpf(x, e); }
|
||||
__global__ void ldexpf_kernel_v2(Dummy x, int e) { float result = ldexpf(x, e); }
|
||||
__global__ void ldexpf_kernel_v3(float x, int* e) { float result = ldexpf(x, e); }
|
||||
__global__ void ldexpf_kernel_v4(float x, Dummy e) { float result = ldexpf(x, e); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kPow{R"(
|
||||
__global__ void pow_kernel_v1(double* x, double e) { double result = pow(x, e); }
|
||||
__global__ void pow_kernel_v2(Dummy x, double e) { double result = pow(x, e); }
|
||||
__global__ void pow_kernel_v3(double x, double* e) { double result = pow(x, e); }
|
||||
__global__ void pow_kernel_v4(double x, Dummy e) { double result = pow(x, e); }
|
||||
__global__ void powf_kernel_v1(float* x, float e) { float result = powf(x, e); }
|
||||
__global__ void powf_kernel_v2(Dummy x, float e) { float result = powf(x, e); }
|
||||
__global__ void powf_kernel_v3(float x, float* e) { float result = powf(x, e); }
|
||||
__global__ void powf_kernel_v4(float x, Dummy e) { float result = powf(x, e); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kPowi{R"(
|
||||
__global__ void powi_kernel_v1(double* x, int e) { double result = powi(x, e); }
|
||||
__global__ void powi_kernel_v2(Dummy x, int e) { double result = powi(x, e); }
|
||||
__global__ void powi_kernel_v3(double x, int* e) { double result = powi(x, e); }
|
||||
__global__ void powi_kernel_v4(double x, Dummy e) { double result = powi(x, e); }
|
||||
__global__ void powif_kernel_v1(float* x, int e) { float result = powif(x, e); }
|
||||
__global__ void powif_kernel_v2(Dummy x, int e) { float result = powif(x, e); }
|
||||
__global__ void powif_kernel_v3(float x, int* e) { float result = powif(x, e); }
|
||||
__global__ void powif_kernel_v4(float x, Dummy e) { float result = powif(x, e); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kScalbn{R"(
|
||||
__global__ void scalbn_kernel_v1(double* x, int e) { double result = scalbn(x, e); }
|
||||
__global__ void scalbn_kernel_v2(Dummy x, int e) { double result = scalbn(x, e); }
|
||||
__global__ void scalbn_kernel_v3(double x, int* e) { double result = scalbn(x, e); }
|
||||
__global__ void scalbn_kernel_v4(double x, Dummy e) { double result = scalbn(x, e); }
|
||||
__global__ void scalbnf_kernel_v1(float* x, int e) { float result = scalbnf(x, e); }
|
||||
__global__ void scalbnf_kernel_v2(Dummy x, int e) { float result = scalbnf(x, e); }
|
||||
__global__ void scalbnf_kernel_v3(float x, int* e) { float result = scalbnf(x, e); }
|
||||
__global__ void scalbnf_kernel_v4(float x, Dummy e) { float result = scalbnf(x, e); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kScalbln{R"(
|
||||
__global__ void scalbln_kernel_v1(double* x, long int n) { double result = scalbln(x, n); }
|
||||
__global__ void scalbln_kernel_v2(Dummy x, long int n) { double result = scalbln(x, n); }
|
||||
__global__ void scalbln_kernel_v3(double x, long int* n) { double result = scalbln(x, n); }
|
||||
__global__ void scalbln_kernel_v4(double x, Dummy n) { double result = scalbln(x, n); }
|
||||
__global__ void scalblnf_kernel_v1(float* x, long int n) { float result = scalblnf(x, n); }
|
||||
__global__ void scalblnf_kernel_v2(Dummy x, long int n) { float result = scalblnf(x, n); }
|
||||
__global__ void scalblnf_kernel_v3(float x, long int* n) { float result = scalblnf(x, n); }
|
||||
__global__ void scalblnf_kernel_v4(float x, Dummy n) { float result = scalblnf(x, n); }
|
||||
)"};
|
||||
@@ -0,0 +1,113 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x, double y) { auto result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(double x, double* y) { auto result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, double y) { auto result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(double x, Dummy y) { auto result = func_name(x, y); } \
|
||||
__global__ void func_name##f_kernel_v1(float* x, float y) { auto result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v2(float x, float* y) { auto result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v3(Dummy x, float y) { auto result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v4(float x, Dummy y) { auto result = func_name##f(x, y); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL(fmod)
|
||||
NEGATIVE_KERNELS_SHELL(remainder)
|
||||
|
||||
__global__ void remquo_kernel_v1(double* x, double y, int* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v2(Dummy x, double y, int* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v3(double x, double* y, int* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v4(double x, Dummy y, int* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v5(double x, double y, char* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v6(double x, double y, short* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v7(double x, double y, long* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v8(double x, double y, long long* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v9(double x, double y, float* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v10(double x, double y, double* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v11(double x, double y, Dummy* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v12(double x, double y, const int* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
|
||||
__global__ void remquof_kernel_v1(float* x, float y, int* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v2(Dummy x, float y, int* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v3(float x, float* y, int* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v4(float x, Dummy y, int* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v5(float x, float y, char* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v6(float x, float y, short* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v7(float x, float y, long* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v8(float x, float y, long long* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v9(float x, float y, float* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v10(float x, float y, double* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v11(float x, float y, Dummy* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v12(float x, float y, const int* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
|
||||
__global__ void modf_kernel_v1(double* x, double* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v2(Dummy x, double* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v3(double x, int* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v4(double x, char* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v5(double x, short* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v6(double x, long* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v7(double x, long long* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v8(double x, float* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v9(double x, Dummy* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v10(double x, const double* iptr) { auto result = modf(x, iptr); }
|
||||
|
||||
__global__ void modff_kernel_v1(float* x, float* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v2(Dummy x, float* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v3(float x, int* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v4(float x, char* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v5(float x, short* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v6(float x, long* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v7(float x, long long* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v8(float x, double* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v9(float x, Dummy* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v10(float x, const float* iptr) { auto result = modff(x, iptr); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL(fdim)
|
||||
@@ -0,0 +1,276 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
/*
|
||||
Negative kernels used for the math remainder and rounding negative Test Cases that are using RTC.
|
||||
*/
|
||||
|
||||
static constexpr auto kTrunc{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void trunc_kernel_v1(double* x) { auto result = trunc(x); }
|
||||
__global__ void trunc_kernel_v2(Dummy x) { auto result = trunc(x); }
|
||||
__global__ void truncf_kernel_v1(float* x) { auto result = truncf(x); }
|
||||
__global__ void truncf_kernel_v2(Dummy x) { auto result = truncf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kRound{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void round_kernel_v1(double* x) { auto result = round(x); }
|
||||
__global__ void round_kernel_v2(Dummy x) { auto result = round(x); }
|
||||
__global__ void roundf_kernel_v1(float* x) { auto result = roundf(x); }
|
||||
__global__ void roundf_kernel_v2(Dummy x) { auto result = roundf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kRint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void rint_kernel_v1(double* x) { auto result = rint(x); }
|
||||
__global__ void rint_kernel_v2(Dummy x) { auto result = rint(x); }
|
||||
__global__ void rintf_kernel_v1(float* x) { auto result = rintf(x); }
|
||||
__global__ void rintf_kernel_v2(Dummy x) { auto result = rintf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kNearbyint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void nearbyint_kernel_v1(double* x) { auto result = nearbyint(x); }
|
||||
__global__ void nearbyint_kernel_v2(Dummy x) { auto result = nearbyint(x); }
|
||||
__global__ void nearbyintf_kernel_v1(float* x) { auto result = nearbyintf(x); }
|
||||
__global__ void nearbyintf_kernel_v2(Dummy x) { auto result = nearbyintf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kCeil{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void ceil_kernel_v1(double* x) { auto result = ceil(x); }
|
||||
__global__ void ceil_kernel_v2(Dummy x) { auto result = ceil(x); }
|
||||
__global__ void ceilf_kernel_v1(float* x) { auto result = ceilf(x); }
|
||||
__global__ void ceilf_kernel_v2(Dummy x) { auto result = ceilf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFloor{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void floor_kernel_v1(double* x) { auto result = floor(x); }
|
||||
__global__ void floor_kernel_v2(Dummy x) { auto result = floor(x); }
|
||||
__global__ void floorf_kernel_v1(float* x) { auto result = floorf(x); }
|
||||
__global__ void floorf_kernel_v2(Dummy x) { auto result = floorf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLrint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void lrint_kernel_v1(double* x) { auto result = lrint(x); }
|
||||
__global__ void lrint_kernel_v2(Dummy x) { auto result = lrint(x); }
|
||||
__global__ void lrintf_kernel_v1(float* x) { auto result = lrintf(x); }
|
||||
__global__ void lrintf_kernel_v2(Dummy x) { auto result = lrintf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLround{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void lround_kernel_v1(double* x) { auto result = lround(x); }
|
||||
__global__ void lround_kernel_v2(Dummy x) { auto result = lround(x); }
|
||||
__global__ void lroundf_kernel_v1(float* x) { auto result = lroundf(x); }
|
||||
__global__ void lroundf_kernel_v2(Dummy x) { auto result = lroundf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLlrint{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void llrint_kernel_v1(double* x) { auto result = llrint(x); }
|
||||
__global__ void llrint_kernel_v2(Dummy x) { auto result = llrint(x); }
|
||||
__global__ void llrintf_kernel_v1(float* x) { auto result = llrintf(x); }
|
||||
__global__ void llrintf_kernel_v2(Dummy x) { auto result = llrintf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLlround{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void llround_kernel_v1(double* x) { auto result = llround(x); }
|
||||
__global__ void llround_kernel_v2(Dummy x) { auto result = llround(x); }
|
||||
__global__ void llroundf_kernel_v1(float* x) { auto result = llroundf(x); }
|
||||
__global__ void llroundf_kernel_v2(Dummy x) { auto result = llroundf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFmod{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void fmod_kernel_v1(double* x, double y) { auto result = fmod(x, y); }
|
||||
__global__ void fmod_kernel_v2(double x, double* y) { auto result = fmod(x, y); }
|
||||
__global__ void fmod_kernel_v3(Dummy x, double y) { auto result = fmod(x, y); }
|
||||
__global__ void fmod_kernel_v4(double x, Dummy y) { auto result = fmod(x, y); }
|
||||
__global__ void fmodf_kernel_v1(float* x, float y) { auto result = fmodf(x, y); }
|
||||
__global__ void fmodf_kernel_v2(float x, float* y) { auto result = fmodf(x, y); }
|
||||
__global__ void fmodf_kernel_v3(Dummy x, float y) { auto result = fmodf(x, y); }
|
||||
__global__ void fmodf_kernel_v4(float x, Dummy y) { auto result = fmodf(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kRemainder{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void remainder_kernel_v1(double* x, double y) { auto result = remainder(x, y); }
|
||||
__global__ void remainder_kernel_v2(double x, double* y) { auto result = remainder(x, y); }
|
||||
__global__ void remainder_kernel_v3(Dummy x, double y) { auto result = remainder(x, y); }
|
||||
__global__ void remainder_kernel_v4(double x, Dummy y) { auto result = remainder(x, y); }
|
||||
__global__ void remainderf_kernel_v1(float* x, float y) { auto result = remainderf(x, y); }
|
||||
__global__ void remainderf_kernel_v2(float x, float* y) { auto result = remainderf(x, y); }
|
||||
__global__ void remainderf_kernel_v3(Dummy x, float y) { auto result = remainderf(x, y); }
|
||||
__global__ void remainderf_kernel_v4(float x, Dummy y) { auto result = remainderf(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kRemquo{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void remquo_kernel_v1(double* x, double y, int* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v2(Dummy x, double y, int* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v3(double x, double* y, int* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v4(double x, Dummy y, int* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v5(double x, double y, char* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v6(double x, double y, short* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v7(double x, double y, long* quo) { auto result = remquo(x, y, quo); }
|
||||
__global__ void remquo_kernel_v8(double x, double y, long long* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v9(double x, double y, float* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v10(double x, double y, double* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v11(double x, double y, Dummy* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquo_kernel_v12(double x, double y, const int* quo) {
|
||||
auto result = remquo(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v1(float* x, float y, int* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v2(Dummy x, float y, int* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v3(float x, float* y, int* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v4(float x, Dummy y, int* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v5(float x, float y, char* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v6(float x, float y, short* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v7(float x, float y, long* quo) { auto result = remquof(x, y, quo); }
|
||||
__global__ void remquof_kernel_v8(float x, float y, long long* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v9(float x, float y, float* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v10(float x, float y, double* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v11(float x, float y, Dummy* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
__global__ void remquof_kernel_v12(float x, float y, const int* quo) {
|
||||
auto result = remquof(x, y, quo);
|
||||
}
|
||||
)"};
|
||||
|
||||
static constexpr auto kModf{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void modf_kernel_v1(double* x, double* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v2(Dummy x, double* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v3(double x, int* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v4(double x, char* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v5(double x, short* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v6(double x, long* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v7(double x, long long* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v8(double x, float* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v9(double x, Dummy* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modf_kernel_v10(double x, const double* iptr) { auto result = modf(x, iptr); }
|
||||
__global__ void modff_kernel_v1(float* x, float* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v2(Dummy x, float* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v3(float x, int* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v4(float x, char* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v5(float x, short* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v6(float x, long* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v7(float x, long long* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v8(float x, double* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v9(float x, Dummy* iptr) { auto result = modff(x, iptr); }
|
||||
__global__ void modff_kernel_v10(float x, const float* iptr) { auto result = modff(x, iptr); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFdim{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void fdim_kernel_v1(double* x, double y) { auto result = fdim(x, y); }
|
||||
__global__ void fdim_kernel_v2(double x, double* y) { auto result = fdim(x, y); }
|
||||
__global__ void fdim_kernel_v3(Dummy x, double y) { auto result = fdim(x, y); }
|
||||
__global__ void fdim_kernel_v4(double x, Dummy y) { auto result = fdim(x, y); }
|
||||
__global__ void fdimf_kernel_v1(float* x, float y) { auto result = fdimf(x, y); }
|
||||
__global__ void fdimf_kernel_v2(float x, float* y) { auto result = fdimf(x, y); }
|
||||
__global__ void fdimf_kernel_v3(Dummy x, float y) { auto result = fdimf(x, y); }
|
||||
__global__ void fdimf_kernel_v4(float x, Dummy y) { auto result = fdimf(x, y); }
|
||||
)"};
|
||||
@@ -0,0 +1,107 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_ONE_ARG(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); } \
|
||||
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_TWO_ARGS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x, double y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(double x, double* y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, double y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(double x, Dummy y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##f_kernel_v1(float* x, float y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v2(float x, float* y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v3(Dummy x, float y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v4(float x, Dummy y) { float result = func_name##f(x, y); }
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_ARRAY_ARG(func_name) \
|
||||
__global__ void func_name##_kernel_v1(int* dim, const double* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(Dummy dim, const double* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(int dim, const int* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v4(int dim, const char* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v5(int dim, const short* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v6(int dim, const long* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v7(int dim, const long long* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v8(int dim, const float* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v9(int dim, const Dummy* a) { \
|
||||
double result = func_name(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v1(int* dim, const float* a) { \
|
||||
float result = func_name##f(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v2(Dummy dim, const float* a) { \
|
||||
float result = func_name##f(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v3(int dim, const int* a) { \
|
||||
float result = func_name##f(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v4(int dim, const char* a) { \
|
||||
float result = func_name##f(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v5(int dim, const short* a) { \
|
||||
float result = func_name##f(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v6(int dim, const long* a) { \
|
||||
float result = func_name##f(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v7(int dim, const long long* a) { \
|
||||
float result = func_name##f(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v8(int dim, const double* a) { \
|
||||
float result = func_name##f(dim, a); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v9(int dim, const Dummy* a) { \
|
||||
double result = func_name##f(dim, a); \
|
||||
}
|
||||
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(sqrt)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(rsqrt)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(cbrt)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(rcbrt)
|
||||
NEGATIVE_KERNELS_SHELL_TWO_ARGS(hypot)
|
||||
NEGATIVE_KERNELS_SHELL_TWO_ARGS(rhypot)
|
||||
NEGATIVE_KERNELS_SHELL_ARRAY_ARG(norm)
|
||||
NEGATIVE_KERNELS_SHELL_ARRAY_ARG(rnorm)
|
||||
@@ -0,0 +1,119 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_THREE_ARGS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x, double y, double z) { \
|
||||
double result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(double x, double* y, double z) { \
|
||||
double result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(double x, double y, double* z) { \
|
||||
double result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v4(Dummy x, double y, double z) { \
|
||||
double result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v5(double x, Dummy y, double z) { \
|
||||
double result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v6(double x, double y, Dummy z) { \
|
||||
double result = func_name(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v1(float* x, float y, float z) { \
|
||||
float result = func_name##f(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v2(float x, float* y, float z) { \
|
||||
float result = func_name##f(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v3(float x, float y, float* z) { \
|
||||
float result = func_name##f(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v4(Dummy x, float y, float z) { \
|
||||
float result = func_name##f(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v5(float x, Dummy y, float z) { \
|
||||
float result = func_name##f(x, y, z); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v6(float x, float y, Dummy z) { \
|
||||
float result = func_name##f(x, y, z); \
|
||||
}
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_FOUR_ARGS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x, double y, double z, double w) { \
|
||||
double result = func_name(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v2(double x, double* y, double z, double w) { \
|
||||
double result = func_name(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v3(double x, double y, double* z, double w) { \
|
||||
double result = func_name(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v4(double x, double y, double z, double* w) { \
|
||||
double result = func_name(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v5(Dummy x, double y, double z, double w) { \
|
||||
double result = func_name(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v6(double x, Dummy y, double z, double w) { \
|
||||
double result = func_name(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v7(double x, double y, Dummy z, double w) { \
|
||||
double result = func_name(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##_kernel_v8(double x, double y, double z, Dummy w) { \
|
||||
double result = func_name(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v1(float* x, float y, float z, float w) { \
|
||||
float result = func_name##f(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v2(float x, float* y, float z, float w) { \
|
||||
float result = func_name##f(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v3(float x, float y, float* z, float w) { \
|
||||
float result = func_name##f(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v4(float x, float y, float z, float* w) { \
|
||||
float result = func_name##f(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v5(Dummy x, float y, float z, float w) { \
|
||||
float result = func_name##f(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v6(float x, Dummy y, float z, float w) { \
|
||||
float result = func_name##f(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v7(float x, float y, Dummy z, float w) { \
|
||||
float result = func_name##f(x, y, z, w); \
|
||||
} \
|
||||
__global__ void func_name##f_kernel_v8(float x, float y, float z, Dummy w) { \
|
||||
float result = func_name##f(x, y, z, w); \
|
||||
}
|
||||
|
||||
NEGATIVE_KERNELS_SHELL_THREE_ARGS(norm3d)
|
||||
NEGATIVE_KERNELS_SHELL_THREE_ARGS(rnorm3d)
|
||||
NEGATIVE_KERNELS_SHELL_FOUR_ARGS(norm4d)
|
||||
NEGATIVE_KERNELS_SHELL_FOUR_ARGS(rnorm4d)
|
||||
@@ -0,0 +1,428 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
/*
|
||||
Negative kernels used for the math root negative Test Cases that are using RTC.
|
||||
*/
|
||||
|
||||
static constexpr auto kSqrt{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void sqrt_kernel_v1(double* x) { double result = sqrt(x); }
|
||||
__global__ void sqrt_kernel_v2(Dummy x) { double result = sqrt(x); }
|
||||
__global__ void sqrtf_kernel_v1(float* x) { float result = sqrtf(x); }
|
||||
__global__ void sqrtf_kernel_v2(Dummy x) { float result = sqrtf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kRsqrt{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void rsqrt_kernel_v1(double* x) { double result = rsqrt(x); }
|
||||
__global__ void rsqrt_kernel_v2(Dummy x) { double result = rsqrt(x); }
|
||||
__global__ void rsqrtf_kernel_v1(float* x) { float result = rsqrtf(x); }
|
||||
__global__ void rsqrtf_kernel_v2(Dummy x) { float result = rsqrtf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kCbrt{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void cbrt_kernel_v1(double* x) { double result = cbrt(x); }
|
||||
__global__ void cbrt_kernel_v2(Dummy x) { double result = cbrt(x); }
|
||||
__global__ void cbrtf_kernel_v1(float* x) { float result = cbrtf(x); }
|
||||
__global__ void cbrtf_kernel_v2(Dummy x) { float result = cbrtf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kRcbrt{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void rcbrt_kernel_v1(double* x) { double result = rcbrt(x); }
|
||||
__global__ void rcbrt_kernel_v2(Dummy x) { double result = rcbrt(x); }
|
||||
__global__ void rcbrtf_kernel_v1(float* x) { float result = rcbrtf(x); }
|
||||
__global__ void rcbrtf_kernel_v2(Dummy x) { float result = rcbrtf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kHypot{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void hypot_kernel_v1(double* x, double y) { double result = hypot(x, y); }
|
||||
__global__ void hypot_kernel_v2(double x, double* y) { double result = hypot(x, y); }
|
||||
__global__ void hypot_kernel_v3(Dummy x, double y) { double result = hypot(x, y); }
|
||||
__global__ void hypot_kernel_v4(double x, Dummy y) { double result = hypot(x, y); }
|
||||
__global__ void hypotf_kernel_v1(float* x, float y) { float result = hypotf(x, y); }
|
||||
__global__ void hypotf_kernel_v2(float x, float* y) { float result = hypotf(x, y); }
|
||||
__global__ void hypotf_kernel_v3(Dummy x, float y) { float result = hypotf(x, y); }
|
||||
__global__ void hypotf_kernel_v4(float x, Dummy y) { float result = hypotf(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kRhypot{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void rhypot_kernel_v1(double* x, double y) { double result = rhypot(x, y); }
|
||||
__global__ void rhypot_kernel_v2(double x, double* y) { double result = rhypot(x, y); }
|
||||
__global__ void rhypot_kernel_v3(Dummy x, double y) { double result = rhypot(x, y); }
|
||||
__global__ void rhypot_kernel_v4(double x, Dummy y) { double result = rhypot(x, y); }
|
||||
__global__ void rhypotf_kernel_v1(float* x, float y) { float result = rhypotf(x, y); }
|
||||
__global__ void rhypotf_kernel_v2(float x, float* y) { float result = rhypotf(x, y); }
|
||||
__global__ void rhypotf_kernel_v3(Dummy x, float y) { float result = rhypotf(x, y); }
|
||||
__global__ void rhypotf_kernel_v4(float x, Dummy y) { float result = rhypotf(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kNorm3D{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void norm3d_kernel_v1(double* x, double y, double z) {
|
||||
double result = norm3d(x, y, z);
|
||||
}
|
||||
__global__ void norm3d_kernel_v2(double x, double* y, double z) {
|
||||
double result = norm3d(x, y, z);
|
||||
}
|
||||
__global__ void norm3d_kernel_v3(double x, double y, double* z) {
|
||||
double result = norm3d(x, y, z);
|
||||
}
|
||||
__global__ void norm3d_kernel_v4(Dummy x, double y, double z) {
|
||||
double result = norm3d(x, y, z);
|
||||
}
|
||||
__global__ void norm3d_kernel_v5(double x, Dummy y, double z) {
|
||||
double result = norm3d(x, y, z);
|
||||
}
|
||||
__global__ void norm3d_kernel_v6(double x, double y, Dummy z) {
|
||||
double result = norm3d(x, y, z);
|
||||
}
|
||||
__global__ void norm3df_kernel_v1(float* x, float y, float z) {
|
||||
float result = norm3df(x, y, z);
|
||||
}
|
||||
__global__ void norm3df_kernel_v2(float x, float* y, float z) {
|
||||
float result = norm3df(x, y, z);
|
||||
}
|
||||
__global__ void norm3df_kernel_v3(float x, float y, float* z) {
|
||||
float result = norm3df(x, y, z);
|
||||
}
|
||||
__global__ void norm3df_kernel_v4(Dummy x, float y, float z) {
|
||||
float result = norm3df(x, y, z);
|
||||
}
|
||||
__global__ void norm3df_kernel_v5(float x, Dummy y, float z) {
|
||||
float result = norm3df(x, y, z);
|
||||
}
|
||||
__global__ void norm3df_kernel_v6(float x, float y, Dummy z) {
|
||||
float result = norm3df(x, y, z);
|
||||
}
|
||||
)"};
|
||||
|
||||
static constexpr auto kRnorm3D{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void rnorm3d_kernel_v1(double* x, double y, double z) {
|
||||
double result = rnorm3d(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3d_kernel_v2(double x, double* y, double z) {
|
||||
double result = rnorm3d(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3d_kernel_v3(double x, double y, double* z) {
|
||||
double result = rnorm3d(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3d_kernel_v4(Dummy x, double y, double z) {
|
||||
double result = rnorm3d(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3d_kernel_v5(double x, Dummy y, double z) {
|
||||
double result = rnorm3d(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3d_kernel_v6(double x, double y, Dummy z) {
|
||||
double result = rnorm3d(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3df_kernel_v1(float* x, float y, float z) {
|
||||
float result = rnorm3df(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3df_kernel_v2(float x, float* y, float z) {
|
||||
float result = rnorm3df(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3df_kernel_v3(float x, float y, float* z) {
|
||||
float result = rnorm3df(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3df_kernel_v4(Dummy x, float y, float z) {
|
||||
float result = rnorm3df(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3df_kernel_v5(float x, Dummy y, float z) {
|
||||
float result = rnorm3df(x, y, z);
|
||||
}
|
||||
__global__ void rnorm3df_kernel_v6(float x, float y, Dummy z) {
|
||||
float result = rnorm3df(x, y, z);
|
||||
}
|
||||
)"};
|
||||
|
||||
static constexpr auto kNorm4D{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void norm4d_kernel_v1(double* x, double y, double z, double w) {
|
||||
double result = norm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4d_kernel_v2(double x, double* y, double z, double w) {
|
||||
double result = norm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4d_kernel_v3(double x, double y, double* z, double w) {
|
||||
double result = norm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4d_kernel_v4(double x, double y, double z, double* w) {
|
||||
double result = norm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4d_kernel_v5(Dummy x, double y, double z, double w) {
|
||||
double result = norm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4d_kernel_v6(double x, Dummy y, double z, double w) {
|
||||
double result = norm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4d_kernel_v7(double x, double y, Dummy z, double w) {
|
||||
double result = norm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4d_kernel_v8(double x, double y, double z, Dummy w) {
|
||||
double result = norm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4df_kernel_v1(float* x, float y, float z, float w) {
|
||||
float result = norm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4df_kernel_v2(float x, float* y, float z, float w) {
|
||||
float result = norm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4df_kernel_v3(float x, float y, float* z, float w) {
|
||||
float result = norm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4df_kernel_v4(float x, float y, float z, float* w) {
|
||||
float result = norm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4df_kernel_v5(Dummy x, float y, float z, float w) {
|
||||
float result = norm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4df_kernel_v6(float x, Dummy y, float z, float w) {
|
||||
float result = norm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4df_kernel_v7(float x, float y, Dummy z, float w) {
|
||||
float result = norm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void norm4df_kernel_v8(float x, float y, float z, Dummy w) {
|
||||
float result = norm4df(x, y, z, w);
|
||||
}
|
||||
)"};
|
||||
|
||||
static constexpr auto kRnorm4D{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void rnorm4d_kernel_v1(double* x, double y, double z, double w) {
|
||||
double result = rnorm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4d_kernel_v2(double x, double* y, double z, double w) {
|
||||
double result = rnorm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4d_kernel_v3(double x, double y, double* z, double w) {
|
||||
double result = rnorm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4d_kernel_v4(double x, double y, double z, double* w) {
|
||||
double result = rnorm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4d_kernel_v5(Dummy x, double y, double z, double w) {
|
||||
double result = rnorm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4d_kernel_v6(double x, Dummy y, double z, double w) {
|
||||
double result = rnorm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4d_kernel_v7(double x, double y, Dummy z, double w) {
|
||||
double result = rnorm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4d_kernel_v8(double x, double y, double z, Dummy w) {
|
||||
double result = rnorm4d(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4df_kernel_v1(float* x, float y, float z, float w) {
|
||||
float result = rnorm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4df_kernel_v2(float x, float* y, float z, float w) {
|
||||
float result = rnorm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4df_kernel_v3(float x, float y, float* z, float w) {
|
||||
float result = rnorm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4df_kernel_v4(float x, float y, float z, float* w) {
|
||||
float result = rnorm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4df_kernel_v5(Dummy x, float y, float z, float w) {
|
||||
float result = rnorm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4df_kernel_v6(float x, Dummy y, float z, float w) {
|
||||
float result = rnorm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4df_kernel_v7(float x, float y, Dummy z, float w) {
|
||||
float result = rnorm4df(x, y, z, w);
|
||||
}
|
||||
__global__ void rnorm4df_kernel_v8(float x, float y, float z, Dummy w) {
|
||||
float result = rnorm4df(x, y, z, w);
|
||||
}
|
||||
)"};
|
||||
|
||||
static constexpr auto kNorm{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void norm_kernel_v1(int* dim, const double* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void norm_kernel_v2(Dummy dim, const double* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void norm_kernel_v3(int dim, const int* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void norm_kernel_v4(int dim, const char* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void norm_kernel_v5(int dim, const short* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void norm_kernel_v6(int dim, const long* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void norm_kernel_v7(int dim, const long long* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void norm_kernel_v8(int dim, const float* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void norm_kernel_v9(int dim, const Dummy* a) {
|
||||
double result = norm(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v1(int* dim, const float* a) {
|
||||
float result = normf(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v2(Dummy dim, const float* a) {
|
||||
float result = normf(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v3(int dim, const int* a) {
|
||||
float result = normf(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v4(int dim, const char* a) {
|
||||
float result = normf(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v5(int dim, const short* a) {
|
||||
float result = normf(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v6(int dim, const long* a) {
|
||||
float result = normf(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v7(int dim, const long long* a) {
|
||||
float result = normf(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v8(int dim, const double* a) {
|
||||
float result = normf(dim, a);
|
||||
}
|
||||
__global__ void normf_kernel_v9(int dim, const Dummy* a) {
|
||||
double result = normf(dim, a);
|
||||
}
|
||||
)"};
|
||||
|
||||
static constexpr auto kRnorm{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void rnorm_kernel_v1(int* dim, const double* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnorm_kernel_v2(Dummy dim, const double* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnorm_kernel_v3(int dim, const int* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnorm_kernel_v4(int dim, const char* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnorm_kernel_v5(int dim, const short* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnorm_kernel_v6(int dim, const long* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnorm_kernel_v7(int dim, const long long* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnorm_kernel_v8(int dim, const float* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnorm_kernel_v9(int dim, const Dummy* a) {
|
||||
double result = rnorm(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v1(int* dim, const float* a) {
|
||||
float result = rnormf(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v2(Dummy dim, const float* a) {
|
||||
float result = rnormf(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v3(int dim, const int* a) {
|
||||
float result = rnormf(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v4(int dim, const char* a) {
|
||||
float result = rnormf(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v5(int dim, const short* a) {
|
||||
float result = rnormf(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v6(int dim, const long* a) {
|
||||
float result = rnormf(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v7(int dim, const long long* a) {
|
||||
float result = rnormf(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v8(int dim, const double* a) {
|
||||
float result = rnormf(dim, a);
|
||||
}
|
||||
__global__ void rnormf_kernel_v9(int dim, const Dummy* a) {
|
||||
double result = rnormf(dim, a);
|
||||
}
|
||||
)"};
|
||||
@@ -0,0 +1,43 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x) { auto result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { auto result = func_name(x); } \
|
||||
__global__ void func_name##f_kernel_v1(float* x) { auto result = func_name##f(x); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x) { auto result = func_name##f(x); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL(trunc)
|
||||
NEGATIVE_KERNELS_SHELL(round)
|
||||
NEGATIVE_KERNELS_SHELL(rint)
|
||||
NEGATIVE_KERNELS_SHELL(nearbyint)
|
||||
NEGATIVE_KERNELS_SHELL(ceil)
|
||||
NEGATIVE_KERNELS_SHELL(floor)
|
||||
NEGATIVE_KERNELS_SHELL(lrint)
|
||||
NEGATIVE_KERNELS_SHELL(lround)
|
||||
NEGATIVE_KERNELS_SHELL(llrint)
|
||||
NEGATIVE_KERNELS_SHELL(llround)
|
||||
@@ -0,0 +1,60 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_ONE_ARG(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); } \
|
||||
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
|
||||
|
||||
#define NEGATIVE_KERNELS_SHELL_TWO_ARGS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(int* x, double y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(int x, double* y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, double y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(int x, Dummy y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##f_kernel_v1(int* x, float y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v2(int x, float* y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v3(Dummy x, float y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v4(int x, Dummy y) { float result = func_name##f(x, y); }
|
||||
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(erf)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(erfc)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(erfinv)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(erfcinv)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(erfcx)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(normcdf)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(normcdfinv)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(lgamma)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(tgamma)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(j0)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(j1)
|
||||
NEGATIVE_KERNELS_SHELL_TWO_ARGS(jn)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(y0)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(y1)
|
||||
NEGATIVE_KERNELS_SHELL_TWO_ARGS(yn)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(cyl_bessel_i0)
|
||||
NEGATIVE_KERNELS_SHELL_ONE_ARG(cyl_bessel_i1)
|
||||
@@ -0,0 +1,236 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
/*
|
||||
Negative kernels used for the math special function negative Test Cases that are using RTC.
|
||||
*/
|
||||
|
||||
static constexpr auto kErf{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void erf_kernel_v1(double* x) { double result = erf(x); }
|
||||
__global__ void erf_kernel_v2(Dummy x) { double result = erf(x); }
|
||||
__global__ void erff_kernel_v1(float* x) { float result = erff(x); }
|
||||
__global__ void erff_kernel_v2(Dummy x) { float result = erff(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kErfc{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void erfc_kernel_v1(double* x) { double result = erfc(x); }
|
||||
__global__ void erfc_kernel_v2(Dummy x) { double result = erfc(x); }
|
||||
__global__ void erfcf_kernel_v1(float* x) { float result = erfcf(x); }
|
||||
__global__ void erfcf_kernel_v2(Dummy x) { float result = erfcf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kErfinv{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void erfinv_kernel_v1(double* x) { double result = erfinv(x); }
|
||||
__global__ void erfinv_kernel_v2(Dummy x) { double result = erfinv(x); }
|
||||
__global__ void erfinvf_kernel_v1(float* x) { float result = erfinvf(x); }
|
||||
__global__ void erfinvf_kernel_v2(Dummy x) { float result = erfinvf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kErfcinv{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void erfcinv_kernel_v1(double* x) { double result = erfcinv(x); }
|
||||
__global__ void erfcinv_kernel_v2(Dummy x) { double result = erfcinv(x); }
|
||||
__global__ void erfcinvf_kernel_v1(float* x) { float result = erfcinvf(x); }
|
||||
__global__ void erfcinvf_kernel_v2(Dummy x) { float result = erfcinvf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kErfcx{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void erfcx_kernel_v1(double* x) { double result = erfcx(x); }
|
||||
__global__ void erfcx_kernel_v2(Dummy x) { double result = erfcx(x); }
|
||||
__global__ void erfcxf_kernel_v1(float* x) { float result = erfcxf(x); }
|
||||
__global__ void erfcxf_kernel_v2(Dummy x) { float result = erfcxf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kNormcdf{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void normcdf_kernel_v1(double* x) { double result = normcdf(x); }
|
||||
__global__ void normcdf_kernel_v2(Dummy x) { double result = normcdf(x); }
|
||||
__global__ void normcdff_kernel_v1(float* x) { float result = normcdff(x); }
|
||||
__global__ void normcdff_kernel_v2(Dummy x) { float result = normcdff(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kNormcdfinv{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void normcdfinv_kernel_v1(double* x) { double result = normcdfinv(x); }
|
||||
__global__ void normcdfinv_kernel_v2(Dummy x) { double result = normcdfinv(x); }
|
||||
__global__ void normcdfinvf_kernel_v1(float* x) { float result = normcdfinvf(x); }
|
||||
__global__ void normcdfinvf_kernel_v2(Dummy x) { float result = normcdfinvf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kLgamma{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void lgamma_kernel_v1(double* x) { double result = lgamma(x); }
|
||||
__global__ void lgamma_kernel_v2(Dummy x) { double result = lgamma(x); }
|
||||
__global__ void lgammaf_kernel_v1(float* x) { float result = lgammaf(x); }
|
||||
__global__ void lgammaf_kernel_v2(Dummy x) { float result = lgammaf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kTgamma{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void tgamma_kernel_v1(double* x) { double result = tgamma(x); }
|
||||
__global__ void tgamma_kernel_v2(Dummy x) { double result = tgamma(x); }
|
||||
__global__ void tgammaf_kernel_v1(float* x) { float result = tgammaf(x); }
|
||||
__global__ void tgammaf_kernel_v2(Dummy x) { float result = tgammaf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kJ0{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void j0_kernel_v1(double* x) { double result = j0(x); }
|
||||
__global__ void j0_kernel_v2(Dummy x) { double result = j0(x); }
|
||||
__global__ void j0f_kernel_v1(float* x) { float result = j0f(x); }
|
||||
__global__ void j0f_kernel_v2(Dummy x) { float result = j0f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kJ1{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void j1_kernel_v1(double* x) { double result = j1(x); }
|
||||
__global__ void j1_kernel_v2(Dummy x) { double result = j1(x); }
|
||||
__global__ void j1f_kernel_v1(float* x) { float result = j1f(x); }
|
||||
__global__ void j1f_kernel_v2(Dummy x) { float result = j1f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kJn{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void jn_kernel_v1(int* x, double y) { double result = jn(x, y); }
|
||||
__global__ void jn_kernel_v2(int x, double* y) { double result = jn(x, y); }
|
||||
__global__ void jn_kernel_v3(Dummy x, double y) { double result = jn(x, y); }
|
||||
__global__ void jn_kernel_v4(int x, Dummy y) { double result = jn(x, y); }
|
||||
__global__ void jnf_kernel_v1(int* x, float y) { float result = jnf(x, y); }
|
||||
__global__ void jnf_kernel_v2(int x, float* y) { float result = jnf(x, y); }
|
||||
__global__ void jnf_kernel_v3(Dummy x, float y) { float result = jnf(x, y); }
|
||||
__global__ void jnf_kernel_v4(int x, Dummy y) { float result = jnf(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kY0{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void y0_kernel_v1(double* x) { double result = y0(x); }
|
||||
__global__ void y0_kernel_v2(Dummy x) { double result = y0(x); }
|
||||
__global__ void y0f_kernel_v1(float* x) { float result = y0f(x); }
|
||||
__global__ void y0f_kernel_v2(Dummy x) { float result = y0f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kY1{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void y1_kernel_v1(double* x) { double result = y1(x); }
|
||||
__global__ void y1_kernel_v2(Dummy x) { double result = y1(x); }
|
||||
__global__ void y1f_kernel_v1(float* x) { float result = y1f(x); }
|
||||
__global__ void y1f_kernel_v2(Dummy x) { float result = y1f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kYn{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void yn_kernel_v1(int* x, double y) { double result = yn(x, y); }
|
||||
__global__ void yn_kernel_v2(int x, double* y) { double result = yn(x, y); }
|
||||
__global__ void yn_kernel_v3(Dummy x, double y) { double result = yn(x, y); }
|
||||
__global__ void yn_kernel_v4(int x, Dummy y) { double result = yn(x, y); }
|
||||
__global__ void ynf_kernel_v1(int* x, float y) { float result = ynf(x, y); }
|
||||
__global__ void ynf_kernel_v2(int x, float* y) { float result = ynf(x, y); }
|
||||
__global__ void ynf_kernel_v3(Dummy x, float y) { float result = ynf(x, y); }
|
||||
__global__ void ynf_kernel_v4(int x, Dummy y) { float result = ynf(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kCylBesselI0{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void cyl_bessel_i0_kernel_v1(double* x) { double result = cyl_bessel_i0(x); }
|
||||
__global__ void cyl_bessel_i0_kernel_v2(Dummy x) { double result = cyl_bessel_i0(x); }
|
||||
__global__ void cyl_bessel_i0f_kernel_v1(float* x) { float result = cyl_bessel_i0f(x); }
|
||||
__global__ void cyl_bessel_i0f_kernel_v2(Dummy x) { float result = cyl_bessel_i0f(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kCylBesselI1{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void cyl_bessel_i1_kernel_v1(double* x) { double result = cyl_bessel_i1(x); }
|
||||
__global__ void cyl_bessel_i1_kernel_v2(Dummy x) { double result = cyl_bessel_i1(x); }
|
||||
__global__ void cyl_bessel_i1f_kernel_v1(float* x) { float result = cyl_bessel_i1f(x); }
|
||||
__global__ void cyl_bessel_i1f_kernel_v2(Dummy x) { float result = cyl_bessel_i1f(x); }
|
||||
)"};
|
||||
@@ -0,0 +1,294 @@
|
||||
//
|
||||
// Copyright (c) 2017 The Khronos Group Inc.
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
//
|
||||
|
||||
// Disclaimer:
|
||||
// This code is based on the work found in OpenCL-CTS authored by The Khronos Group.
|
||||
// The original code can be found at https://github.com/KhronosGroup/OpenCL-CTS.
|
||||
// We acknowledge the contributions of The Khronos Group to the development of this code.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <limits>
|
||||
|
||||
/*-----------------------------------------------------------------------------
|
||||
HEX_FLT, HEXT_DBL, HEX_LDBL -- Create hex floating point literal of type
|
||||
float, double, long double respectively. Arguments:
|
||||
|
||||
sm -- sign of number,
|
||||
int -- integer part of mantissa (without `0x' prefix),
|
||||
fract -- fractional part of mantissa (without decimal point and `L' or
|
||||
`LL' suffixes),
|
||||
se -- sign of exponent,
|
||||
exp -- absolute value of (binary) exponent.
|
||||
|
||||
Example:
|
||||
|
||||
double yhi = HEX_DBL(+, 1, 5555555555555, -, 2); // 0x1.5555555555555p-2
|
||||
|
||||
Note:
|
||||
|
||||
We have to pass signs as separate arguments because gcc pass negative
|
||||
integer values (e. g. `-2') into a macro as two separate tokens, so
|
||||
`HEX_FLT(1, 0, -2)' produces result `0x1.0p- 2' (note a space between minus
|
||||
and two) which is not a correct floating point literal.
|
||||
-----------------------------------------------------------------------------*/
|
||||
#if defined(_MSC_VER) && !defined(__INTEL_COMPILER)
|
||||
// If compiler does not support hex floating point literals:
|
||||
#define HEX_FLT(sm, int, fract, se, exp) \
|
||||
sm ldexpf((float)(0x##int##fract##UL), \
|
||||
se exp + ilogbf((float)0x##int) - ilogbf((float)(0x##int##fract##UL)))
|
||||
#define HEX_DBL(sm, int, fract, se, exp) \
|
||||
sm ldexp((double)(0x##int##fract##ULL), \
|
||||
se exp + ilogb((double)0x##int) - ilogb((double)(0x##int##fract##ULL)))
|
||||
#define HEX_LDBL(sm, int, fract, se, exp) \
|
||||
sm ldexpl((long double)(0x##int##fract##ULL), \
|
||||
se exp + ilogbl((long double)0x##int) - ilogbl((long double)(0x##int##fract##ULL)))
|
||||
#else
|
||||
// If compiler supports hex floating point literals: just concatenate all the
|
||||
// parts into a literal.
|
||||
#define HEX_FLT(sm, int, fract, se, exp) sm 0x##int##.##fract##p##se##exp##F
|
||||
#define HEX_DBL(sm, int, fract, se, exp) sm 0x##int##.##fract##p##se##exp
|
||||
#define HEX_LDBL(sm, int, fract, se, exp) sm 0x##int##.##fract##p##se##exp##L
|
||||
#endif
|
||||
|
||||
inline constexpr std::array kSpecialValuesDouble{
|
||||
-std::numeric_limits<double>::quiet_NaN(),
|
||||
-std::numeric_limits<double>::infinity(),
|
||||
-std::numeric_limits<double>::max(),
|
||||
HEX_DBL(-, 1, 0000000000001, +, 64),
|
||||
HEX_DBL(-, 1, 0, +, 64),
|
||||
HEX_DBL(-, 1, fffffffffffff, +, 63),
|
||||
HEX_DBL(-, 1, 0000000000001, +, 63),
|
||||
HEX_DBL(-, 1, 0, +, 63),
|
||||
HEX_DBL(-, 1, fffffffffffff, +, 62),
|
||||
HEX_DBL(-, 1, 000002, +, 32),
|
||||
HEX_DBL(-, 1, 0, +, 32),
|
||||
HEX_DBL(-, 1, fffffffffffff, +, 31),
|
||||
HEX_DBL(-, 1, 0000000000001, +, 31),
|
||||
HEX_DBL(-, 1, 0, +, 31),
|
||||
HEX_DBL(-, 1, fffffffffffff, +, 30),
|
||||
-1000.0,
|
||||
-100.0,
|
||||
-4.0,
|
||||
-3.5,
|
||||
-3.0,
|
||||
HEX_DBL(-, 1, 8000000000001, +, 1),
|
||||
-2.5,
|
||||
HEX_DBL(-, 1, 7ffffffffffff, +, 1),
|
||||
-2.0,
|
||||
HEX_DBL(-, 1, 8000000000001, +, 0),
|
||||
-1.5,
|
||||
HEX_DBL(-, 1, 7ffffffffffff, +, 0),
|
||||
HEX_DBL(-, 1, 0000000000001, +, 0),
|
||||
-1.0,
|
||||
HEX_DBL(-, 1, fffffffffffff, -, 1),
|
||||
HEX_DBL(-, 1, 0000000000001, -, 1),
|
||||
-0.5,
|
||||
HEX_DBL(-, 1, fffffffffffff, -, 2),
|
||||
HEX_DBL(-, 1, 0000000000001, -, 2),
|
||||
-0.25,
|
||||
HEX_DBL(-, 1, fffffffffffff, -, 3),
|
||||
HEX_DBL(-, 1, 0000000000001, -, 1022),
|
||||
-std::numeric_limits<double>::min(),
|
||||
HEX_DBL(-, 0, fffffffffffff, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000fff, -, 1022),
|
||||
HEX_DBL(-, 0, 00000000000fe, -, 1022),
|
||||
HEX_DBL(-, 0, 000000000000e, -, 1022),
|
||||
HEX_DBL(-, 0, 000000000000c, -, 1022),
|
||||
HEX_DBL(-, 0, 000000000000a, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000008, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000007, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000006, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000005, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000004, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000003, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000002, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000001, -, 1022),
|
||||
-0.0,
|
||||
|
||||
std::numeric_limits<double>::quiet_NaN(),
|
||||
std::numeric_limits<double>::infinity(),
|
||||
std::numeric_limits<double>::max(),
|
||||
HEX_DBL(+, 1, 0000000000001, +, 64),
|
||||
HEX_DBL(+, 1, 0, +, 64),
|
||||
HEX_DBL(+, 1, fffffffffffff, +, 63),
|
||||
HEX_DBL(+, 1, 0000000000001, +, 63),
|
||||
HEX_DBL(+, 1, 0, +, 63),
|
||||
HEX_DBL(+, 1, fffffffffffff, +, 62),
|
||||
HEX_DBL(+, 1, 000002, +, 32),
|
||||
HEX_DBL(+, 1, 0, +, 32),
|
||||
HEX_DBL(+, 1, fffffffffffff, +, 31),
|
||||
HEX_DBL(+, 1, 0000000000001, +, 31),
|
||||
HEX_DBL(+, 1, 0, +, 31),
|
||||
HEX_DBL(+, 1, fffffffffffff, +, 30),
|
||||
+1000.0,
|
||||
+100.0,
|
||||
+4.0,
|
||||
+3.5,
|
||||
+3.0,
|
||||
HEX_DBL(+, 1, 8000000000001, +, 1),
|
||||
+2.5,
|
||||
HEX_DBL(+, 1, 7ffffffffffff, +, 1),
|
||||
+2.0,
|
||||
HEX_DBL(+, 1, 8000000000001, +, 0),
|
||||
+1.5,
|
||||
HEX_DBL(+, 1, 7ffffffffffff, +, 0),
|
||||
HEX_DBL(+, 1, 0000000000001, +, 0),
|
||||
+1.0,
|
||||
HEX_DBL(+, 1, fffffffffffff, -, 1),
|
||||
HEX_DBL(+, 1, 0000000000001, -, 1),
|
||||
+0.5,
|
||||
HEX_DBL(+, 1, fffffffffffff, -, 2),
|
||||
HEX_DBL(+, 1, 0000000000001, -, 2),
|
||||
+0.25,
|
||||
HEX_DBL(+, 1, fffffffffffff, -, 3),
|
||||
HEX_DBL(+, 1, 0000000000001, -, 1022),
|
||||
+std::numeric_limits<double>::min(),
|
||||
HEX_DBL(+, 0, fffffffffffff, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000fff, -, 1022),
|
||||
HEX_DBL(+, 0, 00000000000fe, -, 1022),
|
||||
HEX_DBL(+, 0, 000000000000e, -, 1022),
|
||||
HEX_DBL(+, 0, 000000000000c, -, 1022),
|
||||
HEX_DBL(+, 0, 000000000000a, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000008, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000007, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000006, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000005, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000004, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000003, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000002, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000001, -, 1022),
|
||||
+0.0,
|
||||
};
|
||||
|
||||
inline constexpr std::array kSpecialValuesFloat{
|
||||
-std::numeric_limits<float>::quiet_NaN(),
|
||||
-std::numeric_limits<float>::infinity(),
|
||||
-std::numeric_limits<float>::max(),
|
||||
HEX_FLT(-, 1, 000002, +, 64),
|
||||
HEX_FLT(-, 1, 0, +, 64),
|
||||
HEX_FLT(-, 1, fffffe, +, 63),
|
||||
HEX_FLT(-, 1, 000002, +, 63),
|
||||
HEX_FLT(-, 1, 0, +, 63),
|
||||
HEX_FLT(-, 1, fffffe, +, 62),
|
||||
HEX_FLT(-, 1, 000002, +, 32),
|
||||
HEX_FLT(-, 1, 0, +, 32),
|
||||
HEX_FLT(-, 1, fffffe, +, 31),
|
||||
HEX_FLT(-, 1, 000002, +, 31),
|
||||
HEX_FLT(-, 1, 0, +, 31),
|
||||
HEX_FLT(-, 1, fffffe, +, 30),
|
||||
-1000.f,
|
||||
-100.f,
|
||||
-4.0f,
|
||||
-3.5f,
|
||||
-3.0f,
|
||||
HEX_FLT(-, 1, 800002, +, 1),
|
||||
-2.5f,
|
||||
HEX_FLT(-, 1, 7ffffe, +, 1),
|
||||
-2.0f,
|
||||
HEX_FLT(-, 1, 800002, +, 0),
|
||||
-1.5f,
|
||||
HEX_FLT(-, 1, 7ffffe, +, 0),
|
||||
HEX_FLT(-, 1, 000002, +, 0),
|
||||
-1.0f,
|
||||
HEX_FLT(-, 1, fffffe, -, 1),
|
||||
HEX_FLT(-, 1, 000002, -, 1),
|
||||
-0.5f,
|
||||
HEX_FLT(-, 1, fffffe, -, 2),
|
||||
HEX_FLT(-, 1, 000002, -, 2),
|
||||
-0.25f,
|
||||
HEX_FLT(-, 1, fffffe, -, 3),
|
||||
HEX_FLT(-, 1, 000002, -, 126),
|
||||
-std::numeric_limits<float>::min(),
|
||||
HEX_FLT(-, 0, fffffe, -, 126),
|
||||
HEX_FLT(-, 0, 000ffe, -, 126),
|
||||
HEX_FLT(-, 0, 0000fe, -, 126),
|
||||
HEX_FLT(-, 0, 00000e, -, 126),
|
||||
HEX_FLT(-, 0, 00000c, -, 126),
|
||||
HEX_FLT(-, 0, 00000a, -, 126),
|
||||
HEX_FLT(-, 0, 000008, -, 126),
|
||||
HEX_FLT(-, 0, 000006, -, 126),
|
||||
HEX_FLT(-, 0, 000004, -, 126),
|
||||
HEX_FLT(-, 0, 000002, -, 126),
|
||||
-0.0f,
|
||||
|
||||
std::numeric_limits<float>::quiet_NaN(),
|
||||
std::numeric_limits<float>::infinity(),
|
||||
std::numeric_limits<float>::max(),
|
||||
HEX_FLT(+, 1, 000002, +, 64),
|
||||
HEX_FLT(+, 1, 0, +, 64),
|
||||
HEX_FLT(+, 1, fffffe, +, 63),
|
||||
HEX_FLT(+, 1, 000002, +, 63),
|
||||
HEX_FLT(+, 1, 0, +, 63),
|
||||
HEX_FLT(+, 1, fffffe, +, 62),
|
||||
HEX_FLT(+, 1, 000002, +, 32),
|
||||
HEX_FLT(+, 1, 0, +, 32),
|
||||
HEX_FLT(+, 1, fffffe, +, 31),
|
||||
HEX_FLT(+, 1, 000002, +, 31),
|
||||
HEX_FLT(+, 1, 0, +, 31),
|
||||
HEX_FLT(+, 1, fffffe, +, 30),
|
||||
+1000.f,
|
||||
+100.f,
|
||||
+4.0f,
|
||||
+3.5f,
|
||||
+3.0f,
|
||||
HEX_FLT(+, 1, 800002, +, 1),
|
||||
2.5f,
|
||||
HEX_FLT(+, 1, 7ffffe, +, 1),
|
||||
+2.0f,
|
||||
HEX_FLT(+, 1, 800002, +, 0),
|
||||
1.5f,
|
||||
HEX_FLT(+, 1, 7ffffe, +, 0),
|
||||
HEX_FLT(+, 1, 000002, +, 0),
|
||||
+1.0f,
|
||||
HEX_FLT(+, 1, fffffe, -, 1),
|
||||
HEX_FLT(+, 1, 000002, -, 1),
|
||||
+0.5f,
|
||||
HEX_FLT(+, 1, fffffe, -, 2),
|
||||
HEX_FLT(+, 1, 000002, -, 2),
|
||||
+0.25f,
|
||||
HEX_FLT(+, 1, fffffe, -, 3),
|
||||
HEX_FLT(+, 1, 000002, -, 126),
|
||||
+std::numeric_limits<float>::min(),
|
||||
HEX_FLT(+, 0, fffffe, -, 126),
|
||||
HEX_FLT(+, 0, 000ffe, -, 126),
|
||||
HEX_FLT(+, 0, 0000fe, -, 126),
|
||||
HEX_FLT(+, 0, 00000e, -, 126),
|
||||
HEX_FLT(+, 0, 00000c, -, 126),
|
||||
HEX_FLT(+, 0, 00000a, -, 126),
|
||||
HEX_FLT(+, 0, 000008, -, 126),
|
||||
HEX_FLT(+, 0, 000006, -, 126),
|
||||
HEX_FLT(+, 0, 000004, -, 126),
|
||||
HEX_FLT(+, 0, 000002, -, 126),
|
||||
+0.0f,
|
||||
};
|
||||
|
||||
inline constexpr std::array kSpecialValuesInt{
|
||||
0, 1, 2, 3, 126, 127, 128, 1022, 1023, 1024, 0x02000001, 0x04000001, 1465264071, 1488522147,
|
||||
std::numeric_limits<int>::max(), -1, -2, -3, -126, -127, -128, -1022, -1023, -11024, -0x02000001,
|
||||
-0x04000001, -1465264071, -1488522147, std::numeric_limits<int>::min(), -std::numeric_limits<int>::max()
|
||||
};
|
||||
|
||||
template <typename T> struct SpecialVals {
|
||||
const T* const data;
|
||||
const size_t size;
|
||||
};
|
||||
|
||||
inline constexpr auto kSpecialValRegistry =
|
||||
std::make_tuple(SpecialVals<float>{kSpecialValuesFloat.data(), kSpecialValuesFloat.size()},
|
||||
SpecialVals<double>{kSpecialValuesDouble.data(), kSpecialValuesDouble.size()},
|
||||
SpecialVals<int>{kSpecialValuesInt.data(), kSpecialValuesInt.size()});
|
||||
@@ -0,0 +1,96 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "misc_negative_kernels_rtc.hh"
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "binary_common.hh"
|
||||
#include "ternary_common.hh"
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(fabs, std::fabs, 0, 0)
|
||||
TEST_CASE("Unit_Device_fabs_fabsf_Negative_RTC") { NegativeTestRTCWrapper<4>(kFabs); }
|
||||
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(copysign, std::copysign, 0, 0)
|
||||
TEST_CASE("Unit_Device_copysign_copysignf_Negative_RTC") { NegativeTestRTCWrapper<8>(kCopySign); }
|
||||
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(fmax, std::fmax, 0, 0)
|
||||
TEST_CASE("Unit_Device_fmax_fmaxf_Negative_RTC") { NegativeTestRTCWrapper<8>(kFmax); }
|
||||
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(fmin, std::fmin, 0, 0)
|
||||
TEST_CASE("Unit_Device_fmin_fminf_Negative_RTC") { NegativeTestRTCWrapper<8>(kFmin); }
|
||||
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(nextafter, std::nextafter, 0, 0)
|
||||
TEST_CASE("Unit_Device_nextafter_nextafterf_Negative_RTC") {
|
||||
NegativeTestRTCWrapper<8>(kNextAfter);
|
||||
}
|
||||
|
||||
MATH_TERNARY_WITHIN_ULP_TEST_DEF(fma, std::fma, 0, 0)
|
||||
TEST_CASE("Unit_Device_fma_fmaf_Negative_RTC") { NegativeTestRTCWrapper<12>(kFma); }
|
||||
|
||||
__global__ void fdividef_kernel(float* const ys, const size_t num_xs, float* const x1s,
|
||||
float* const x2s) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
ys[i] = fdividef(x1s[i], x2s[i]);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_fdividef_Accuracy_Positive") {
|
||||
double (*ref)(double, double) = [](double x1, double x2) { return x1 / x2; };
|
||||
BinaryFloatingPointTest(fdividef_kernel, ref, ULPValidatorBuilderFactory<float>(0));
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_fdividef_Negative_RTC") { NegativeTestRTCWrapper<4>(kFdividef); }
|
||||
|
||||
#define MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(kern_name, ref_func) \
|
||||
template <typename T> \
|
||||
__global__ void kern_name##_kernel(bool* const ys, const size_t num_xs, T* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = kern_name(xs[i]); \
|
||||
} \
|
||||
} \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - float") { \
|
||||
bool (*ref)(double) = ref_func; \
|
||||
UnarySinglePrecisionTest(kern_name##_kernel<float>, ref, EqValidatorBuilderFactory<bool>()); \
|
||||
} \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - double") { \
|
||||
bool (*ref)(long double) = ref_func; \
|
||||
UnaryDoublePrecisionTest(kern_name##_kernel<double>, ref, EqValidatorBuilderFactory<bool>()); \
|
||||
}
|
||||
|
||||
MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(isfinite, std::isfinite)
|
||||
TEST_CASE("Unit_Device_isfinite_Negative_RTC") { NegativeTestRTCWrapper<4>(kIsFinite); }
|
||||
|
||||
MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(isinf, std::isinf)
|
||||
TEST_CASE("Unit_Device_isinf_Negative_RTC") { NegativeTestRTCWrapper<4>(kIsInf); }
|
||||
|
||||
MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(isnan, std::isnan)
|
||||
TEST_CASE("Unit_Device_isnan_Negative_RTC") { NegativeTestRTCWrapper<4>(kIsNan); }
|
||||
|
||||
MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(signbit, std::signbit)
|
||||
TEST_CASE("Unit_Device_signbit_Negative_RTC") { NegativeTestRTCWrapper<4>(kSignBit); }
|
||||
@@ -0,0 +1,87 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define MISC_UNARY_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); } \
|
||||
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); }
|
||||
|
||||
#define MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(float* x) { bool result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { bool result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v3(double* x) { bool result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v4(Dummy x) { bool result = func_name(x); }
|
||||
|
||||
#define MISC_BINARY_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##f_kernel_v1(float* x, float y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x, float y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v3(float x, float* y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##f_kernel_v4(float x, Dummy y) { float result = func_name##f(x, y); } \
|
||||
__global__ void func_name##_kernel_v1(double* x, double y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x, double y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(double x, double* y) { double result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(double x, Dummy y) { double result = func_name(x, y); }
|
||||
|
||||
/*Expecting 4 errors*/
|
||||
MISC_UNARY_NEGATIVE_KERNELS(fabs)
|
||||
|
||||
/*Expecting 8 errors per macro invocation - 40 total*/
|
||||
MISC_BINARY_NEGATIVE_KERNELS(copysign)
|
||||
MISC_BINARY_NEGATIVE_KERNELS(fmax)
|
||||
MISC_BINARY_NEGATIVE_KERNELS(fmin)
|
||||
MISC_BINARY_NEGATIVE_KERNELS(nextafter)
|
||||
MISC_BINARY_NEGATIVE_KERNELS(fma)
|
||||
|
||||
/*Expecting 4 errors*/
|
||||
__global__ void fdividef_kernel_v1(float* x, float y) { float result = fdividef(x, y); }
|
||||
__global__ void fdividef_kernel_v2(Dummy x, float y) { float result = fdivide(x); }
|
||||
__global__ void fdividef_kernel_v3(float x, float* y) { float result = fdivide(x); }
|
||||
__global__ void fdividef_kernel_v4(float x, Dummy y) { float result = fdivide(x); }
|
||||
|
||||
/*Expecting 4 errors per macro invocation - 16 total*/
|
||||
MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(isfinite)
|
||||
MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(isinf)
|
||||
MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(isnan)
|
||||
MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(signbit)
|
||||
|
||||
/*Expecting 12 errors*/
|
||||
__global__ void fmaf_kernel_v1(float* x, float y, float z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v2(Dummy x, float y, float z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v3(float x, float* y, float z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v4(float x, Dummy y, float z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v5(float x, float y, float* z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v6(float x, float y, Dummy z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v1(double* x, double y, double z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v2(Dummy x, double y, double z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v3(double x, double* y, double z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v4(double x, Dummy y, double z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v5(double x, double y, double* z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v6(double x, double y, Dummy z) { double result = fmaf(x, y, z); }
|
||||
@@ -0,0 +1,177 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
|
||||
static constexpr auto kFabs{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void fabsf_kernel_v1(float* x) { float result = fabsf(x); }
|
||||
__global__ void fabsf_kernel_v2(Dummy x) { float result = fabsf(x); }
|
||||
__global__ void fabs_kernel_v1(double* x) { double result = fabs(x); }
|
||||
__global__ void fabs_kernel_v2(Dummy x) { double result = fabs(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kCopySign{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void copysignf_kernel_v1(float* x, float y) { float result = copysignf(x, y); }
|
||||
__global__ void copysignf_kernel_v2(Dummy x, float y) { float result = copysignf(x, y); }
|
||||
__global__ void copysignf_kernel_v3(float x, float* y) { float result = copysignf(x, y); }
|
||||
__global__ void copysignf_kernel_v4(float x, Dummy y) { float result = copysignf(x, y); }
|
||||
__global__ void copysign_kernel_v1(double* x, double y) { double result = copysign(x, y); }
|
||||
__global__ void copysign_kernel_v2(Dummy x, double y) { double result = copysign(x, y); }
|
||||
__global__ void copysign_kernel_v3(double x, double* y) { double result = copysign(x, y); }
|
||||
__global__ void copysign_kernel_v4(double x, Dummy y) { double result = copysign(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFmax{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void fmaxf_kernel_v1(float* x, float y) { float result = fmaxf(x, y); }
|
||||
__global__ void fmaxf_kernel_v2(Dummy x, float y) { float result = fmaxf(x, y); }
|
||||
__global__ void fmaxf_kernel_v3(float x, float* y) { float result = fmaxf(x, y); }
|
||||
__global__ void fmaxf_kernel_v4(float x, Dummy y) { float result = fmaxf(x, y); }
|
||||
__global__ void fmax_kernel_v1(double* x, double y) { double result = fmax(x, y); }
|
||||
__global__ void fmax_kernel_v2(Dummy x, double y) { double result = fmax(x, y); }
|
||||
__global__ void fmax_kernel_v3(double x, double* y) { double result = fmax(x, y); }
|
||||
__global__ void fmax_kernel_v4(double x, Dummy y) { double result = fmax(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFmin{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void fminf_kernel_v1(float* x, float y) { float result = fminf(x, y); }
|
||||
__global__ void fminf_kernel_v2(Dummy x, float y) { float result = fminf(x, y); }
|
||||
__global__ void fminf_kernel_v3(float x, float* y) { float result = fminf(x, y); }
|
||||
__global__ void fminf_kernel_v4(float x, Dummy y) { float result = fminf(x, y); }
|
||||
__global__ void fmin_kernel_v1(double* x, double y) { double result = fmin(x, y); }
|
||||
__global__ void fmin_kernel_v2(Dummy x, double y) { double result = fmin(x, y); }
|
||||
__global__ void fmin_kernel_v3(double x, double* y) { double result = fmin(x, y); }
|
||||
__global__ void fmin_kernel_v4(double x, Dummy y) { double result = fmin(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kNextAfter{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void nextafterf_kernel_v1(float* x, float y) { float result = nextafterf(x, y); }
|
||||
__global__ void nextafterf_kernel_v2(Dummy x, float y) { float result = nextafterf(x, y); }
|
||||
__global__ void nextafterf_kernel_v3(float x, float* y) { float result = nextafterf(x, y); }
|
||||
__global__ void nextafterf_kernel_v4(float x, Dummy y) { float result = nextafterf(x, y); }
|
||||
__global__ void nextafter_kernel_v1(double* x, double y) { double result = nextafter(x, y); }
|
||||
__global__ void nextafter_kernel_v2(Dummy x, double y) { double result = nextafter(x, y); }
|
||||
__global__ void nextafter_kernel_v3(double x, double* y) { double result = nextafter(x, y); }
|
||||
__global__ void nextafter_kernel_v4(double x, Dummy y) { double result = nextafter(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFma{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void fmaf_kernel_v1(float* x, float y, float z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v2(Dummy x, float y, float z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v3(float x, float* y, float z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v4(float x, Dummy y, float z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v5(float x, float y, float* z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fmaf_kernel_v6(float x, float y, Dummy z) { float result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v1(double* x, double y, double z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v2(Dummy x, double y, double z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v3(double x, double* y, double z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v4(double x, Dummy y, double z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v5(double x, double y, double* z) { double result = fmaf(x, y, z); }
|
||||
__global__ void fma_kernel_v6(double x, double y, Dummy z) { double result = fmaf(x, y, z); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kFdividef{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void fdividef_kernel_v1(float* x, float y) { float result = fdividef(x, y); }
|
||||
__global__ void fdividef_kernel_v2(Dummy x, float y) { float result = fdivide(x); }
|
||||
__global__ void fdividef_kernel_v3(float x, float* y) { float result = fdivide(x); }
|
||||
__global__ void fdividef_kernel_v4(float x, Dummy y) { float result = fdivide(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kIsFinite{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void isfinite_kernel_v1(float* x) { bool result = isfinite(x); }
|
||||
__global__ void isfinite_kernel_v2(Dummy x) { bool result = isfinite(x); }
|
||||
__global__ void isfinite_kernel_v3(double* x) { bool result = isfinite(x); }
|
||||
__global__ void isfinite_kernel_v4(Dummy x) { bool result = isfinite(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kIsInf{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void isinf_kernel_v1(float* x) { bool result = isinf(x); }
|
||||
__global__ void isinf_kernel_v2(Dummy x) { bool result = isinf(x); }
|
||||
__global__ void isinf_kernel_v3(double* x) { bool result = isinf(x); }
|
||||
__global__ void isinf_kernel_v4(Dummy x) { bool result = isinf(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kIsNan{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void isnan_kernel_v1(float* x) { bool result = isnan(x); }
|
||||
__global__ void isnan_kernel_v2(Dummy x) { bool result = isnan(x); }
|
||||
__global__ void isnan_kernel_v3(double* x) { bool result = isnan(x); }
|
||||
__global__ void isnan_kernel_v4(Dummy x) { bool result = isnan(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kSignBit{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void signbit_kernel_v1(float* x) { bool result = signbit(x); }
|
||||
__global__ void signbit_kernel_v2(Dummy x) { bool result = signbit(x); }
|
||||
__global__ void signbit_kernel_v3(double* x) { bool result = signbit(x); }
|
||||
__global__ void signbit_kernel_v4(Dummy x) { bool result = signbit(x); }
|
||||
)"};
|
||||
@@ -0,0 +1,134 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "math_common.hh"
|
||||
#include "math_special_values.hh"
|
||||
|
||||
#include <hip/hip_cooperative_groups.h>
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
#define MATH_POW_INT_KERNEL_DEF(func_name) \
|
||||
template <typename T1, typename T2> \
|
||||
__global__ void func_name##_kernel(T1* const ys, const size_t num_xs, T1* const x1s, \
|
||||
T2* const x2s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
if constexpr (std::is_same_v<float, T1>) { \
|
||||
ys[i] = func_name##f(x1s[i], x2s[i]); \
|
||||
} else if constexpr (std::is_same_v<double, T1>) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i]); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
template <typename T1, typename T2>
|
||||
using kernel_pow_int_sig = void (*)(T1*, const size_t, T1*, T2*);
|
||||
|
||||
template <typename T1, typename T2> using ref_pow_int_sig = T1 (*)(T1, T2);
|
||||
|
||||
template <typename T1, typename T2, typename RT1, typename RT2, typename ValidatorBuilder>
|
||||
void PowIntFloatingPointBruteForceTest(kernel_pow_int_sig<T1, T2> kernel,
|
||||
ref_pow_int_sig<RT1, RT2> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const uint64_t num_iterations = GetTestIterationCount();
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(T1) * 2 + sizeof(T2)), num_iterations);
|
||||
LinearAllocGuard<T1> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(T1)};
|
||||
LinearAllocGuard<T2> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(T2)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
thread_pool.Post([=, &x1s, &x2s] {
|
||||
const auto generator1 = [=] {
|
||||
static thread_local std::mt19937 rng(std::random_device{}());
|
||||
std::uniform_real_distribution<RefType_t<T1>> unif_dist(std::numeric_limits<T1>::lowest(),
|
||||
std::numeric_limits<T1>::max());
|
||||
return static_cast<T1>(unif_dist(rng));
|
||||
};
|
||||
const auto generator2 = [] {
|
||||
static thread_local std::mt19937 rng(std::random_device{}());
|
||||
std::uniform_int_distribution<T2> unif_dist(std::numeric_limits<T2>::lowest(),
|
||||
std::numeric_limits<T2>::max());
|
||||
return unif_dist(rng);
|
||||
};
|
||||
std::generate(x1s.ptr() + base_idx, x1s.ptr() + base_idx + sub_batch_size, generator1);
|
||||
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator2);
|
||||
});
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
|
||||
x2s.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T1, typename T2, typename RT1, typename RT2, typename ValidatorBuilder>
|
||||
void PowIntFloatingPointSpecialValuesTest(kernel_pow_int_sig<T1, T2> kernel,
|
||||
ref_pow_int_sig<RT1, RT2> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const auto values1 = std::get<SpecialVals<T1>>(kSpecialValRegistry);
|
||||
const auto values2 = std::get<SpecialVals<int>>(kSpecialValRegistry);
|
||||
|
||||
const auto size = values1.size * values2.size;
|
||||
LinearAllocGuard<T1> x1s{LinearAllocs::hipHostMalloc, size * sizeof(T1)};
|
||||
LinearAllocGuard<T2> x2s{LinearAllocs::hipHostMalloc, size * sizeof(T2)};
|
||||
|
||||
for (auto i = 0u; i < values1.size; ++i) {
|
||||
for (auto j = 0u; j < values2.size; ++j) {
|
||||
x1s.ptr()[i * values2.size + j] = values1.data[i];
|
||||
x2s.ptr()[i * values2.size + j] = static_cast<T2>(values2.data[j]);
|
||||
}
|
||||
}
|
||||
|
||||
MathTest math_test(kernel, size);
|
||||
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, size, x1s.ptr(),
|
||||
x2s.ptr());
|
||||
}
|
||||
|
||||
template <typename T1, typename T2, typename RT1, typename RT2, typename ValidatorBuilder>
|
||||
void PowIntFloatingPointTest(kernel_pow_int_sig<T1, T2> kernel, ref_pow_int_sig<RT1, RT2> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Special values") {
|
||||
PowIntFloatingPointSpecialValuesTest(kernel, ref_func, validator_builder);
|
||||
}
|
||||
|
||||
SECTION("Brute force") { PowIntFloatingPointBruteForceTest(kernel, ref_func, validator_builder); }
|
||||
}
|
||||
@@ -0,0 +1,455 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "binary_common.hh"
|
||||
#include "pow_common.hh"
|
||||
#include "math_pow_negative_kernels_rtc.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup PowMathFuncs PowMathFuncs
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
/********** Unary Functions **********/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `expf(x)` for all possible inputs and `exp(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::exp(T)`. The maximum ulp error for single
|
||||
* precision is 2 and for double precision is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(exp, 2, 1)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for expf and exp.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_exp_expf_Negative_RTC") { NegativeTestRTCWrapper<4>(kExp); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `exp2f(x)` for all possible inputs and `exp2(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::exp2(T)`. The maximum ulp error for single
|
||||
* precision is 2 and for double precision is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(exp2, 2, 1)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for exp2f and exp2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_exp2_exp2f_Negative_RTC") { NegativeTestRTCWrapper<4>(kExp2); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `expm1f(x)` for all possible inputs and `expm1(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::exp(T)`. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(expm1, 1, 1)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for expm1f and expm1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_expm1_expm1f_Negative_RTC") { NegativeTestRTCWrapper<4>(kExpm1); }
|
||||
|
||||
MATH_UNARY_KERNEL_DEF(exp10)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `exp10f(x)` for all possible inputs. The maximum ulp error
|
||||
* is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_exp10f_Accuracy_Positive") {
|
||||
auto exp10_ref = [](double arg) -> double { return std::pow(10, arg); };
|
||||
double (*ref)(double) = exp10_ref;
|
||||
UnarySinglePrecisionTest(exp10_kernel<float>, ref, ULPValidatorBuilderFactory<float>(2));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `exp10(x)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_exp10_Accuracy_Positive") {
|
||||
auto exp10_ref = [](long double arg) -> long double { return std::pow(10, arg); };
|
||||
long double (*ref)(long double) = exp10_ref;
|
||||
UnaryDoublePrecisionTest(exp10_kernel<double>, ref, ULPValidatorBuilderFactory<double>(1));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for exp10f and exp10.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_exp10_exp10f_Negative_RTC") { NegativeTestRTCWrapper<4>(kExp10); }
|
||||
|
||||
template <typename T>
|
||||
__global__ void frexp_kernel(std::pair<T, int>* const ys, const size_t num_xs, T* const xs) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
if constexpr (std::is_same_v<float, T>) {
|
||||
ys[i].first = frexpf(xs[i], &ys[i].second);
|
||||
} else if constexpr (std::is_same_v<double, T>) {
|
||||
ys[i].first = frexp(xs[i], &ys[i].second);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> std::pair<T, int> frexp_ref(T arg) {
|
||||
int exp_v;
|
||||
T res = std::frexp(arg, &exp_v);
|
||||
return {res, exp_v};
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `frexpf(x, exp)` for all possible inputs. The results are
|
||||
* compared against reference function `double std::frexp(double, int*)`. The maximum ulp error is
|
||||
* 0.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_frexpf_Accuracy_Positive") {
|
||||
UnarySinglePrecisionTest(
|
||||
frexp_kernel<float>, frexp_ref<double>,
|
||||
PairValidatorBuilderFactory<float, int>(ULPValidatorBuilderFactory<float>(0),
|
||||
EqValidatorBuilderFactory<int>()));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `frexp(x, exp)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are
|
||||
* compared against reference function `long double std::frexp(long double, int*)`. The maximum ulp
|
||||
* error is 0.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_frexp_Accuracy_Positive") {
|
||||
UnaryDoublePrecisionTest(
|
||||
frexp_kernel<double>, frexp_ref<long double>,
|
||||
PairValidatorBuilderFactory<double, int>(ULPValidatorBuilderFactory<double>(0),
|
||||
EqValidatorBuilderFactory<int>()));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for frexpf and frexp.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_frexp_frexpf_Negative_RTC") { NegativeTestRTCWrapper<20>(kFrexp); }
|
||||
|
||||
|
||||
/********** Binary Functions **********/
|
||||
|
||||
MATH_BINARY_KERNEL_DEF(pow)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `powf(x, y)` and `pow(x, y)`against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::pow(T, T)`. The maximum ulp error
|
||||
* for single precision is 4 and for double precision is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_pow_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto pow_ref = [](RT arg1, RT arg2) -> RT {
|
||||
if (std::isinf(arg1) && arg2 < 0) return 0;
|
||||
return std::pow(arg1, arg2);
|
||||
};
|
||||
RT (*ref)(RT, RT) = pow_ref;
|
||||
const auto ulp = std::is_same_v<float, TestType> ? 4 : 2;
|
||||
BinaryFloatingPointTest(pow_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(ulp));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for powf and pow.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_pow_powf_Negative_RTC") { NegativeTestRTCWrapper<8>(kPow); }
|
||||
|
||||
MATH_POW_INT_KERNEL_DEF(ldexp)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `ldexpf(x, exp)` and `ldexp(x, exp)`against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::ldexp(T, int)`. The maximum ulp error is 0.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_ldexp_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
RT (*ref)(RT, int) = std::ldexp;
|
||||
PowIntFloatingPointTest(ldexp_kernel<TestType, int>, ref,
|
||||
ULPValidatorBuilderFactory<TestType>(0));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for ldexpf and ldexp.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_ldexp_ldexpf_Negative_RTC") { NegativeTestRTCWrapper<8>(kLdexp); }
|
||||
|
||||
MATH_POW_INT_KERNEL_DEF(powi)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `powi(x, exp)` and `powi(x, exp)`against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::pow(T, T)`. The maximum ulp error
|
||||
* for single precision is 4 and for double precision is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_powi_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto pow_ref = [](RT arg1, int arg2) -> RT {
|
||||
if (std::isinf(arg1) && arg2 < 0) return 0;
|
||||
return std::pow(arg1, static_cast<RT>(arg2));
|
||||
};
|
||||
RT (*ref)(RT, int) = pow_ref;
|
||||
const auto ulp = std::is_same_v<float, TestType> ? 4 : 2;
|
||||
PowIntFloatingPointTest(powi_kernel<TestType, int>, ref,
|
||||
ULPValidatorBuilderFactory<TestType>(ulp));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for powif and powi.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_powi_powif_Negative_RTC") { NegativeTestRTCWrapper<8>(kPowi); }
|
||||
|
||||
MATH_POW_INT_KERNEL_DEF(scalbn)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `scalbnf(x, n)` and `scalbn(x, n)`against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::scalbn(T, int)`. The maximum ulp error is 0.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_scalbn_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
RT (*ref)(RT, int) = std::scalbn;
|
||||
PowIntFloatingPointTest(scalbn_kernel<TestType, int>, ref,
|
||||
ULPValidatorBuilderFactory<TestType>(0));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for scalbnf and scalbn.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_scalbn_scalbnf_Negative_RTC") { NegativeTestRTCWrapper<8>(kScalbn); }
|
||||
|
||||
MATH_POW_INT_KERNEL_DEF(scalbln)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `scalblnf(x, l)` and `scalbln(x, l)`against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::scalbn(T, long int)`. The maximum ulp error is 0.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_scalbln_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
RT (*ref)(RT, long int) = std::scalbln;
|
||||
PowIntFloatingPointTest(scalbln_kernel<TestType, long int>, ref,
|
||||
ULPValidatorBuilderFactory<TestType>(0));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for scalblnf and scalbln.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/pow_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_scalbln_scalblnf_Negative_RTC") { NegativeTestRTCWrapper<8>(kScalbln); }
|
||||
@@ -0,0 +1,246 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "math_common.hh"
|
||||
#include "math_special_values.hh"
|
||||
|
||||
#include <hip/hip_cooperative_groups.h>
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
#define MATH_QUATERNARY_KERNEL_DEF(func_name) \
|
||||
template <typename T> \
|
||||
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, T* const x1s, T* const x2s, \
|
||||
T* const x3s, T* const x4s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
if constexpr (std::is_same_v<float, T>) { \
|
||||
ys[i] = func_name##f(x1s[i], x2s[i], x3s[i], x4s[i]); \
|
||||
} else if constexpr (std::is_same_v<double, T>) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i], x3s[i], x4s[i]); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
inline constexpr std::array kSpecialValuesReducedDouble{
|
||||
-std::numeric_limits<double>::quiet_NaN(),
|
||||
-std::numeric_limits<double>::infinity(),
|
||||
-std::numeric_limits<double>::max(),
|
||||
HEX_DBL(-, 1, 0000000000001, +, 64),
|
||||
HEX_DBL(-, 1, fffffffffffff, +, 63),
|
||||
HEX_DBL(-, 1, fffffffffffff, +, 62),
|
||||
HEX_DBL(-, 1, 0, +, 32),
|
||||
HEX_DBL(-, 1, 0000000000001, +, 31),
|
||||
HEX_DBL(-, 1, fffffffffffff, +, 30),
|
||||
-1000.0,
|
||||
-3.5,
|
||||
HEX_DBL(-, 1, 8000000000001, +, 1),
|
||||
-2.5,
|
||||
HEX_DBL(-, 1, 8000000000001, +, 0),
|
||||
-1.5,
|
||||
-0.5,
|
||||
-0.25,
|
||||
HEX_DBL(-, 1, fffffffffffff, -, 3),
|
||||
-std::numeric_limits<double>::min(),
|
||||
HEX_DBL(-, 0, fffffffffffff, -, 1022),
|
||||
HEX_DBL(-, 0, 0000000000001, -, 1022),
|
||||
-0.0,
|
||||
|
||||
std::numeric_limits<double>::quiet_NaN(),
|
||||
std::numeric_limits<double>::infinity(),
|
||||
std::numeric_limits<double>::max(),
|
||||
HEX_DBL(+, 1, 0, +, 64),
|
||||
HEX_DBL(+, 1, 0000000000001, +, 63),
|
||||
HEX_DBL(+, 1, 000002, +, 32),
|
||||
HEX_DBL(+, 1, fffffffffffff, +, 31),
|
||||
HEX_DBL(+, 1, 0, +, 31),
|
||||
HEX_DBL(+, 1, fffffffffffff, +, 30),
|
||||
+100.0,
|
||||
+3.0,
|
||||
HEX_DBL(+, 1, 7ffffffffffff, +, 1),
|
||||
+2.0,
|
||||
HEX_DBL(+, 1, 7ffffffffffff, +, 0),
|
||||
+1.0,
|
||||
HEX_DBL(+, 1, fffffffffffff, -, 2),
|
||||
+std::numeric_limits<double>::min(),
|
||||
HEX_DBL(+, 0, 0000000000fff, -, 1022),
|
||||
HEX_DBL(+, 0, 0000000000007, -, 1022),
|
||||
+0.0,
|
||||
};
|
||||
|
||||
inline constexpr std::array kSpecialValuesReducedFloat{
|
||||
-std::numeric_limits<float>::quiet_NaN(),
|
||||
-std::numeric_limits<float>::infinity(),
|
||||
-std::numeric_limits<float>::max(),
|
||||
HEX_FLT(-, 1, 000002, +, 64),
|
||||
HEX_FLT(-, 1, fffffe, +, 63),
|
||||
HEX_FLT(-, 1, fffffe, +, 62),
|
||||
HEX_FLT(-, 1, 0, +, 32),
|
||||
HEX_FLT(-, 1, fffffe, +, 31),
|
||||
HEX_FLT(-, 1, fffffe, +, 30),
|
||||
-1000.f,
|
||||
-3.5f,
|
||||
HEX_FLT(-, 1, 800002, +, 1),
|
||||
-2.5f,
|
||||
HEX_FLT(-, 1, 800002, +, 0),
|
||||
-1.5f,
|
||||
-0.5f,
|
||||
-0.25f,
|
||||
HEX_FLT(-, 1, fffffe, -, 3),
|
||||
-std::numeric_limits<float>::min(),
|
||||
HEX_FLT(-, 0, fffffe, -, 126),
|
||||
HEX_FLT(-, 0, 000002, -, 126),
|
||||
-0.0f,
|
||||
|
||||
std::numeric_limits<float>::quiet_NaN(),
|
||||
std::numeric_limits<float>::infinity(),
|
||||
std::numeric_limits<float>::max(),
|
||||
HEX_FLT(+, 1, 0, +, 64),
|
||||
HEX_FLT(+, 1, 000002, +, 63),
|
||||
HEX_FLT(+, 1, 000002, +, 32),
|
||||
HEX_FLT(+, 1, 000002, +, 31),
|
||||
HEX_FLT(+, 1, fffffe, +, 30),
|
||||
+100.f,
|
||||
+4.0f,
|
||||
HEX_FLT(+, 1, 7ffffe, +, 1),
|
||||
+2.0f,
|
||||
HEX_FLT(+, 1, 7ffffe, +, 0),
|
||||
+1.0f,
|
||||
HEX_FLT(+, 1, fffffe, -, 2),
|
||||
+std::numeric_limits<float>::min(),
|
||||
HEX_FLT(+, 0, 000ffe, -, 126),
|
||||
HEX_FLT(+, 0, 000006, -, 126),
|
||||
+0.0f,
|
||||
};
|
||||
|
||||
inline constexpr auto kSpecialValReducedRegistry = std::make_tuple(
|
||||
SpecialVals<float>{kSpecialValuesReducedFloat.data(), kSpecialValuesReducedFloat.size()},
|
||||
SpecialVals<double>{kSpecialValuesReducedDouble.data(), kSpecialValuesReducedDouble.size()});
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void QuaternaryFloatingPointBruteForceTest(kernel_sig<T, TArg, TArg, TArg, TArg> kernel,
|
||||
ref_sig<RT, RTArg, RTArg, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder,
|
||||
const TArg a = std::numeric_limits<TArg>::lowest(),
|
||||
const TArg b = std::numeric_limits<TArg>::max()) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const uint64_t num_iterations = GetTestIterationCount();
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(TArg) * 4 + sizeof(T)), num_iterations);
|
||||
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x3s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x4s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
thread_pool.Post([=, &x1s, &x2s, &x3s, &x4s] {
|
||||
const auto generator = [=] {
|
||||
static thread_local std::mt19937 rng(std::random_device{}());
|
||||
std::uniform_real_distribution<RefType_t<TArg>> unif_dist(a, b);
|
||||
return static_cast<TArg>(unif_dist(rng));
|
||||
};
|
||||
std::generate(x1s.ptr() + base_idx, x1s.ptr() + base_idx + sub_batch_size, generator);
|
||||
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator);
|
||||
std::generate(x3s.ptr() + base_idx, x3s.ptr() + base_idx + sub_batch_size, generator);
|
||||
std::generate(x4s.ptr() + base_idx, x4s.ptr() + base_idx + sub_batch_size, generator);
|
||||
});
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
|
||||
x2s.ptr(), x3s.ptr(), x4s.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void QuaternaryFloatingPointSpecialValuesTest(kernel_sig<T, TArg, TArg, TArg, TArg> kernel,
|
||||
ref_sig<RT, RTArg, RTArg, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const auto values = std::get<SpecialVals<TArg>>(kSpecialValReducedRegistry);
|
||||
|
||||
const auto size = values.size * values.size * values.size * values.size;
|
||||
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x3s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x4s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
|
||||
for (auto i = 0u; i < values.size; ++i) {
|
||||
for (auto j = 0u; j < values.size; ++j) {
|
||||
for (auto k = 0u; k < values.size; ++k) {
|
||||
for (auto l = 0u; l < values.size; ++l) {
|
||||
x1s.ptr()[((i * values.size + j) * values.size + k) * values.size + l] = values.data[i];
|
||||
x2s.ptr()[((i * values.size + j) * values.size + k) * values.size + l] = values.data[j];
|
||||
x3s.ptr()[((i * values.size + j) * values.size + k) * values.size + l] = values.data[k];
|
||||
x4s.ptr()[((i * values.size + j) * values.size + k) * values.size + l] = values.data[l];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MathTest math_test(kernel, size);
|
||||
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, size, x1s.ptr(),
|
||||
x2s.ptr(), x3s.ptr(), x4s.ptr());
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void QuaternaryFloatingPointTest(kernel_sig<T, TArg, TArg, TArg, TArg> kernel,
|
||||
ref_sig<RT, RTArg, RTArg, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Special values") {
|
||||
QuaternaryFloatingPointSpecialValuesTest(kernel, ref_func, validator_builder);
|
||||
}
|
||||
|
||||
SECTION("Brute force") {
|
||||
QuaternaryFloatingPointBruteForceTest(kernel, ref_func, validator_builder);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#define MATH_QUATERNARY_WITHIN_ULP_TEST_DEF(kern_name, ref_func, sp_ulp, dp_ulp) \
|
||||
MATH_QUATERNARY_KERNEL_DEF(kern_name) \
|
||||
\
|
||||
TEMPLATE_TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive", "", float, double) { \
|
||||
using RT = RefType_t<TestType>; \
|
||||
RT (*ref)(RT, RT, RT, RT) = ref_func; \
|
||||
const auto ulp = std::is_same_v<float, TestType> ? sp_ulp : dp_ulp; \
|
||||
\
|
||||
QuaternaryFloatingPointTest(kern_name##_kernel<TestType>, ref, \
|
||||
ULPValidatorBuilderFactory<TestType>(ulp)); \
|
||||
}
|
||||
@@ -0,0 +1,153 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "binary_common.hh"
|
||||
#include "math_remainder_rounding_negative_kernels_rtc.hh"
|
||||
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(fmod, std::fmod, 0, 0)
|
||||
TEST_CASE("Unit_Device_fmod_fmodf_Negative_RTC") { NegativeTestRTCWrapper<8>(kFmod); }
|
||||
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(remainder, std::remainder, 0, 0)
|
||||
TEST_CASE("Unit_Device_remainder_remainder_Negative_RTC") { NegativeTestRTCWrapper<8>(kRemainder); }
|
||||
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(fdim, std::fdim, 0, 0)
|
||||
TEST_CASE("Unit_Device_fdim_fdimf_Negative_RTC") { NegativeTestRTCWrapper<8>(kFdim); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(trunc, std::trunc, 0, 0)
|
||||
TEST_CASE("Unit_Device_trunc_truncf_Negative_RTC") { NegativeTestRTCWrapper<4>(kTrunc); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(round, std::round, 0, 0)
|
||||
TEST_CASE("Unit_Device_round_roundf_Negative_RTC") { NegativeTestRTCWrapper<4>(kRound); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(rint, std::rint, 0, 0)
|
||||
TEST_CASE("Unit_Device_rint_rintf_Negative_RTC") { NegativeTestRTCWrapper<4>(kRint); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(nearbyint, std::nearbyint, 0, 0)
|
||||
TEST_CASE("Unit_Device_nearbyint_nearbyintf_Negative_RTC") {
|
||||
NegativeTestRTCWrapper<4>(kNearbyint);
|
||||
}
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(ceil, std::ceil, 0, 0)
|
||||
TEST_CASE("Unit_Device_ceil_ceilf_Negative_RTC") { NegativeTestRTCWrapper<4>(kCeil); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(floor, std::floor, 0, 0)
|
||||
TEST_CASE("Unit_Device_floor_floorf_Negative_RTC") { NegativeTestRTCWrapper<4>(kFloor); }
|
||||
|
||||
|
||||
#define LONG_CONVERSION_FUNCTION_TEST_DEF(kern_name, ref_func, lt) \
|
||||
MATH_UNARY_KERNEL_DEF(kern_name) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - float") { \
|
||||
lt (*ref)(double) = ref_func; \
|
||||
UnarySinglePrecisionRangeTest(kern_name##_kernel<float, lt>, ref, \
|
||||
EqValidatorBuilderFactory<lt>(), \
|
||||
static_cast<float>(std::numeric_limits<lt>::lowest()), \
|
||||
static_cast<float>(std::numeric_limits<lt>::max())); \
|
||||
} \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - double") { \
|
||||
lt (*ref)(long double) = ref_func; \
|
||||
UnaryDoublePrecisionBruteForceTest(kern_name##_kernel<double, lt>, ref, \
|
||||
EqValidatorBuilderFactory<lt>(), \
|
||||
static_cast<double>(std::numeric_limits<lt>::lowest()), \
|
||||
static_cast<double>(std::numeric_limits<lt>::max())); \
|
||||
}
|
||||
|
||||
LONG_CONVERSION_FUNCTION_TEST_DEF(lrint, std::lrint, long)
|
||||
TEST_CASE("Unit_Device_lrint_lrintf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLrint); }
|
||||
|
||||
LONG_CONVERSION_FUNCTION_TEST_DEF(lround, std::lround, long)
|
||||
TEST_CASE("Unit_Device_lround_lroundf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLround); }
|
||||
|
||||
LONG_CONVERSION_FUNCTION_TEST_DEF(llrint, std::llrint, long long)
|
||||
TEST_CASE("Unit_Device_llrint_llrintf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLlrint); }
|
||||
|
||||
LONG_CONVERSION_FUNCTION_TEST_DEF(llround, std::llround, long long)
|
||||
TEST_CASE("Unit_Device_llround_llroundf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLlround); }
|
||||
|
||||
|
||||
template <typename T>
|
||||
__global__ void remquo_kernel(std::pair<T, int>* const ys, const size_t num_xs, T* const x1s,
|
||||
T* const x2s) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
if constexpr (std::is_same_v<float, T>) {
|
||||
ys[i].first = remquof(x1s[i], x2s[i], &ys[i].second);
|
||||
} else if constexpr (std::is_same_v<double, T>) {
|
||||
ys[i].first = remquo(x1s[i], x2s[i], &ys[i].second);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> std::pair<T, int> remquo_wrapper(T x1, T x2) {
|
||||
std::pair<T, int> ret;
|
||||
ret.first = std::remquo(x1, x2, &ret.second);
|
||||
return ret;
|
||||
}
|
||||
|
||||
TEMPLATE_TEST_CASE("Unit_Device_remquo_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
std::pair<RT, int> (*ref)(RT, RT) = remquo_wrapper;
|
||||
const auto ulp_builder = ULPValidatorBuilderFactory<TestType>(0);
|
||||
const auto eq_builder = EqValidatorBuilderFactory<int>();
|
||||
|
||||
BinaryFloatingPointTest(remquo_kernel<TestType>, ref,
|
||||
PairValidatorBuilderFactory<TestType, int>(ulp_builder, eq_builder));
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_remquo_remquof_Negative_RTC") { NegativeTestRTCWrapper<24>(kRemquo); }
|
||||
|
||||
template <typename T>
|
||||
__global__ void modf_kernel(std::pair<T, T>* const ys, const size_t num_xs, T* const xs) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
if constexpr (std::is_same_v<float, T>) {
|
||||
ys[i].first = modff(xs[i], &ys[i].second);
|
||||
} else if constexpr (std::is_same_v<double, T>) {
|
||||
ys[i].first = modf(xs[i], &ys[i].second);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> std::pair<T, T> modf_wrapper(T x) {
|
||||
std::pair<T, T> ret;
|
||||
ret.first = std::modf(x, &ret.second);
|
||||
return ret;
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_modf_Accuracy_Positive - float") {
|
||||
UnarySinglePrecisionTest(
|
||||
modf_kernel<float>, modf_wrapper<double>,
|
||||
PairValidatorBuilderFactory<float>(ULPValidatorBuilderFactory<float>(0)));
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_modf_Accuracy_Positive - double") {
|
||||
UnaryDoublePrecisionTest(
|
||||
modf_kernel<double>, modf_wrapper<long double>,
|
||||
PairValidatorBuilderFactory<double>(ULPValidatorBuilderFactory<double>(0)));
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_modf_modff_Negative_RTC") { NegativeTestRTCWrapper<20>(kModf); }
|
||||
@@ -0,0 +1,604 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "binary_common.hh"
|
||||
#include "ternary_common.hh"
|
||||
#include "quaternary_common.hh"
|
||||
#include "math_root_negative_kernels_rtc.hh"
|
||||
|
||||
/**
|
||||
* @addtogroup RootMathFuncs RootMathFuncs
|
||||
* @{
|
||||
* @ingroup MathTest
|
||||
*/
|
||||
|
||||
/********** Unary Functions **********/
|
||||
|
||||
MATH_UNARY_KERNEL_DEF(sqrt)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `sqrtf(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::exp(float)`. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_sqrtf_Accuracy_Positive") {
|
||||
float (*ref)(float) = std::sqrt;
|
||||
UnarySinglePrecisionTest(sqrt_kernel<float>, ref, ULPValidatorBuilderFactory<float>(1));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `sqrt(x)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The results are
|
||||
* compared against reference function `double std::sqrt(double)`. The error bounds are
|
||||
* IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_sqrt_Accuracy_Positive") {
|
||||
double (*ref)(double) = std::sqrt;
|
||||
UnaryDoublePrecisionTest<double>(sqrt_kernel<double>, ref, ULPValidatorBuilderFactory<double>(0));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for sqrtf and sqrt.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_sqrt_sqrtf_Negative_RTC") { NegativeTestRTCWrapper<4>(kSqrt); }
|
||||
|
||||
MATH_UNARY_KERNEL_DEF(rsqrt)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `rsqrtf(x)` for all possible inputs. The maximum ulp error
|
||||
* is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rsqrtf_Accuracy_Positive") {
|
||||
auto rsqrt_ref = [](double arg) -> double { return 1. / std::sqrt(arg); };
|
||||
double (*ref)(double) = rsqrt_ref;
|
||||
UnarySinglePrecisionTest(rsqrt_kernel<float>, ref, ULPValidatorBuilderFactory<float>(2));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `rsqrt(x)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rsqrt_Accuracy_Positive") {
|
||||
auto rsqrt_ref = [](long double arg) -> long double { return 1.L / std::sqrt(arg); };
|
||||
long double (*ref)(long double) = rsqrt_ref;
|
||||
UnaryDoublePrecisionTest(rsqrt_kernel<double>, ref, ULPValidatorBuilderFactory<double>(1));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for rsqrtf and rsqrt.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rsqrt_rsqrtf_Negative_RTC") { NegativeTestRTCWrapper<4>(kRsqrt); }
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `cbrtf(x)` for all possible inputs and `cbrt(x)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The results
|
||||
* are compared against reference function `T std::cbrt(T)`. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(cbrt, std::cbrt, 1, 1)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for cbrtf and cbrt.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_cbrt_cbrtf_Negative_RTC") { NegativeTestRTCWrapper<4>(kCbrt); }
|
||||
|
||||
MATH_UNARY_KERNEL_DEF(rcbrt)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `rcbrtf(x)` for all possible inputs. The maximum ulp error
|
||||
* is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rcbrtf_Accuracy_Positive") {
|
||||
auto rcbrt_ref = [](double arg) -> double { return 1. / std::cbrt(arg); };
|
||||
double (*ref)(double) = rcbrt_ref;
|
||||
UnarySinglePrecisionTest(rcbrt_kernel<float>, ref, ULPValidatorBuilderFactory<float>(1));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `rcbrt(x)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The maximum ulp error is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rcbrt_Accuracy_Positive") {
|
||||
auto rcbrt_ref = [](long double arg) -> long double { return 1. / std::cbrt(arg); };
|
||||
long double (*ref)(long double) = rcbrt_ref;
|
||||
UnaryDoublePrecisionTest(rcbrt_kernel<double>, ref, ULPValidatorBuilderFactory<double>(1));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass argument of invalid type for rcbrtf and rcbrt.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rcbrt_rcbrtf_Negative_RTC") { NegativeTestRTCWrapper<4>(kRcbrt); }
|
||||
|
||||
/********** Binary Functions **********/
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `hypotf(x, y)` and `hypot(x, y)` against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The results are
|
||||
* compared against reference function `T std::hypot(T, T)`. The maximum ulp error for single
|
||||
* precision is 3 and for double precision is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(hypot, std::hypot, 3, 2)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for hypotf and hypot.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_hypot_hypotf_Negative_RTC") { NegativeTestRTCWrapper<8>(kHypot); }
|
||||
|
||||
MATH_BINARY_KERNEL_DEF(rhypot)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `rhypotf(x, y)` and `rhypot(x, y)`against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The maximum ulp error
|
||||
* for single precision is 2 and for double precision is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_rhypot_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto rhypot_ref = [](RT arg1, RT arg2) -> RT { return 1. / std::hypot(arg1, arg2); };
|
||||
RT (*ref)(RT, RT) = rhypot_ref;
|
||||
const auto ulp = std::is_same_v<float, TestType> ? 2 : 1;
|
||||
BinaryFloatingPointTest(rhypot_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(ulp));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for rhypotf and rhypot.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rhypot_rhypotf_Negative_RTC") { NegativeTestRTCWrapper<8>(kRhypot); }
|
||||
|
||||
/********** Ternary Functions **********/
|
||||
|
||||
MATH_TERNARY_KERNEL_DEF(norm3d)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `norm3df(x, y, z)` and `norm3d(x, y, z)` against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The maximum ulp error
|
||||
* for single precision is 3 and for double precision is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_norm3d_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto norm3d_ref = [](RT arg1, RT arg2, RT arg3) -> RT {
|
||||
if (std::isinf(arg1) || std::isinf(arg2) || std::isinf(arg3)) {
|
||||
return std::numeric_limits<RT>::infinity();
|
||||
}
|
||||
return std::sqrt(arg1 * arg1 + arg2 * arg2 + arg3 * arg3);
|
||||
};
|
||||
RT (*ref)(RT, RT, RT) = norm3d_ref;
|
||||
const auto ulp = std::is_same_v<float, TestType> ? 3 : 2;
|
||||
TernaryFloatingPointTest(norm3d_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(ulp));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for norm3df and norm3d.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_norm3d_norm3df_Negative_RTC") { NegativeTestRTCWrapper<12>(kNorm3D); }
|
||||
|
||||
MATH_TERNARY_KERNEL_DEF(rnorm3d)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `rnorm3df(x, y, z)` and `rnorm3d(x, y, z)`against a table of
|
||||
* difficult values, followed by a large number of randomly generated values. The maximum ulp error
|
||||
* for single precision is 2 and for double precision is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_rnorm3d_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto rnorm3d_ref = [](RT arg1, RT arg2, RT arg3) -> RT {
|
||||
if (std::isinf(arg1) || std::isinf(arg2) || std::isinf(arg3)) {
|
||||
return 0;
|
||||
}
|
||||
return 1. / std::sqrt(arg1 * arg1 + arg2 * arg2 + arg3 * arg3);
|
||||
};
|
||||
RT (*ref)(RT, RT, RT) = rnorm3d_ref;
|
||||
const auto ulp = std::is_same_v<float, TestType> ? 2 : 1;
|
||||
TernaryFloatingPointTest(rnorm3d_kernel<TestType>, ref,
|
||||
ULPValidatorBuilderFactory<TestType>(ulp));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for rnorm3df and rnorm3d.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rnorm3d_rnorm3df_Negative_RTC") { NegativeTestRTCWrapper<12>(kRnorm3D); }
|
||||
|
||||
/********** Quaternary Functions **********/
|
||||
|
||||
MATH_QUATERNARY_KERNEL_DEF(norm4d)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `norm4df(x, y, z, t)` and `norm4d(x, y, z, t)` against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The maximum
|
||||
* ulp error for single precision is 3 and for double precision is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_norm4d_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto norm4d_ref = [](RT arg1, RT arg2, RT arg3, RT arg4) -> RT {
|
||||
if (std::isinf(arg1) || std::isinf(arg2) || std::isinf(arg3) || std::isinf(arg4)) {
|
||||
return std::numeric_limits<RT>::infinity();
|
||||
}
|
||||
return std::sqrt(arg1 * arg1 + arg2 * arg2 + arg3 * arg3 + arg4 * arg4);
|
||||
};
|
||||
RT (*ref)(RT, RT, RT, RT) = norm4d_ref;
|
||||
const auto ulp = std::is_same_v<float, TestType> ? 3 : 2;
|
||||
QuaternaryFloatingPointTest(norm4d_kernel<TestType>, ref,
|
||||
ULPValidatorBuilderFactory<TestType>(ulp));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for norm4df and norm4d.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_norm4d_norm4df_Negative_RTC") { NegativeTestRTCWrapper<16>(kNorm4D); }
|
||||
|
||||
MATH_QUATERNARY_KERNEL_DEF(rnorm4d)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `rnorm4df(x, y, z, t)` and `rnorm4d(x, y, z, t)`against a
|
||||
* table of difficult values, followed by a large number of randomly generated values. The maximum
|
||||
* ulp error for single precision is 2 and for double precision is 1.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_rnorm4d_Accuracy_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto rnorm4d_ref = [](RT arg1, RT arg2, RT arg3, RT arg4) -> RT {
|
||||
if (std::isinf(arg1) || std::isinf(arg2) || std::isinf(arg3) || std::isinf(arg4)) {
|
||||
return 0;
|
||||
}
|
||||
return 1. / std::sqrt(arg1 * arg1 + arg2 * arg2 + arg3 * arg3 + arg4 * arg4);
|
||||
};
|
||||
RT (*ref)(RT, RT, RT, RT) = rnorm4d_ref;
|
||||
const auto ulp = std::is_same_v<float, TestType> ? 2 : 1;
|
||||
QuaternaryFloatingPointTest(rnorm4d_kernel<TestType>, ref,
|
||||
ULPValidatorBuilderFactory<TestType>(ulp));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for rnorm4df and rnorm4d.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rnorm4d_rnorm4df_Negative_RTC") { NegativeTestRTCWrapper<16>(kRnorm4D); }
|
||||
|
||||
/********** norm Function **********/
|
||||
|
||||
#define MATH_NORM_KERNEL_DEF(func_name) \
|
||||
template <typename T> __global__ void func_name##_kernel(T* const ys, int dim, T* const x1s) { \
|
||||
if constexpr (std::is_same_v<float, T>) { \
|
||||
*ys = func_name##f(dim, x1s); \
|
||||
} else if constexpr (std::is_same_v<double, T>) { \
|
||||
*ys = func_name(dim, x1s); \
|
||||
} \
|
||||
}
|
||||
|
||||
template <typename T, typename F, typename RF, typename ValidatorBuilder>
|
||||
void NormSimpleTest(F kernel, RF ref_func, const ValidatorBuilder& validator_builder) {
|
||||
const auto max_dim = 10000;
|
||||
|
||||
LinearAllocGuard<T> x{LinearAllocs::hipHostMalloc, max_dim * sizeof(T)};
|
||||
LinearAllocGuard<T> x_dev{LinearAllocs::hipMalloc, max_dim * sizeof(T)};
|
||||
LinearAllocGuard<T> y{LinearAllocs::hipHostMalloc, sizeof(T)};
|
||||
LinearAllocGuard<T> y_dev{LinearAllocs::hipMalloc, sizeof(T)};
|
||||
|
||||
std::fill_n(x.ptr(), max_dim, 1);
|
||||
HIP_CHECK(hipMemcpy(x_dev.ptr(), x.ptr(), max_dim * sizeof(T), hipMemcpyHostToDevice));
|
||||
|
||||
for (uint64_t i = 1u; i < max_dim; i++) {
|
||||
kernel<<<1, 1>>>(y_dev.ptr(), i, x_dev.ptr());
|
||||
HIP_CHECK(hipGetLastError());
|
||||
|
||||
HIP_CHECK(hipMemcpy(y.ptr(), y_dev.ptr(), sizeof(T), hipMemcpyDeviceToHost));
|
||||
const auto actual_val = *y.ptr();
|
||||
const auto ref_val = static_cast<T>(ref_func(i, x.ptr()));
|
||||
const auto validator = validator_builder(ref_val);
|
||||
|
||||
if (!validator->match(actual_val)) {
|
||||
std::stringstream ss;
|
||||
ss << std::scientific << std::setprecision(std::numeric_limits<T>::max_digits10 - 1);
|
||||
ss << "Validation fails for dim: " << i << " " << actual_val << " " << ref_val;
|
||||
INFO(ss.str());
|
||||
REQUIRE(false);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MATH_NORM_KERNEL_DEF(norm)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `normf(dim, arr)` and `norm(dim, arr)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_norm_Sanity_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto norm_ref = [](int dim, TestType* args) -> RT {
|
||||
RT sum = 0;
|
||||
for (int i = 0; i < dim; i++) {
|
||||
if (std::isinf(args[i])) return std::numeric_limits<RT>::infinity();
|
||||
sum += static_cast<RT>(args[i]) * static_cast<RT>(args[i]);
|
||||
}
|
||||
return std::sqrt(sum);
|
||||
};
|
||||
RT (*ref)(int, TestType*) = norm_ref;
|
||||
|
||||
NormSimpleTest<TestType>(norm_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(10));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for normf and norm.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_norm_normf_Negative_RTC") { NegativeTestRTCWrapper<18>(kNorm); }
|
||||
|
||||
MATH_NORM_KERNEL_DEF(rnorm)
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Sanity test for `rnormf(dim, arr)` and `rnorm(dim, arr)`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEMPLATE_TEST_CASE("Unit_Device_rnorm_Sanity_Positive", "", float, double) {
|
||||
using RT = RefType_t<TestType>;
|
||||
auto rnorm_ref = [](int dim, TestType* args) -> RT {
|
||||
RT sum = 0;
|
||||
for (int i = 0; i < dim; i++) {
|
||||
if (std::isinf(args[i])) return std::numeric_limits<RT>::infinity();
|
||||
sum += static_cast<RT>(args[i]) * static_cast<RT>(args[i]);
|
||||
}
|
||||
return 1. / std::sqrt(sum);
|
||||
};
|
||||
RT (*ref)(int, TestType*) = rnorm_ref;
|
||||
|
||||
NormSimpleTest<TestType>(rnorm_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(10));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - RTCs kernels that pass combinations of arguments of invalid types for rnormf and rnorm.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/root_funcs.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_Device_rnorm_rnormf_Negative_RTC") { NegativeTestRTCWrapper<18>(kRnorm); }
|
||||
@@ -0,0 +1,530 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "binary_common.hh"
|
||||
#include "ternary_common.hh"
|
||||
|
||||
/********** Unary Functions **********/
|
||||
|
||||
#define MATH_UNARY_SP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(float* const ys, const size_t num_xs, float* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(xs[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_UNARY_SP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
UnarySinglePrecisionTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_UNARY_SP_TEST_DEF(func_name, ref_func) \
|
||||
MATH_UNARY_SP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x)
|
||||
|
||||
|
||||
static float __frcp_rn_ref(float x) { return 1.0f / x; }
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__frcp_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__frcp_rn(x)` for all possible inputs. The error bounds are
|
||||
* IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF_IMPL(__frcp_rn, __frcp_rn_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__fsqrt_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__fsqrt_rn(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::sqrt(float)`. The error bounds are
|
||||
* IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF_IMPL(__fsqrt_rn, static_cast<float (*)(float)>(std::sqrt),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __frsqrt_rn_ref(float x) { return 1.0f / std::sqrt(x); }
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__frsqrt_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__frsqrt_rn(x)` for all possible inputs. The results are
|
||||
* compared against reference function `float std::sqrt(float)`. The error bounds are
|
||||
* IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF_IMPL(__frsqrt_rn, __frsqrt_rn_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__expf) {
|
||||
const int64_t ulp_err = 2 + static_cast<int64_t>(std::floor(std::abs(1.16f * x)));
|
||||
return ULPValidatorBuilderFactory<float>(ulp_err)(target);
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__expf);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__expf(x)` for all possible inputs. The results are
|
||||
* compared against reference function `double std::exp(double)`. The maximum ulp error is `2 +
|
||||
* floor(abs(1.16 * x))`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF(__expf, static_cast<double (*)(double)>(std::exp));
|
||||
|
||||
|
||||
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__exp10f) {
|
||||
const int64_t ulp_err = 2 + static_cast<int64_t>(std::floor(std::abs(2.95f * x)));
|
||||
return ULPValidatorBuilderFactory<float>(ulp_err)(target);
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__exp10f);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__exp10f(x)` for all possible inputs. The results are
|
||||
* compared against reference function `double exp10(double)`. The maximum ulp error is `2 +
|
||||
* floor(abs(2.95 * x))`.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF(__exp10f, static_cast<double (*)(double)>(exp10));
|
||||
|
||||
|
||||
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__logf) {
|
||||
if (0.5f <= x && x <= 2.0f) {
|
||||
const auto abs_err = std::pow(2.0, -21.41);
|
||||
return AbsValidatorBuilderFactory<float>(abs_err)(target);
|
||||
} else {
|
||||
return ULPValidatorBuilderFactory<float>(3)(target);
|
||||
}
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__logf);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__logf(x)` for all possible inputs. The results are
|
||||
* compared against reference function `double std::log(double)`. For `x` in [0.5, 2], the maximum
|
||||
* absolute error is 2^-21.41, otherwise, the maximum ulp error is 3.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF(__logf, static_cast<double (*)(double)>(std::log));
|
||||
|
||||
|
||||
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__log2f) {
|
||||
if (0.5f <= x && x <= 2.0f) {
|
||||
const auto abs_err = std::pow(2.0, -22.0);
|
||||
return AbsValidatorBuilderFactory<float>(abs_err)(target);
|
||||
} else {
|
||||
return ULPValidatorBuilderFactory<float>(2)(target);
|
||||
}
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__log2f);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__log2f(x)` for all possible inputs. The results are
|
||||
* compared against reference function `double std::log2(double)`. For `x` in [0.5, 2], the maximum
|
||||
* absolute error is 2^-22, otherwise, the maximum ulp error is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF(__log2f, static_cast<double (*)(double)>(std::log2));
|
||||
|
||||
|
||||
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__log10f) {
|
||||
if (0.5f <= x && x <= 2.0f) {
|
||||
const auto abs_err = std::pow(2.0, -24.0);
|
||||
return AbsValidatorBuilderFactory<float>(abs_err)(target);
|
||||
} else {
|
||||
return ULPValidatorBuilderFactory<float>(3)(target);
|
||||
}
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__log10f);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__log10f(x)` for all possible inputs. The results are
|
||||
* compared against reference function `double std::log10(double)`. For `x` in [0.5, 2], the maximum
|
||||
* absolute error is 2^-24, otherwise, the maximum ulp error is 3.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF(__log10f, static_cast<double (*)(double)>(std::log10));
|
||||
|
||||
|
||||
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__sinf) {
|
||||
if (-M_PI <= x && x <= M_PI) {
|
||||
const auto abs_err = std::pow(2.0, -21.41);
|
||||
return AbsValidatorBuilderFactory<float>(abs_err)(target);
|
||||
} else {
|
||||
return NopValidatorBuilderFactory<float>()();
|
||||
}
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__sinf);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__sinf(x)` for all possible inputs. The results are
|
||||
* compared against reference function `double std::sin(double)`. For `x` in [-PI, PI], the maximum
|
||||
* absolute error is 2^-21.41, and larger otherwise.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF(__sinf, static_cast<double (*)(double)>(std::sin));
|
||||
|
||||
|
||||
__device__ float __sincosf_sin(float x) {
|
||||
float sin, cos;
|
||||
__sincosf(x, &sin, &cos);
|
||||
return sin;
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__sincosf_sin);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__sincosf(x, sptr, cptr)` for all possible inputs. The
|
||||
* results in `sptr` are compared against reference function `double std::sin(double)`. For `x` in
|
||||
* [-PI, PI], the maximum absolute error is 2^-21.41, and larger otherwise.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF_IMPL(__sincosf_sin, static_cast<double (*)(double)>(std::sin),
|
||||
__sinf_validator_builder);
|
||||
|
||||
|
||||
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__cosf) {
|
||||
if (-M_PI <= x && x <= M_PI) {
|
||||
const auto abs_err = std::pow(2.0, -21.19);
|
||||
return AbsValidatorBuilderFactory<float>(abs_err)(target);
|
||||
} else {
|
||||
return NopValidatorBuilderFactory<float>()();
|
||||
}
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__cosf);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__cosf(x)` for all possible inputs. The results are
|
||||
* compared against reference function `double std::cos(double)`. For `x` in [-PI, PI], the maximum
|
||||
* absolute error is 2^-21.19, and larger otherwise.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF(__cosf, static_cast<double (*)(double)>(std::cos));
|
||||
|
||||
|
||||
__device__ float __sincosf_cos(float x) {
|
||||
float sin, cos;
|
||||
__sincosf(x, &sin, &cos);
|
||||
return cos;
|
||||
}
|
||||
|
||||
MATH_UNARY_SP_KERNEL_DEF(__sincosf_cos);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__sincosf(x, sptr, cptr)` for all possible inputs. The
|
||||
* results in `cptr` are compared against reference function `double std::cos(double)`. For `x` in
|
||||
* [-PI, PI], the maximum absolute error is 2^-21.19, and larger otherwise.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_UNARY_SP_TEST_DEF_IMPL(__sincosf_cos, static_cast<double (*)(double)>(std::cos),
|
||||
__cosf_validator_builder);
|
||||
|
||||
|
||||
/********** Binary Functions **********/
|
||||
|
||||
#define MATH_BINARY_SP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(float* const ys, const size_t num_xs, float* const x1s, \
|
||||
float* const x2s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_BINARY_SP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
BinaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_BINARY_SP_TEST_DEF(func_name, ref_func) \
|
||||
MATH_BINARY_SP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_BINARY_SP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x1, \
|
||||
float x2)
|
||||
|
||||
static float __fadd_rn_ref(float x1, float x2) { return x1 + x2; }
|
||||
|
||||
MATH_BINARY_SP_KERNEL_DEF(__fadd_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__fadd_rn(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_SP_TEST_DEF_IMPL(__fadd_rn, __fadd_rn_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __fsub_rn_ref(float x1, float x2) { return x1 - x2; }
|
||||
|
||||
MATH_BINARY_SP_KERNEL_DEF(__fsub_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__fsub_rn(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_SP_TEST_DEF_IMPL(__fsub_rn, __fsub_rn_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __fmul_rn_ref(float x1, float x2) { return x1 * x2; }
|
||||
|
||||
MATH_BINARY_SP_KERNEL_DEF(__fmul_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__fmul_rn(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_SP_TEST_DEF_IMPL(__fmul_rn, __fmul_rn_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
static float __fdiv_rn_ref(float x1, float x2) { return x1 / x2; }
|
||||
|
||||
MATH_BINARY_SP_KERNEL_DEF(__fdiv_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__fdiv_rn(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_SP_TEST_DEF_IMPL(__fdiv_rn, __fdiv_rn_ref, EqValidatorBuilderFactory<float>());
|
||||
|
||||
|
||||
MATH_BINARY_SP_VALIDATOR_BUILDER_DEF(__fdividef) {
|
||||
x1 = 2.0f;
|
||||
const auto abs_x2 = std::abs(x2);
|
||||
if (std::pow(x1, -126.0f) <= abs_x2 && abs_x2 <= std::pow(x1, 126.0f)) {
|
||||
return ULPValidatorBuilderFactory<float>(2)(target);
|
||||
} else {
|
||||
return NopValidatorBuilderFactory<float>()();
|
||||
}
|
||||
}
|
||||
|
||||
MATH_BINARY_SP_KERNEL_DEF(__fdividef);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__fdividef(x,y)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. For `|y|` in [2^-126, 2^126], the
|
||||
* maximum ulp error is 2.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_BINARY_SP_TEST_DEF(__fdividef, __fdiv_rn_ref);
|
||||
|
||||
|
||||
/********** Ternary Functions **********/
|
||||
|
||||
#define MATH_TERNARY_SP_KERNEL_DEF(func_name) \
|
||||
__global__ void func_name##_kernel(float* const ys, const size_t num_xs, float* const x1s, \
|
||||
float* const x2s, float* const x3s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i], x3s[i]); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define MATH_TERNARY_SP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
|
||||
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
|
||||
TernaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
|
||||
}
|
||||
|
||||
#define MATH_TERNARY_SP_TEST_DEF(func_name, ref_func, validator_builder) \
|
||||
MATH_TERNARY_SP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
|
||||
|
||||
#define MATH_TERNARY_SP_VALIDATOR_BUILDER_DEF(func_name) \
|
||||
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x1, \
|
||||
float x2, float x3)
|
||||
|
||||
|
||||
MATH_TERNARY_SP_KERNEL_DEF(__fmaf_rn);
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Tests the numerical accuracy of `__fmaf(x,y,z)` against a table of difficult values,
|
||||
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
|
||||
*
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - unit/math/single_precision_intrinsics.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
MATH_TERNARY_SP_TEST_DEF_IMPL(__fmaf_rn, static_cast<float (*)(float, float, float)>(std::fma),
|
||||
EqValidatorBuilderFactory<float>());
|
||||
@@ -0,0 +1,56 @@
|
||||
/*
|
||||
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(float* x) { float result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { float result = func_name(x); }
|
||||
|
||||
#define INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(float* x, float y) { float result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v2(float x, float* y) { float result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v3(Dummy x, float y) { float result = func_name(x, y); } \
|
||||
__global__ void func_name##_kernel_v4(float x, Dummy y) { float result = func_name(x, y); }
|
||||
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__fsqrt_rn)
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__expf)
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__exp10f)
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__logf)
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__log2f)
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__log10f)
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__sinf)
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__cosf)
|
||||
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__tanf)
|
||||
|
||||
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fadd_rn)
|
||||
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fsub_rn)
|
||||
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fmul_rn)
|
||||
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fdiv_rn)
|
||||
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fdividef)
|
||||
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__powf)
|
||||
@@ -0,0 +1,145 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "math_common.hh"
|
||||
#include "math_special_values.hh"
|
||||
|
||||
#include <hip/hip_cooperative_groups.h>
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
#define MATH_BESSEL_N_KERNEL_DEF(func_name) \
|
||||
template <typename T> \
|
||||
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, int* n, T* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
if constexpr (std::is_same_v<float, T>) { \
|
||||
ys[i] = func_name##f(n[i], xs[i]); \
|
||||
} else if constexpr (std::is_same_v<double, T>) { \
|
||||
ys[i] = func_name(n[i], xs[i]); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
template <typename T> using kernel_bessel_n_sig = void (*)(T*, const size_t, int*, T*);
|
||||
|
||||
template <typename T> using ref_bessel_n_sig = T (*)(int, T);
|
||||
|
||||
template <typename ValidatorBuilder>
|
||||
void BesselDoublePrecisionBruteForceTest(kernel_bessel_n_sig<double> kernel,
|
||||
ref_bessel_n_sig<long double> ref_func,
|
||||
const ValidatorBuilder& validator_builder, int n_input = 0,
|
||||
const double a = std::numeric_limits<double>::lowest(),
|
||||
const double b = std::numeric_limits<double>::max()) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const uint64_t num_iterations = GetTestIterationCount();
|
||||
const auto max_batch_size = std::min(
|
||||
GetMaxAllowedDeviceMemoryUsage() / (sizeof(double) * 2 + sizeof(int)), num_iterations);
|
||||
LinearAllocGuard<int> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(int)};
|
||||
LinearAllocGuard<double> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(double)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
std::fill_n(x1s.ptr(), max_batch_size, n_input);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
thread_pool.Post([=, &x2s] {
|
||||
const auto generator = [=] {
|
||||
static thread_local std::mt19937 rng(std::random_device{}());
|
||||
std::uniform_real_distribution<RefType_t<double>> unif_dist(a, b);
|
||||
return static_cast<double>(unif_dist(rng));
|
||||
};
|
||||
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator);
|
||||
});
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
|
||||
x2s.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename ValidatorBuilder>
|
||||
void BesselSinglePrecisionRangeTest(kernel_bessel_n_sig<float> kernel,
|
||||
ref_bessel_n_sig<double> ref_func,
|
||||
const ValidatorBuilder& validator_builder, int n_input,
|
||||
const float a, const float b) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const auto max_batch_size = GetMaxAllowedDeviceMemoryUsage() / (sizeof(float) * 2 + sizeof(int));
|
||||
LinearAllocGuard<int> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(int)};
|
||||
LinearAllocGuard<float> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(float)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
std::fill_n(x1s.ptr(), max_batch_size, n_input);
|
||||
|
||||
size_t inserted = 0u;
|
||||
for (float v = a; v != b; v = std::nextafter(v, b)) {
|
||||
x2s.ptr()[inserted++] = v;
|
||||
if (inserted < max_batch_size) continue;
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, inserted, x1s.ptr(),
|
||||
x2s.ptr());
|
||||
inserted = 0u;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename F, typename ValidatorBuilder>
|
||||
void SpecialSimpleTest(F kernel, const ValidatorBuilder& validator_builder, const T* x,
|
||||
const T* ref, size_t num_args) {
|
||||
LinearAllocGuard<T> x_dev{LinearAllocs::hipMalloc, num_args * sizeof(T)};
|
||||
LinearAllocGuard<T> y{LinearAllocs::hipHostMalloc, num_args * sizeof(T)};
|
||||
LinearAllocGuard<T> y_dev{LinearAllocs::hipMalloc, num_args * sizeof(T)};
|
||||
|
||||
HIP_CHECK(hipMemcpy(x_dev.ptr(), x, num_args * sizeof(T), hipMemcpyHostToDevice));
|
||||
|
||||
kernel<<<1, num_args>>>(y_dev.ptr(), num_args, x_dev.ptr());
|
||||
HIP_CHECK(hipGetLastError());
|
||||
|
||||
HIP_CHECK(hipMemcpy(y.ptr(), y_dev.ptr(), num_args * sizeof(T), hipMemcpyDeviceToHost));
|
||||
|
||||
for (auto i = 0u; i < num_args; ++i) {
|
||||
const auto actual_val = y.ptr()[i];
|
||||
const auto ref_val = ref[i];
|
||||
const auto validator = validator_builder(ref_val);
|
||||
|
||||
if (!validator->match(actual_val)) {
|
||||
std::stringstream ss;
|
||||
ss << "Input value(s): " << std::scientific
|
||||
<< std::setprecision(std::numeric_limits<T>::max_digits10 - 1);
|
||||
ss << x[i] << " " << actual_val << " " << ref_val << "\n";
|
||||
INFO(ss.str());
|
||||
REQUIRE(false);
|
||||
}
|
||||
}
|
||||
}
|
||||
Plik diff jest za duży
Load Diff
@@ -0,0 +1,151 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "math_common.hh"
|
||||
#include "math_special_values.hh"
|
||||
|
||||
#include <hip/hip_cooperative_groups.h>
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
#define MATH_TERNARY_KERNEL_DEF(func_name) \
|
||||
template <typename T> \
|
||||
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, T* const x1s, T* const x2s, \
|
||||
T* const x3s) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
if constexpr (std::is_same_v<float, T>) { \
|
||||
ys[i] = func_name##f(x1s[i], x2s[i], x3s[i]); \
|
||||
} else if constexpr (std::is_same_v<double, T>) { \
|
||||
ys[i] = func_name(x1s[i], x2s[i], x3s[i]); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void TernaryFloatingPointBruteForceTest(kernel_sig<T, TArg, TArg, TArg> kernel,
|
||||
ref_sig<RT, RTArg, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder,
|
||||
const TArg a = std::numeric_limits<TArg>::lowest(),
|
||||
const TArg b = std::numeric_limits<TArg>::max()) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const uint64_t num_iterations = GetTestIterationCount();
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(TArg) * 3 + sizeof(T)), num_iterations);
|
||||
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x3s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
thread_pool.Post([=, &x1s, &x2s, &x3s] {
|
||||
const auto generator = [=] {
|
||||
static thread_local std::mt19937 rng(std::random_device{}());
|
||||
if constexpr (std::is_same_v<TArg, Float16>) {
|
||||
std::uniform_real_distribution<RefType_t<Float16>> unif_dist(-FLOAT16_MAX, FLOAT16_MAX);
|
||||
return static_cast<Float16>(unif_dist(rng));
|
||||
} else {
|
||||
std::uniform_real_distribution<RefType_t<TArg>> unif_dist(a, b);
|
||||
return static_cast<TArg>(unif_dist(rng));
|
||||
}
|
||||
};
|
||||
std::generate(x1s.ptr() + base_idx, x1s.ptr() + base_idx + sub_batch_size, generator);
|
||||
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator);
|
||||
std::generate(x3s.ptr() + base_idx, x3s.ptr() + base_idx + sub_batch_size, generator);
|
||||
});
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
|
||||
x2s.ptr(), x3s.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void TernaryFloatingPointSpecialValuesTest(kernel_sig<T, TArg, TArg, TArg> kernel,
|
||||
ref_sig<RT, RTArg, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
using SpecialValsType = std::conditional_t<std::is_same_v<TArg, Float16>, float, TArg>;
|
||||
const auto values = std::get<SpecialVals<SpecialValsType>>(kSpecialValRegistry);
|
||||
|
||||
const auto size = values.size * values.size * values.size;
|
||||
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
LinearAllocGuard<TArg> x3s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
|
||||
|
||||
for (auto i = 0u; i < values.size; ++i) {
|
||||
for (auto j = 0u; j < values.size; ++j) {
|
||||
for (auto k = 0u; k < values.size; ++k) {
|
||||
x1s.ptr()[(i * values.size + j) * values.size + k] = values.data[i];
|
||||
x2s.ptr()[(i * values.size + j) * values.size + k] = values.data[j];
|
||||
x3s.ptr()[(i * values.size + j) * values.size + k] = values.data[k];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MathTest math_test(kernel, size);
|
||||
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, size, x1s.ptr(),
|
||||
x2s.ptr(), x3s.ptr());
|
||||
}
|
||||
|
||||
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void TernaryFloatingPointTest(kernel_sig<T, TArg, TArg, TArg> kernel,
|
||||
ref_sig<RT, RTArg, RTArg, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Special values") {
|
||||
TernaryFloatingPointSpecialValuesTest(kernel, ref_func, validator_builder);
|
||||
}
|
||||
|
||||
SECTION("Brute force") {
|
||||
TernaryFloatingPointBruteForceTest(kernel, ref_func, validator_builder);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#define MATH_TERNARY_WITHIN_ULP_TEST_DEF(kern_name, ref_func, sp_ulp, dp_ulp) \
|
||||
MATH_TERNARY_KERNEL_DEF(kern_name) \
|
||||
\
|
||||
TEMPLATE_TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive", "", float, double) { \
|
||||
using RT = RefType_t<TestType>; \
|
||||
RT (*ref)(RT, RT, RT) = ref_func; \
|
||||
const auto ulp = std::is_same_v<float, TestType> ? sp_ulp : dp_ulp; \
|
||||
\
|
||||
TernaryFloatingPointTest(kern_name##_kernel<TestType>, ref, \
|
||||
ULPValidatorBuilderFactory<TestType>(ulp)); \
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <thread>
|
||||
|
||||
#include <boost/asio/post.hpp>
|
||||
#include <boost/asio/thread_pool.hpp>
|
||||
|
||||
// This is a simple wrapper around boost::asio::thread_pool that keeps track of the number of
|
||||
// currently active tasks using an atomic counter.
|
||||
class ThreadPool {
|
||||
public:
|
||||
ThreadPool(size_t thread_count = std::thread::hardware_concurrency())
|
||||
: thread_count_(thread_count) {}
|
||||
|
||||
~ThreadPool() { thread_pool_.join(); }
|
||||
|
||||
// Submits a task to the thread pool and increments the number of active tasks. The task is
|
||||
// wrapped in a lambda that decrements the number of active tasks upon completion.
|
||||
template <typename T> void Post(T&& task) {
|
||||
++active_tasks_;
|
||||
auto&& task_wrapper = [task, this] {
|
||||
task();
|
||||
--active_tasks_;
|
||||
};
|
||||
boost::asio::post(thread_pool_, task_wrapper);
|
||||
}
|
||||
|
||||
// Busy waits for the number of active tasks to reach zero.
|
||||
void Wait() const {
|
||||
while (active_tasks_.load(std::memory_order_relaxed))
|
||||
;
|
||||
}
|
||||
|
||||
size_t thread_count() const { return thread_count_; }
|
||||
|
||||
private:
|
||||
const size_t thread_count_;
|
||||
boost::asio::thread_pool thread_pool_{thread_count_};
|
||||
std::atomic<size_t> active_tasks_;
|
||||
};
|
||||
|
||||
inline ThreadPool thread_pool{};
|
||||
@@ -0,0 +1,108 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define TRIG_DP_UNARY_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
|
||||
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); }
|
||||
|
||||
/*Expecting 2 errors per macro invocation - 26 total*/
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(sin)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(cos)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(tan)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(asin)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(acos)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(atan)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(sinh)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(cosh)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(tanh)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(asinh)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(atanh)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(sinpi)
|
||||
TRIG_DP_UNARY_NEGATIVE_KERNELS(cospi)
|
||||
|
||||
/*Expecting 4 errors*/
|
||||
__global__ void atan2_kernel_v1(double* x, double y) { double result = atan2(x, y); }
|
||||
__global__ void atan2_kernel_v2(double x, double* y) { double result = atan2(x, y); }
|
||||
__global__ void atan2_kernel_v3(Dummy x, double y) { double result = atan2(x, y); }
|
||||
__global__ void atan2_kernel_v4(double x, Dummy y) { double result = atan2(x, y); }
|
||||
|
||||
/*Expecting 18 errors*/
|
||||
__global__ void sincos_kernel_v1(double* x, double* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v2(Dummy x, double* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v3(double x, char* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v4(double x, short* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v5(double x, int* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v6(double x, long* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v7(double x, long long* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v8(double x, float* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v9(double x, Dummy* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v10(double x, const double* sptr, double* cptr) {
|
||||
sincos(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincos_kernel_v11(double x, double* sptr, char* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v12(double x, double* sptr, short* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v13(double x, double* sptr, int* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v14(double x, double* sptr, long* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v15(double x, double* sptr, long long* cptr) {
|
||||
sincos(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincos_kernel_v16(double x, double* sptr, float* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v17(double x, double* sptr, Dummy* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v18(double x, double* sptr, const double* cptr) {
|
||||
sincos(x, sptr, cptr);
|
||||
}
|
||||
|
||||
/*Expecting 18 errors*/
|
||||
__global__ void sincospi_kernel_v1(float* x, float* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v2(Dummy x, float* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v3(float x, char* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v4(float x, short* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v5(float x, int* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v6(float x, long* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v7(float x, long long* sptr, float* cptr) {
|
||||
sincospi(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospi_kernel_v8(float x, double* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v9(float x, Dummy* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v10(float x, const float* sptr, float* cptr) {
|
||||
sincospi(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospi_kernel_v11(float x, float* sptr, char* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v12(float x, float* sptr, short* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v13(float x, float* sptr, int* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v14(float x, float* sptr, long* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v15(float x, float* sptr, long long* cptr) {
|
||||
sincospi(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospi_kernel_v16(float x, float* sptr, double* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v17(float x, float* sptr, Dummy* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v18(float x, float* sptr, const float* cptr) {
|
||||
sincospi(x, sptr, cptr);
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "trig_negative_kernels_rtc.hh"
|
||||
|
||||
#include "unary_common.hh"
|
||||
#include "binary_common.hh"
|
||||
|
||||
#include <boost/math/special_functions.hpp>
|
||||
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(sin, std::sin, 2, 2);
|
||||
TEST_CASE("Unit_Device_sin_sinf_Negative_RTC") { NegativeTestRTCWrapper<4>(kSin); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(cos, std::cos, 2, 2)
|
||||
TEST_CASE("Unit_Device_cos_cosf_Negative_RTC") { NegativeTestRTCWrapper<4>(kCos); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(tan, std::tan, 4, 2)
|
||||
TEST_CASE("Unit_Device_tan_tanf_Negative_RTC") { NegativeTestRTCWrapper<4>(kTan); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(asin, std::asin, 2, 2)
|
||||
TEST_CASE("Unit_Device_asin_asinf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAsin); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(acos, std::acos, 2, 2)
|
||||
TEST_CASE("Unit_Device_acos_acosf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAcos); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(atan, std::atan, 2, 2)
|
||||
TEST_CASE("Unit_Device_atan_atanf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAtan); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(sinh, std::sinh, 3, 2)
|
||||
TEST_CASE("Unit_Device_sinh_sinhf_Negative_RTC") { NegativeTestRTCWrapper<4>(kSinh); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(cosh, std::cosh, 2, 1)
|
||||
TEST_CASE("Unit_Device_cosh_coshf_Negative_RTC") { NegativeTestRTCWrapper<4>(kCosh); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(tanh, std::tanh, 2, 1)
|
||||
TEST_CASE("Unit_Device_tanh_tanhf_Negative_RTC") { NegativeTestRTCWrapper<4>(kTanh); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(asinh, std::asinh, 3, 2)
|
||||
TEST_CASE("Unit_Device_asinh_asinhf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAsinh); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(acosh, std::acosh, 4, 2)
|
||||
TEST_CASE("Unit_Device_acosh_acoshf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAcosh); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(atanh, std::atanh, 3, 2)
|
||||
TEST_CASE("Unit_Device_atanh_atanhf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAtanh); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(sinpi, boost::math::sin_pi, 2, 2);
|
||||
TEST_CASE("Unit_Device_sinpi_sinpif_Negative_RTC") { NegativeTestRTCWrapper<4>(kSinpi); }
|
||||
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(cospi, boost::math::cos_pi, 2, 2);
|
||||
TEST_CASE("Unit_Device_cospi_cospif_Negative_RTC") { NegativeTestRTCWrapper<4>(kCospi); }
|
||||
|
||||
MATH_BINARY_WITHIN_ULP_TEST_DEF(atan2, std::atan2, 3, 2);
|
||||
TEST_CASE("Unit_Device_atan2_atan2f_Negative_RTC") { NegativeTestRTCWrapper<8>(kAtan2); }
|
||||
|
||||
|
||||
template <typename T>
|
||||
__global__ void sincos_kernel(std::pair<T, T>* const ys, const size_t num_xs, T* const xs) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
if constexpr (std::is_same_v<float, T>) {
|
||||
sincosf(xs[i], &ys[i].first, &ys[i].second);
|
||||
} else if constexpr (std::is_same_v<double, T>) {
|
||||
sincos(xs[i], &ys[i].first, &ys[i].second);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> std::pair<T, T> sincos(T x) { return {std::sin(x), std::cos(x)}; }
|
||||
|
||||
TEST_CASE("Unit_Device_sincos_Accuracy_Positive - float") {
|
||||
UnarySinglePrecisionTest(
|
||||
sincos_kernel<float>, sincos<double>,
|
||||
PairValidatorBuilderFactory<float>(ULPValidatorBuilderFactory<float>(2)));
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_sincos_Accuracy_Positive - double") {
|
||||
const auto validator_builder =
|
||||
PairValidatorBuilderFactory<double>(ULPValidatorBuilderFactory<double>(2));
|
||||
UnaryDoublePrecisionTest(sincos_kernel<double>, sincos<long double>, validator_builder);
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_sincos_sincosf_Negative_RTC") { NegativeTestRTCWrapper<36>(kSincos); }
|
||||
|
||||
|
||||
template <typename T>
|
||||
__global__ void sincospi_kernel(std::pair<T, T>* const ys, const size_t num_xs, T* const xs) {
|
||||
const auto tid = cg::this_grid().thread_rank();
|
||||
const auto stride = cg::this_grid().size();
|
||||
|
||||
for (auto i = tid; i < num_xs; i += stride) {
|
||||
if constexpr (std::is_same_v<float, T>) {
|
||||
sincospif(xs[i], &ys[i].first, &ys[i].second);
|
||||
} else if constexpr (std::is_same_v<double, T>) {
|
||||
sincospi(xs[i], &ys[i].first, &ys[i].second);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T> std::pair<T, T> sincospi(T x) {
|
||||
return {boost::math::sin_pi(x), boost::math::cos_pi(x)};
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_sincospi_Accuracy_Positive - float") {
|
||||
UnarySinglePrecisionTest(
|
||||
sincospi_kernel<float>, sincospi<double>,
|
||||
PairValidatorBuilderFactory<float>(ULPValidatorBuilderFactory<float>(2)));
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_sincospi_Accuracy_Positive - double") {
|
||||
const auto validator_builder =
|
||||
PairValidatorBuilderFactory<double>(ULPValidatorBuilderFactory<double>(2));
|
||||
UnaryDoublePrecisionTest(sincospi_kernel<double>, sincospi<long double>, validator_builder);
|
||||
}
|
||||
|
||||
TEST_CASE("Unit_Device_sincospi_sincospif_Negative_RTC") { NegativeTestRTCWrapper<36>(kSincospi); }
|
||||
@@ -0,0 +1,320 @@
|
||||
// #define TRIG_UNARY_NEGATIVE_KERNELS(func_name)
|
||||
// class Dummy {
|
||||
// public:
|
||||
// __device__ Dummy() {}
|
||||
// __device__ ~Dummy() {}
|
||||
// };
|
||||
// __global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); }
|
||||
// __global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
|
||||
// __global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); }
|
||||
// __global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); }
|
||||
|
||||
static constexpr auto kSin{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void sinf_kernel_v1(float* x) { float result = sinf(x); }
|
||||
__global__ void sinf_kernel_v2(Dummy x) { float result = sinf(x); }
|
||||
__global__ void sin_kernel_v1(double* x) { double result = sin(x); }
|
||||
__global__ void sin_kernel_v2(Dummy x) { double result = sin(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kCos{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void cosf_kernel_v1(float* x) { float result = cosf(x); }
|
||||
__global__ void cosf_kernel_v2(Dummy x) { float result = cosf(x); }
|
||||
__global__ void cos_kernel_v1(double* x) { double result = cos(x); }
|
||||
__global__ void cos_kernel_v2(Dummy x) { double result = cos(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kTan{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void tanf_kernel_v1(float* x) { float result = tanf(x); }
|
||||
__global__ void tanf_kernel_v2(Dummy x) { float result = tanf(x); }
|
||||
__global__ void tan_kernel_v1(double* x) { double result = tan(x); }
|
||||
__global__ void tan_kernel_v2(Dummy x) { double result = tan(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kAsin{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void asinf_kernel_v1(float* x) { float result = asinf(x); }
|
||||
__global__ void asinf_kernel_v2(Dummy x) { float result = asinf(x); }
|
||||
__global__ void asin_kernel_v1(double* x) { double result = asin(x); }
|
||||
__global__ void asin_kernel_v2(Dummy x) { double result = asin(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kAcos{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void acosf_kernel_v1(float* x) { float result = acosf(x); }
|
||||
__global__ void acosf_kernel_v2(Dummy x) { float result = acosf(x); }
|
||||
__global__ void acos_kernel_v1(double* x) { double result = acos(x); }
|
||||
__global__ void acos_kernel_v2(Dummy x) { double result = acos(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kAtan{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void atanf_kernel_v1(float* x) { float result = atanf(x); }
|
||||
__global__ void atanf_kernel_v2(Dummy x) { float result = atanf(x); }
|
||||
__global__ void atan_kernel_v1(double* x) { double result = atan(x); }
|
||||
__global__ void atan_kernel_v2(Dummy x) { double result = atan(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kSinh{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void sinhf_kernel_v1(float* x) { float result = sinhf(x); }
|
||||
__global__ void sinhf_kernel_v2(Dummy x) { float result = sinhf(x); }
|
||||
__global__ void sinh_kernel_v1(double* x) { double result = sinh(x); }
|
||||
__global__ void sinh_kernel_v2(Dummy x) { double result = sinh(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kCosh{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void coshf_kernel_v1(float* x) { float result = coshf(x); }
|
||||
__global__ void coshf_kernel_v2(Dummy x) { float result = coshf(x); }
|
||||
__global__ void cosh_kernel_v1(double* x) { double result = cosh(x); }
|
||||
__global__ void cosh_kernel_v2(Dummy x) { double result = cosh(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kTanh{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void tanhf_kernel_v1(float* x) { float result = tanhf(x); }
|
||||
__global__ void tanhf_kernel_v2(Dummy x) { float result = tanhf(x); }
|
||||
__global__ void tanh_kernel_v1(double* x) { double result = tanh(x); }
|
||||
__global__ void tanh_kernel_v2(Dummy x) { double result = tanh(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kAsinh{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void asinhf_kernel_v1(float* x) { float result = asinhf(x); }
|
||||
__global__ void asinhf_kernel_v2(Dummy x) { float result = asinhf(x); }
|
||||
__global__ void asinh_kernel_v1(double* x) { double result = asinh(x); }
|
||||
__global__ void asinh_kernel_v2(Dummy x) { double result = asinh(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kAcosh{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void acoshf_kernel_v1(float* x) { float result = acoshf(x); }
|
||||
__global__ void acoshf_kernel_v2(Dummy x) { float result = acoshf(x); }
|
||||
__global__ void acosh_kernel_v1(double* x) { double result = acosh(x); }
|
||||
__global__ void acosh_kernel_v2(Dummy x) { double result = acosh(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kAtanh{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void atanhf_kernel_v1(float* x) { float result = atanhf(x); }
|
||||
__global__ void atanhf_kernel_v2(Dummy x) { float result = atanhf(x); }
|
||||
__global__ void atanh_kernel_v1(double* x) { double result = atanh(x); }
|
||||
__global__ void atanh_kernel_v2(Dummy x) { double result = atanh(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kSinpi{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void sinpif_kernel_v1(float* x) { float result = sinpif(x); }
|
||||
__global__ void sinpif_kernel_v2(Dummy x) { float result = sinpif(x); }
|
||||
__global__ void sinpi_kernel_v1(double* x) { double result = sinpi(x); }
|
||||
__global__ void sinpi_kernel_v2(Dummy x) { double result = sinpi(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kCospi{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void cospif_kernel_v1(float* x) { float result = cospif(x); }
|
||||
__global__ void cospif_kernel_v2(Dummy x) { float result = cospif(x); }
|
||||
__global__ void cospi_kernel_v1(double* x) { double result = cospi(x); }
|
||||
__global__ void cospi_kernel_v2(Dummy x) { double result = cospi(x); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kAtan2{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void atan2f_kernel_v1(float* x, float y) { float result = atan2f(x, y); }
|
||||
__global__ void atan2f_kernel_v2(float x, float* y) { float result = atan2f(x, y); }
|
||||
__global__ void atan2f_kernel_v3(Dummy x, float y) { float result = atan2f(x, y); }
|
||||
__global__ void atan2f_kernel_v4(float x, Dummy y) { float result = atan2f(x, y); }
|
||||
__global__ void atan2_kernel_v1(double* x, double y) { double result = atan2(x, y); }
|
||||
__global__ void atan2_kernel_v2(double x, double* y) { double result = atan2(x, y); }
|
||||
__global__ void atan2_kernel_v3(Dummy x, double y) { double result = atan2(x, y); }
|
||||
__global__ void atan2_kernel_v4(double x, Dummy y) { double result = atan2(x, y); }
|
||||
)"};
|
||||
|
||||
static constexpr auto kSincos{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void sincosf_kernel_v1(float* x, float* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v2(Dummy x, float* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v3(float x, char* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v4(float x, short* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v5(float x, int* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v6(float x, long* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v7(float x, long long* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v8(float x, double* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v9(float x, Dummy* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v10(float x, const float* sptr, float* cptr) {
|
||||
sincosf(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincosf_kernel_v11(float x, float* sptr, char* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v12(float x, float* sptr, short* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v13(float x, float* sptr, int* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v14(float x, float* sptr, long* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v15(float x, float* sptr, long long* cptr) {
|
||||
sincosf(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincosf_kernel_v16(float x, float* sptr, double* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v17(float x, float* sptr, Dummy* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v18(float x, float* sptr, const float* cptr) {
|
||||
sincosf(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincos_kernel_v1(double* x, double* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v2(Dummy x, double* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v3(double x, char* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v4(double x, short* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v5(double x, int* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v6(double x, long* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v7(double x, long long* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v8(double x, float* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v9(double x, Dummy* sptr, double* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v10(double x, const double* sptr, double* cptr) {
|
||||
sincos(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincos_kernel_v11(double x, double* sptr, char* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v12(double x, double* sptr, short* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v13(double x, double* sptr, int* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v14(double x, double* sptr, long* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v15(double x, double* sptr, long long* cptr) {
|
||||
sincos(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincos_kernel_v16(double x, double* sptr, float* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v17(double x, double* sptr, Dummy* cptr) { sincos(x, sptr, cptr); }
|
||||
__global__ void sincos_kernel_v18(double x, double* sptr, const double* cptr) {
|
||||
sincos(x, sptr, cptr);
|
||||
}
|
||||
)"};
|
||||
|
||||
static constexpr auto kSincospi{R"(
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
__global__ void sincospif_kernel_v1(float* x, float* sptr, float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v2(Dummy x, float* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v3(float x, char* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v4(float x, short* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v5(float x, int* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v6(float x, long* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v7(float x, long long* sptr, float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v8(float x, double* sptr, float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v9(float x, Dummy* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v10(float x, const float* sptr, float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v11(float x, float* sptr, char* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v12(float x, float* sptr, short* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v13(float x, float* sptr, int* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v14(float x, float* sptr, long* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v15(float x, float* sptr, long long* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v16(float x, float* sptr, double* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v17(float x, float* sptr, Dummy* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v18(float x, float* sptr, const float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospi_kernel_v1(float* x, float* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v2(Dummy x, float* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v3(float x, char* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v4(float x, short* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v5(float x, int* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v6(float x, long* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v7(float x, long long* sptr, float* cptr) {
|
||||
sincospi(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospi_kernel_v8(float x, double* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v9(float x, Dummy* sptr, float* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v10(float x, const float* sptr, float* cptr) {
|
||||
sincospi(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospi_kernel_v11(float x, float* sptr, char* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v12(float x, float* sptr, short* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v13(float x, float* sptr, int* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v14(float x, float* sptr, long* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v15(float x, float* sptr, long long* cptr) {
|
||||
sincospi(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospi_kernel_v16(float x, float* sptr, double* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v17(float x, float* sptr, Dummy* cptr) { sincospi(x, sptr, cptr); }
|
||||
__global__ void sincospi_kernel_v18(float x, float* sptr, const float* cptr) {
|
||||
sincospi(x, sptr, cptr);
|
||||
}
|
||||
)"};
|
||||
@@ -0,0 +1,118 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
class Dummy {
|
||||
public:
|
||||
__device__ Dummy() {}
|
||||
__device__ ~Dummy() {}
|
||||
};
|
||||
|
||||
#define TRIG_SP_UNARY_NEGATIVE_KERNELS(func_name) \
|
||||
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
|
||||
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
|
||||
|
||||
/*Expecting 2 errors per macro invocation - 26 total*/
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(sin)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(cos)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(tan)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(asin)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(acos)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(atan)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(sinh)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(cosh)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(tanh)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(asinh)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(atanh)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(sinpi)
|
||||
TRIG_SP_UNARY_NEGATIVE_KERNELS(cospi)
|
||||
|
||||
/*Expecting 4 errors*/
|
||||
__global__ void atan2f_kernel_v1(float* x, float y) { float result = atan2f(x, y); }
|
||||
__global__ void atan2f_kernel_v2(float x, float* y) { float result = atan2f(x, y); }
|
||||
__global__ void atan2f_kernel_v3(Dummy x, float y) { float result = atan2f(x, y); }
|
||||
__global__ void atan2f_kernel_v4(float x, Dummy y) { float result = atan2f(x, y); }
|
||||
|
||||
/*Expecting 18 errors*/
|
||||
__global__ void sincosf_kernel_v1(float* x, float* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v2(Dummy x, float* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v3(float x, char* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v4(float x, short* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v5(float x, int* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v6(float x, long* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v7(float x, long long* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v8(float x, double* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v9(float x, Dummy* sptr, float* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v10(float x, const float* sptr, float* cptr) {
|
||||
sincosf(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincosf_kernel_v11(float x, float* sptr, char* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v12(float x, float* sptr, short* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v13(float x, float* sptr, int* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v14(float x, float* sptr, long* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v15(float x, float* sptr, long long* cptr) {
|
||||
sincosf(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincosf_kernel_v16(float x, float* sptr, double* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v17(float x, float* sptr, Dummy* cptr) { sincosf(x, sptr, cptr); }
|
||||
__global__ void sincosf_kernel_v18(float x, float* sptr, const float* cptr) {
|
||||
sincosf(x, sptr, cptr);
|
||||
}
|
||||
|
||||
/*Expecting 18 errors*/
|
||||
__global__ void sincospif_kernel_v1(float* x, float* sptr, float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v2(Dummy x, float* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v3(float x, char* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v4(float x, short* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v5(float x, int* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v6(float x, long* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v7(float x, long long* sptr, float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v8(float x, double* sptr, float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v9(float x, Dummy* sptr, float* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v10(float x, const float* sptr, float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v11(float x, float* sptr, char* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v12(float x, float* sptr, short* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v13(float x, float* sptr, int* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v14(float x, float* sptr, long* cptr) { sincospif(x, sptr, cptr); }
|
||||
__global__ void sincospif_kernel_v15(float x, float* sptr, long long* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v16(float x, float* sptr, double* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v17(float x, float* sptr, Dummy* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
__global__ void sincospif_kernel_v18(float x, float* sptr, const float* cptr) {
|
||||
sincospif(x, sptr, cptr);
|
||||
}
|
||||
@@ -0,0 +1,243 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "math_common.hh"
|
||||
#include "math_special_values.hh"
|
||||
|
||||
#include <hip/hip_cooperative_groups.h>
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
#define MATH_UNARY_KERNEL_DEF(func_name) \
|
||||
template <typename T, typename RT = T> \
|
||||
__global__ void func_name##_kernel(RT* const ys, const size_t num_xs, T* const xs) { \
|
||||
const auto tid = cg::this_grid().thread_rank(); \
|
||||
const auto stride = cg::this_grid().size(); \
|
||||
\
|
||||
for (auto i = tid; i < num_xs; i += stride) { \
|
||||
if constexpr (std::is_same_v<float, T>) { \
|
||||
ys[i] = func_name##f(xs[i]); \
|
||||
} else if constexpr (std::is_same_v<double, T>) { \
|
||||
ys[i] = func_name(xs[i]); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void UnaryHalfPrecisionBruteForceTest(kernel_sig<T, Float16> kernel, ref_sig<RT, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
uint64_t stop = std::numeric_limits<uint16_t>::max() + 1ul;
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(Float16) + sizeof(T)), stop);
|
||||
LinearAllocGuard<Float16> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(Float16)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
|
||||
for (uint64_t v = 0u; v < stop;) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, stop - v);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
|
||||
thread_pool.Post([=, &values] {
|
||||
auto t = v;
|
||||
uint16_t val;
|
||||
for (auto j = 0u; j < sub_batch_size; ++j) {
|
||||
val = static_cast<uint16_t>(t++);
|
||||
values.ptr()[base_idx + j] = *reinterpret_cast<Float16*>(&val);
|
||||
}
|
||||
});
|
||||
|
||||
v += sub_batch_size;
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void UnarySinglePrecisionBruteForceTest(kernel_sig<T, float> kernel, ref_sig<RT, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
uint64_t stop = std::numeric_limits<uint32_t>::max() + 1ul;
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(float) + sizeof(T)), stop);
|
||||
LinearAllocGuard<float> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(float)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
|
||||
for (uint64_t v = 0u; v < stop;) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, stop - v);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
|
||||
thread_pool.Post([=, &values] {
|
||||
auto t = v;
|
||||
uint32_t val;
|
||||
for (auto j = 0u; j < sub_batch_size; ++j) {
|
||||
val = static_cast<uint32_t>(t++);
|
||||
values.ptr()[base_idx + j] = *reinterpret_cast<float*>(&val);
|
||||
}
|
||||
});
|
||||
|
||||
v += sub_batch_size;
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void UnarySinglePrecisionRangeTest(kernel_sig<T, float> kernel, ref_sig<RT, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder, const float a,
|
||||
const float b) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const auto max_batch_size = GetMaxAllowedDeviceMemoryUsage() / (sizeof(float) + sizeof(T));
|
||||
LinearAllocGuard<float> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(float)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
size_t inserted = 0u;
|
||||
for (float v = a; v != b; v = std::nextafter(v, b)) {
|
||||
values.ptr()[inserted++] = v;
|
||||
if (inserted < max_batch_size) continue;
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, inserted, values.ptr());
|
||||
inserted = 0u;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void UnaryDoublePrecisionBruteForceTest(kernel_sig<T, double> kernel, ref_sig<RT, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder,
|
||||
const double a = std::numeric_limits<double>::lowest(),
|
||||
const double b = std::numeric_limits<double>::max()) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const uint64_t num_iterations = GetTestIterationCount();
|
||||
const auto max_batch_size =
|
||||
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(double) + sizeof(T)), num_iterations);
|
||||
LinearAllocGuard<double> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(double)};
|
||||
|
||||
MathTest math_test(kernel, max_batch_size);
|
||||
|
||||
auto batch_size = max_batch_size;
|
||||
const auto num_threads = thread_pool.thread_count();
|
||||
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
|
||||
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
|
||||
|
||||
const auto min_sub_batch_size = batch_size / num_threads;
|
||||
const auto tail = batch_size % num_threads;
|
||||
|
||||
auto base_idx = 0u;
|
||||
for (auto i = 0u; i < num_threads; ++i) {
|
||||
const auto sub_batch_size = min_sub_batch_size + (i < tail);
|
||||
thread_pool.Post([=, &values] {
|
||||
const auto generator = [=] {
|
||||
static thread_local std::mt19937 rng(std::random_device{}());
|
||||
std::uniform_real_distribution<long double> unif_dist(a, b);
|
||||
return static_cast<double>(unif_dist(rng));
|
||||
};
|
||||
std::generate(values.ptr() + base_idx, values.ptr() + base_idx + sub_batch_size, generator);
|
||||
});
|
||||
base_idx += sub_batch_size;
|
||||
}
|
||||
|
||||
thread_pool.Wait();
|
||||
|
||||
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void UnaryDoublePrecisionSpecialValuesTest(kernel_sig<T, double> kernel,
|
||||
ref_sig<RT, RTArg> ref_func,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
|
||||
const auto values = std::get<SpecialVals<double>>(kSpecialValRegistry);
|
||||
|
||||
MathTest math_test(kernel, values.size);
|
||||
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, values.size,
|
||||
values.data);
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void UnaryHalfPrecisionTest(kernel_sig<T, Float16> kernel, ref_sig<RT, RTArg> ref,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Brute force") { UnaryHalfPrecisionBruteForceTest(kernel, ref, validator_builder); }
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void UnarySinglePrecisionTest(kernel_sig<T, float> kernel, ref_sig<RT, RTArg> ref,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Brute force") { UnarySinglePrecisionBruteForceTest(kernel, ref, validator_builder); }
|
||||
}
|
||||
|
||||
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
|
||||
void UnaryDoublePrecisionTest(kernel_sig<T, double> kernel, ref_sig<RT, RTArg> ref,
|
||||
const ValidatorBuilder& validator_builder) {
|
||||
SECTION("Special values") {
|
||||
UnaryDoublePrecisionSpecialValuesTest(kernel, ref, validator_builder);
|
||||
}
|
||||
|
||||
SECTION("Brute force") { UnaryDoublePrecisionBruteForceTest(kernel, ref, validator_builder); }
|
||||
}
|
||||
|
||||
#define MATH_UNARY_WITHIN_ULP_TEST_DEF(kern_name, ref_func, sp_ulp, dp_ulp) \
|
||||
MATH_UNARY_KERNEL_DEF(kern_name) \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - float") { \
|
||||
double (*ref)(double) = ref_func; \
|
||||
UnarySinglePrecisionTest(kern_name##_kernel<float>, ref, \
|
||||
ULPValidatorBuilderFactory<float>(sp_ulp)); \
|
||||
} \
|
||||
\
|
||||
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - double") { \
|
||||
long double (*ref)(long double) = ref_func; \
|
||||
UnaryDoublePrecisionTest(kern_name##_kernel<double>, ref, \
|
||||
ULPValidatorBuilderFactory<double>(dp_ulp)); \
|
||||
}
|
||||
|
||||
#define MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(func_name, sp_ulp, dp_ulp) \
|
||||
MATH_UNARY_WITHIN_ULP_TEST_DEF(func_name, std::func_name, sp_ulp, dp_ulp)
|
||||
@@ -0,0 +1,152 @@
|
||||
/*
|
||||
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <catch.hpp>
|
||||
|
||||
// Define a new MatcherBase class with a public 'describe' member function because
|
||||
// Catch::MatcherBase::describe is protected and thus can't be used via a pointer to
|
||||
// Catch::MatcherBase.
|
||||
template <typename T> class MatcherBase : public Catch::MatcherBase<T> {
|
||||
public:
|
||||
virtual std::string describe() const = 0;
|
||||
virtual ~MatcherBase() = default;
|
||||
};
|
||||
|
||||
template <typename T, typename Matcher> class ValidatorBase : public MatcherBase<T> {
|
||||
public:
|
||||
template <typename... Ts>
|
||||
ValidatorBase(T target, Ts&&... args) : matcher_{std::forward<Ts>(args)...}, target_{target} {}
|
||||
|
||||
bool match(const T& val) const override {
|
||||
if (std::isnan(target_)) {
|
||||
return std::isnan(val);
|
||||
}
|
||||
|
||||
return matcher_.match(val);
|
||||
}
|
||||
|
||||
std::string describe() const override {
|
||||
if (std::isnan(target_)) {
|
||||
return "is not NaN";
|
||||
}
|
||||
|
||||
return matcher_.describe();
|
||||
}
|
||||
|
||||
private:
|
||||
Matcher matcher_;
|
||||
T target_;
|
||||
bool nan = false;
|
||||
};
|
||||
|
||||
template <typename T> auto ULPValidatorBuilderFactory(int64_t ulps) {
|
||||
return [=](T target, auto&&...) {
|
||||
return std::make_unique<ValidatorBase<T, Catch::Matchers::Floating::WithinUlpsMatcher>>(
|
||||
target, Catch::WithinULP(target, ulps));
|
||||
};
|
||||
};
|
||||
|
||||
template <typename T> auto AbsValidatorBuilderFactory(double margin) {
|
||||
return [=](T target, auto&&...) {
|
||||
return std::make_unique<ValidatorBase<T, Catch::Matchers::Floating::WithinAbsMatcher>>(
|
||||
target, Catch::WithinAbs(target, margin));
|
||||
};
|
||||
}
|
||||
|
||||
template <typename T> auto RelValidatorBuilderFactory(T margin) {
|
||||
return [=](T target, auto&&...) {
|
||||
return std::make_unique<ValidatorBase<T, Catch::Matchers::Floating::WithinRelMatcher>>(
|
||||
target, Catch::WithinRel(target, margin));
|
||||
};
|
||||
}
|
||||
|
||||
template <typename T> class EqValidator : public MatcherBase<T> {
|
||||
public:
|
||||
EqValidator(T target) : target_{target} {}
|
||||
|
||||
bool match(const T& val) const override {
|
||||
if (std::isnan(target_)) {
|
||||
return std::isnan(val);
|
||||
}
|
||||
|
||||
return target_ == val;
|
||||
}
|
||||
|
||||
std::string describe() const override {
|
||||
std::stringstream ss;
|
||||
ss << " is not equal to " << target_;
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
private:
|
||||
T target_;
|
||||
};
|
||||
|
||||
template <typename T> auto EqValidatorBuilderFactory() {
|
||||
return [](T val, auto&&...) { return std::make_unique<EqValidator<T>>(val); };
|
||||
}
|
||||
|
||||
template <typename T, typename U, typename VBF, typename VBS>
|
||||
class PairValidator : public MatcherBase<std::pair<T, U>> {
|
||||
public:
|
||||
PairValidator(const std::pair<T, U>& target, const VBF& vbf, const VBS& vbs)
|
||||
: first_matcher_{vbf(target.first)}, second_matcher_{vbs(target.second)} {}
|
||||
|
||||
bool match(const std::pair<T, U>& val) const override {
|
||||
return first_matcher_->match(val.first) && second_matcher_->match(val.second);
|
||||
}
|
||||
|
||||
std::string describe() const override {
|
||||
return "<" + first_matcher_->describe() + ", " + second_matcher_->describe() + ">";
|
||||
}
|
||||
|
||||
private:
|
||||
decltype(std::declval<VBF>()(std::declval<T>())) first_matcher_;
|
||||
decltype(std::declval<VBS>()(std::declval<U>())) second_matcher_;
|
||||
};
|
||||
|
||||
template <typename T, typename ValidatorBuilder>
|
||||
auto PairValidatorBuilderFactory(const ValidatorBuilder& vb) {
|
||||
return [=](const std::pair<T, T>& t, auto&&...) {
|
||||
return std::make_unique<PairValidator<T, T, ValidatorBuilder, ValidatorBuilder>>(t, vb, vb);
|
||||
};
|
||||
}
|
||||
|
||||
template <typename T, typename U, typename VBF, typename VBS>
|
||||
auto PairValidatorBuilderFactory(const VBF& vbf, const VBS& vbs) {
|
||||
return [=](const std::pair<T, U>& t, auto&&...) {
|
||||
return std::make_unique<PairValidator<T, U, VBF, VBS>>(t, vbf, vbs);
|
||||
};
|
||||
}
|
||||
|
||||
template <typename T> class NopValidator : public MatcherBase<T> {
|
||||
public:
|
||||
bool match(const T&) const override { return true; }
|
||||
|
||||
std::string describe() const override { return ""; }
|
||||
};
|
||||
|
||||
template <typename T> auto NopValidatorBuilderFactory() {
|
||||
return [](auto&&...) { return std::make_unique<NopValidator<T>>(); };
|
||||
}
|
||||
Reference in New Issue
Block a user