SWDEV-1 - Merge github PRs to amd-staging

Change-Id: I2944a63ddc2eec8dc1403d9790ffffbaec343385
This commit is contained in:
Rakesh Roy
2024-03-04 11:51:34 +05:30
366 zmienionych plików z 55399 dodań i 2073 usunięć
+164
Wyświetl plik
@@ -0,0 +1,164 @@
# Copyright (c) 2023 Advanced Micro Devices, Inc. All Rights Reserved.
#
# Permission is hereby granted, free of charge, to any person obtaining a copy
# of this software and associated documentation files (the "Software"), to deal
# in the Software without restriction, including without limitation the rights
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
# copies of the Software, and to permit persons to whom the Software is
# furnished to do so, subject to the following conditions:
#
# The above copyright notice and this permission notice shall be included in
# all copies or substantial portions of the Software.
#
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
# THE SOFTWARE.
set(TEST_SRC
trig_funcs.cc
misc_funcs.cc
remainder_and_rounding_funcs.cc
single_precision_intrinsics.cc
double_precision_intrinsics.cc
integer_intrinsics.cc
root_funcs.cc
log_funcs.cc
special_funcs.cc
casting_double_funcs.cc
casting_float_funcs.cc
casting_int_funcs.cc
casting_half2int_funcs.cc
casting_int2half_funcs.cc
casting_half_float_funcs.cc
)
if(HIP_PLATFORM MATCHES "nvidia")
set(LINKER_LIBS nvrtc)
elseif(HIP_PLATFORM MATCHES "amd")
set(TEST_SRC ${TEST_SRC}
pow_funcs.cc
casting_half2_funcs.cc
half_precision_math.cc
half_precision_arithmetic.cc
half_precision_comparison.cc
)
set(LINKER_LIBS hiprtc)
endif()
find_package(Boost 1.70.0)
message(STATUS "Boost_FOUND: ${Boost_FOUND}")
if(Boost_FOUND)
hip_add_exe_to_target(NAME MathsTest
TEST_SRC ${TEST_SRC}
TEST_TARGET_NAME build_tests COMMON_SHARED_SRC ${COMMON_SHARED_SRC}
LINKER_LIBS ${LINKER_LIBS})
target_include_directories(MathsTest PRIVATE ${Boost_INCLUDE_DIRS})
else()
message(STATUS "Boost not found. Dependent math tests not enabled.")
endif()
# Below tests fail in PSDB
#add_test(NAME Unit_Device_Single_Precision_Trig_Functions_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# trig_single_precision_negative_kernels.cc 66)
#
#add_test(NAME Unit_Device_Double_Precision_Trig_Functions_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# trig_double_precision_negative_kernels.cc 66)
#add_test(NAME Unit_Device_Misc_Functions_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# misc_negative_kernels.cc 76)
#
#add_test(NAME Unit_Device_remainder_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# math_remainder_negative_kernels.cc 68)
#
#add_test(NAME Unit_Device_rounding_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# math_rounding_negative_kernels.cc 40)
#
#add_test(NAME Unit_Single_Precision_Intrinsics_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# single_precision_intrinsics_negative_kernels.cc 42)
#
#add_test(NAME Unit_Double_Precision_Intrinsics_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# double_precision_intrinsics_negative_kernels.cc 18)
#
#add_test(NAME Unit_Integer_Intrinsics_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# integer_intrinsics_negative_kernels.cc 20)
#add_test(NAME Unit_Device_root_1Dand2D_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# math_root_negative_kernels_1Dand2D.cc 68)
#
#add_test(NAME Unit_Device_root_3Dand4D_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# math_root_negative_kernels_3Dand4D.cc 56)
#add_test(NAME Unit_Device_pow_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# math_pow_negative_kernels.cc 76)
#add_test(NAME Unit_Device_log_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# math_log_negative_kernels.cc 24)
#add_test(NAME Unit_Device_special_funcs_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# math_special_func_kernels.cc 76)
#add_test(NAME Unit_Device_casting_double_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# casting_double_negative_kernels.cc 69)
#add_test(NAME Unit_Device_casting_float_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# casting_float_negative_kernels.cc 54)
#add_test(NAME Unit_Device_casting_int_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# casting_int_negative_kernels.cc 92)
#
#add_test(NAME Unit_Device_casting_half2_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# casting_half2_negative_kernels.cc 53)
#add_test(NAME Unit_Half_Precision_Math_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# half_precision_math_negative_kernels.cc 60)
#add_test(NAME Unit_Half_Precision_Arithmetic_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# half_precision_arithmetic_negative_kernels.cc 88)
#add_test(NAME Unit_Half_Precision_Comparison_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# half_precision_comparison_negative_kernels.cc 168)
#add_test(NAME Unit_Device_casting_half2int_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# casting_half2int_negative_kernels.cc 78)
#add_test(NAME Unit_Device_casting_int2half_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# casting_int2half_negative_kernels.cc 78)
#add_test(NAME Unit_Device_casting_half_float_Negative
# COMMAND python3 ${CMAKE_CURRENT_SOURCE_DIR}/../compileAndCaptureOutput.py
# ${CMAKE_CURRENT_SOURCE_DIR} ${HIP_PLATFORM} ${HIP_PATH}
# casting_half_float_negative_kernels.cc 18)
+55
Wyświetl plik
@@ -0,0 +1,55 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include <hip/hip_fp16.h>
#define FLOAT16_MAX 65504.0f
class Float16 {
public:
__host__ __device__ Float16() = default;
__host__ __device__ Float16(__half x) : x_{x} {}
__host__ __device__ Float16(__half2 x) : x_{__low2half(x)} {}
__host__ __device__ Float16(float x) : x_{__float2half(x)} {}
// __heq doesn't have a __host__ version
__host__ __device__ bool operator==(Float16 other) const { return (static_cast<__half_raw>(x_).x == static_cast<__half_raw>(other.x_).x); }
__host__ __device__ bool operator!=(Float16 other) const { return !(*this == other); }
__host__ __device__ operator __half() const { return x_; }
__host__ __device__ operator __half2() const { return __half2half2(x_); }
__host__ __device__ operator float() const { return __half2float(x_); }
private:
__half x_;
};
namespace {
inline std::ostream& operator<<(std::ostream& o, Float16 x) {
o << static_cast<float>(x);
return o;
}
} // namespace
+141
Wyświetl plik
@@ -0,0 +1,141 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "math_common.hh"
#include "math_special_values.hh"
#include <hip/hip_cooperative_groups.h>
namespace cg = cooperative_groups;
#define MATH_BINARY_KERNEL_DEF(func_name) \
template <typename T, typename RT = T> \
__global__ void func_name##_kernel(RT* const ys, const size_t num_xs, T* const x1s, \
T* const x2s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
if constexpr (std::is_same_v<float, T>) { \
ys[i] = func_name##f(x1s[i], x2s[i]); \
} else if constexpr (std::is_same_v<double, T>) { \
ys[i] = func_name(x1s[i], x2s[i]); \
} \
} \
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void BinaryFloatingPointBruteForceTest(kernel_sig<T, TArg, TArg> kernel,
ref_sig<RT, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder,
const TArg a = std::numeric_limits<TArg>::lowest(),
const TArg b = std::numeric_limits<TArg>::max()) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const uint64_t num_iterations = GetTestIterationCount();
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(TArg) * 2 + sizeof(T)), num_iterations);
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &x1s, &x2s] {
const auto generator = [=] {
static thread_local std::mt19937 rng(std::random_device{}());
if constexpr (std::is_same_v<TArg, Float16>) {
std::uniform_real_distribution<RefType_t<Float16>> unif_dist(-FLOAT16_MAX, FLOAT16_MAX);
return static_cast<Float16>(unif_dist(rng));
} else {
std::uniform_real_distribution<RefType_t<TArg>> unif_dist(a, b);
return static_cast<TArg>(unif_dist(rng));
}
};
std::generate(x1s.ptr() + base_idx, x1s.ptr() + base_idx + sub_batch_size, generator);
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator);
});
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
x2s.ptr());
}
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void BinaryFloatingPointSpecialValuesTest(kernel_sig<T, TArg, TArg> kernel,
ref_sig<RT, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
using SpecialValsType = std::conditional_t<std::is_same_v<TArg, Float16>, float, TArg>;
const auto values = std::get<SpecialVals<SpecialValsType>>(kSpecialValRegistry);
const auto size = values.size * values.size;
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
for (auto i = 0u; i < values.size; ++i) {
for (auto j = 0u; j < values.size; ++j) {
x1s.ptr()[i * values.size + j] = values.data[i];
x2s.ptr()[i * values.size + j] = values.data[j];
}
}
MathTest math_test(kernel, size);
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, size, x1s.ptr(),
x2s.ptr());
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void BinaryFloatingPointTest(kernel_sig<T, TArg, TArg> kernel, ref_sig<RT, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
SECTION("Special values") {
BinaryFloatingPointSpecialValuesTest(kernel, ref_func, validator_builder);
}
SECTION("Brute force") { BinaryFloatingPointBruteForceTest(kernel, ref_func, validator_builder); }
}
#define MATH_BINARY_WITHIN_ULP_TEST_DEF(kern_name, ref_func, sp_ulp, dp_ulp) \
MATH_BINARY_KERNEL_DEF(kern_name) \
\
TEMPLATE_TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive", "", float, double) { \
using RT = RefType_t<TestType>; \
RT (*ref)(RT, RT) = ref_func; \
const auto ulp = std::is_same_v<float, TestType> ? sp_ulp : dp_ulp; \
\
BinaryFloatingPointTest(kern_name##_kernel<TestType>, ref, \
ULPValidatorBuilderFactory<TestType>(ulp)); \
}
+259
Wyświetl plik
@@ -0,0 +1,259 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "unary_common.hh"
#include <fenv.h>
namespace cg = cooperative_groups;
#define CAST_KERNEL_DEF(func_name, T1, T2) \
__global__ void func_name##_kernel(T1* const ys, const size_t num_xs, T2* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(xs[i]); \
} \
}
#define CAST_BINARY_KERNEL_DEF(func_name, T1, T2) \
__global__ void func_name##_kernel(T1* const ys, const size_t num_xs, T2* const x1s, \
T2* const x2s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(x1s[i], x2s[i]); \
} \
}
#define CAST_F2I_REF_DEF(func_name, T1, T2, ref_func) \
T1 func_name##_ref(T2 arg) { \
if (arg >= static_cast<T2>(std::numeric_limits<T1>::max())) \
return std::numeric_limits<T1>::max(); \
else if (arg <= static_cast<T2>(std::numeric_limits<T1>::min())) \
return std::numeric_limits<T1>::min(); \
T2 result = ref_func(arg); \
return result; \
}
#define CAST_F2I_RZ_REF_DEF(func_name, T1, T2) \
T1 func_name##_ref(T2 arg) { \
if (arg >= static_cast<double>(std::numeric_limits<T1>::max())) \
return std::numeric_limits<T1>::max(); \
else if (arg <= static_cast<double>(std::numeric_limits<T1>::min())) \
return std::numeric_limits<T1>::min(); \
T1 result = static_cast<T1>(arg); \
return result; \
}
#define CAST_RND_REF_DEF(func_name, T1, T2, round_dir) \
T1 func_name##_ref(T2 arg) { \
int curr_direction = fegetround(); \
fesetround(round_dir); \
T1 result = static_cast<T1>(arg); \
fesetround(curr_direction); \
return result; \
}
#define CAST_REF_DEF(func_name, T1, T2) \
T1 func_name##_ref(T2 arg) { \
T1 result = static_cast<T1>(arg); \
return result; \
}
template <typename T1, typename T2> T1 type2_as_type1_ref(T2 arg) {
T1 tmp;
memcpy(&tmp, &arg, sizeof(tmp));
return tmp;
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void CastUnaryHalfPrecisionBruteForceTest(kernel_sig<T, Float16> kernel,
ref_sig<RT, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
uint64_t stop = std::numeric_limits<uint16_t>::max() + 1ul;
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(Float16) + sizeof(T)), stop);
LinearAllocGuard<Float16> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(Float16)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t v = 0u; v < stop;) {
batch_size = std::min<uint64_t>(max_batch_size, stop - v);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &values] {
auto t = v;
uint16_t val;
for (auto j = 0u; j < sub_batch_size; ++j) {
val = static_cast<uint16_t>(t++);
values.ptr()[base_idx + j] = *reinterpret_cast<Float16*>(&val);
if (std::isnan(values.ptr()[base_idx + j]) || std::isinf(values.ptr()[base_idx + j])) {
values.ptr()[base_idx + j] = 0;
}
}
});
v += sub_batch_size;
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
}
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void CastUnaryHalfPrecisionTest(kernel_sig<T, Float16> kernel, ref_sig<RT, RTArg> ref,
const ValidatorBuilder& validator_builder) {
SECTION("Brute force") { CastUnaryHalfPrecisionBruteForceTest(kernel, ref, validator_builder); }
}
template <typename T, typename ValidatorBuilder>
void CastDoublePrecisionSpecialValuesTest(kernel_sig<T, double> kernel, ref_sig<T, double> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const auto values = std::get<SpecialVals<double>>(kSpecialValRegistry);
std::vector<double> spec_values;
if (!std::is_same_v<float, T> && !std::is_same_v<double, T> && !std::is_same_v<long double, T>) {
for (int i = 0; i < values.size; i++) {
if (!std::isnan(values.data[i]) && !std::isinf(values.data[i])) {
spec_values.push_back(values.data[i]);
}
}
}
MathTest math_test(kernel, spec_values.size());
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func,
spec_values.size(), spec_values.data());
}
template <typename T, typename ValidatorBuilder>
void CastDoublePrecisionTest(kernel_sig<T, double> kernel, ref_sig<T, double> ref,
const ValidatorBuilder& validator_builder) {
SECTION("Special values") {
CastDoublePrecisionSpecialValuesTest(kernel, ref, validator_builder);
}
SECTION("Brute force") { UnaryDoublePrecisionBruteForceTest(kernel, ref, validator_builder); }
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void CastIntRangeTest(kernel_sig<T, TArg> kernel, ref_sig<RT, RTArg> ref_func,
const ValidatorBuilder& validator_builder,
const TArg a = std::numeric_limits<TArg>::lowest(),
const TArg b = std::numeric_limits<TArg>::max()) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const auto max_batch_size = GetMaxAllowedDeviceMemoryUsage() / (sizeof(T) + sizeof(TArg));
LinearAllocGuard<TArg> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
MathTest math_test(kernel, max_batch_size);
size_t inserted = 0u;
for (TArg v = a; v <= b; v++) {
values.ptr()[inserted++] = v;
if (inserted < max_batch_size) continue;
math_test.Run(validator_builder, grid_size, block_size, ref_func, inserted, values.ptr());
inserted = 0u;
}
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void CastIntBruteForceTest(kernel_sig<T, TArg> kernel, ref_sig<RT, RTArg> ref_func,
const ValidatorBuilder& validator_builder,
const TArg a = std::numeric_limits<TArg>::lowest(),
const TArg b = std::numeric_limits<TArg>::max()) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const uint64_t num_iterations = GetTestIterationCount();
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(T) + sizeof(TArg)), num_iterations);
LinearAllocGuard<TArg> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &values] {
const auto generator = [=] {
static thread_local std::mt19937 rng(std::random_device{}());
std::uniform_int_distribution<TArg> unif_dist(a, b);
return static_cast<TArg>(unif_dist(rng));
};
std::generate(values.ptr() + base_idx, values.ptr() + base_idx + sub_batch_size, generator);
});
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
}
}
template <typename T1, typename T2, typename ValidatorBuilder>
void CastBinaryIntRangeTest(kernel_sig<T1, T2, T2> kernel, ref_sig<T1, T2, T2> ref_func,
const ValidatorBuilder& validator_builder,
const T2 a = std::numeric_limits<T2>::lowest(),
const T2 b = std::numeric_limits<T2>::max()) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const auto max_batch_size = GetMaxAllowedDeviceMemoryUsage() / (sizeof(T1) + 2 * sizeof(T2));
LinearAllocGuard<T2> values1{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(T2)};
LinearAllocGuard<T2> values2{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(T2)};
MathTest math_test(kernel, max_batch_size);
size_t inserted = 0u;
for (T2 v = a; v <= b; v++) {
values1.ptr()[inserted] = v;
values2.ptr()[inserted++] = b - v;
if (inserted < max_batch_size) continue;
math_test.Run(validator_builder, grid_size, block_size, ref_func, inserted, values1.ptr(),
values2.ptr());
inserted = 0u;
}
}
@@ -0,0 +1,597 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "casting_common.hh"
#include "casting_double_negative_kernels_rtc.hh"
/**
* @addtogroup CastingDoubleType CastingDoubleType
* @{
* @ingroup MathTest
*/
#define CAST_DOUBLE2INT_TEST_DEF(kern_name, T, ref_func) \
CAST_KERNEL_DEF(kern_name, T, double) \
CAST_F2I_REF_DEF(kern_name, T, double, ref_func) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T (*ref)(double) = kern_name##_ref; \
CastDoublePrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>()); \
}
#define CAST_DOUBLE2INT_RZ_TEST_DEF(kern_name, T) \
CAST_KERNEL_DEF(kern_name, T, double) \
CAST_F2I_RZ_REF_DEF(kern_name, T, double) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T (*ref)(double) = kern_name##_ref; \
CastDoublePrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__double2int_rd` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::floor`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2INT_TEST_DEF(__double2int_rd, int, std::floor)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2int_rn` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::rint`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2INT_TEST_DEF(__double2int_rn, int, std::rint)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2int_ru` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::ceil`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2INT_TEST_DEF(__double2int_ru, int, std::ceil)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2int_rz` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* performs cast to int.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2INT_RZ_TEST_DEF(__double2int_rz, int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __double2int_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2int_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2Int); }
/**
* Test Description
* ------------------------
* - Tests that checks `__double2uint_rd` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* `std::floor`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2INT_TEST_DEF(__double2uint_rd, unsigned int, std::floor)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2uint_rn` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* `std::rint`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2INT_TEST_DEF(__double2uint_rn, unsigned int, std::rint)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2uint_ru` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* `std::ceil`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2INT_TEST_DEF(__double2uint_ru, unsigned int, std::ceil)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2uint_rz` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* which performs cast to unsigned int.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2INT_RZ_TEST_DEF(__double2uint_rz, unsigned int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __double2uint_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2uint_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2Uint); }
#define CAST_DOUBLE2LL_TEST_DEF(kern_name, T, ref_func) \
CAST_KERNEL_DEF(kern_name, T, double) \
CAST_F2I_REF_DEF(kern_name, T, double, ref_func) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T (*ref)(double) = kern_name##_ref; \
UnaryDoublePrecisionBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
static_cast<double>(std::numeric_limits<T>::min()), \
static_cast<double>(std::numeric_limits<T>::max())); \
}
#define CAST_DOUBLE2LL_RZ_TEST_DEF(kern_name, T) \
CAST_KERNEL_DEF(kern_name, T, double) \
CAST_F2I_RZ_REF_DEF(kern_name, T, double) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T (*ref)(double) = kern_name##_ref; \
UnaryDoublePrecisionBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
static_cast<double>(std::numeric_limits<T>::min()), \
static_cast<double>(std::numeric_limits<T>::max())); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__double2ll_rd` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::floor`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2LL_TEST_DEF(__double2ll_rd, long long int, std::floor)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2ll_rn` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::rint`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2LL_TEST_DEF(__double2ll_rn, long long int, std::rint)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2ll_ru` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::ceil`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2LL_TEST_DEF(__double2ll_ru, long long int, std::ceil)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2ll_rz` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* performs cast to long long int.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2LL_RZ_TEST_DEF(__double2ll_rz, long long int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __double2ll_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2ll_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2LL); }
/**
* Test Description
* ------------------------
* - Tests that checks `__double2ull_rd` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::floor`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2LL_TEST_DEF(__double2ull_rd, unsigned long long int, std::floor)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2ull_rn` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::rint`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2LL_TEST_DEF(__double2ull_rn, unsigned long long int, std::rint)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2ull_ru` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function
* `std::ceil`.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2LL_TEST_DEF(__double2ull_ru, unsigned long long int, std::ceil)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2ull_rz` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* performs cast to unsigned long long int.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2LL_RZ_TEST_DEF(__double2ull_rz, unsigned long long int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __double2ull_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2ull_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2ULL); }
#define CAST_DOUBLE2FLOAT_TEST_DEF(kern_name, round_dir) \
CAST_KERNEL_DEF(kern_name, float, double) \
CAST_RND_REF_DEF(kern_name, float, double, round_dir) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
float (*ref)(double) = kern_name##_ref; \
CastDoublePrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<float>()); \
}
#define CAST_DOUBLE2FLOAT_RN_TEST_DEF(kern_name) \
CAST_KERNEL_DEF(kern_name, float, double) \
CAST_REF_DEF(kern_name, float, double) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
float (*ref)(double) = kern_name##_ref; \
CastDoublePrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<float>()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__double2float_rd` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* which performs cast to float with rounding mode FE_DOWNWARD.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2FLOAT_TEST_DEF(__double2float_rd, FE_DOWNWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2float_rn` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* which performs cast to float.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2FLOAT_RN_TEST_DEF(__double2float_rn)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2float_ru` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* which performs cast to float with rounding mode FE_UPWARD.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2FLOAT_TEST_DEF(__double2float_ru, FE_UPWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__double2float_rz` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* which performs cast to float with rounding mode FE_TOWARDZERO.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_DOUBLE2FLOAT_TEST_DEF(__double2float_rz, FE_TOWARDZERO)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __double2float_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2float_Negative_RTC") { NegativeTestRTCWrapper<12>(kDouble2Float); }
CAST_KERNEL_DEF(__double2hiint, int, double)
int __double2hiint_ref(double arg) {
int tmp[2];
memcpy(tmp, &arg, sizeof(tmp));
return tmp[1];
}
/**
* Test Description
* ------------------------
* - Tests that checks `__double2hiint` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* performs copy of higher part of double value to int variable.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2hiint_Positive") {
int (*ref)(double) = __double2hiint_ref;
CastDoublePrecisionTest(__double2hiint_kernel, ref, EqValidatorBuilderFactory<int>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __double2hiint.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2hiint_Negative_RTC") { NegativeTestRTCWrapper<3>(kDouble2Hiint); }
CAST_KERNEL_DEF(__double2loint, int, double)
int __double2loint_ref(double arg) {
int tmp[2];
memcpy(tmp, &arg, sizeof(tmp));
return tmp[0];
}
/**
* Test Description
* ------------------------
* - Tests that checks `__double2loint` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* performs copy of lower part of double value to int variable.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2loint_Positive") {
int (*ref)(double) = __double2loint_ref;
CastDoublePrecisionTest(__double2loint_kernel, ref, EqValidatorBuilderFactory<int>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __double2loint.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double2loint_Negative_RTC") { NegativeTestRTCWrapper<3>(kDouble2Loint); }
CAST_KERNEL_DEF(__double_as_longlong, long long int, double)
/**
* Test Description
* ------------------------
* - Tests that checks `__double_as_longlong` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* which performs copy of double value to long long int variable.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double_as_longlong_Positive") {
long long int (*ref)(double) = type2_as_type1_ref<long long int, double>;
CastDoublePrecisionTest(__double_as_longlong_kernel, ref,
EqValidatorBuilderFactory<long long int>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __double_as_longlong.
*
* Test source
* ------------------------
* - unit/math/casting_double_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___double_as_longlong_Negative_RTC") {
NegativeTestRTCWrapper<3>(kDoubleAsLonglong);
}
@@ -0,0 +1,55 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL(func_name, T) \
__global__ void func_name##_kernel_v1(T* result, double* x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v2(T* result, Dummy x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v3(Dummy* result, double x) { *result = func_name(x); }
NEGATIVE_KERNELS_SHELL(__double2int_rd, int)
NEGATIVE_KERNELS_SHELL(__double2int_rn, int)
NEGATIVE_KERNELS_SHELL(__double2int_ru, int)
NEGATIVE_KERNELS_SHELL(__double2int_rz, int)
NEGATIVE_KERNELS_SHELL(__double2uint_rd, unsigned int)
NEGATIVE_KERNELS_SHELL(__double2uint_rn, unsigned int)
NEGATIVE_KERNELS_SHELL(__double2uint_ru, unsigned int)
NEGATIVE_KERNELS_SHELL(__double2uint_rz, unsigned int)
NEGATIVE_KERNELS_SHELL(__double2ll_rd, long long int)
NEGATIVE_KERNELS_SHELL(__double2ll_rn, long long int)
NEGATIVE_KERNELS_SHELL(__double2ll_ru, long long int)
NEGATIVE_KERNELS_SHELL(__double2ll_rz, long long int)
NEGATIVE_KERNELS_SHELL(__double2ull_rd, unsigned long long int)
NEGATIVE_KERNELS_SHELL(__double2ull_rn, unsigned long long int)
NEGATIVE_KERNELS_SHELL(__double2ull_ru, unsigned long long int)
NEGATIVE_KERNELS_SHELL(__double2ull_rz, unsigned long long int)
NEGATIVE_KERNELS_SHELL(__double2float_rd, float)
NEGATIVE_KERNELS_SHELL(__double2float_rn, float)
NEGATIVE_KERNELS_SHELL(__double2float_ru, float)
NEGATIVE_KERNELS_SHELL(__double2float_rz, float)
NEGATIVE_KERNELS_SHELL(__double2hiint, int)
NEGATIVE_KERNELS_SHELL(__double2loint, int)
NEGATIVE_KERNELS_SHELL(__double_as_longlong, long long int)
@@ -0,0 +1,157 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
/*
Negative kernels used for the double type casting negative Test Cases that are using RTC.
*/
static constexpr auto kDouble2Int{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void double2int_rd_kernel_v1(int* result, double* x) { *result = __double2int_rd(x); }
__global__ void double2int_rd_kernel_v2(int* result, Dummy x) { *result = __double2int_rd(x); }
__global__ void double2int_rd_kernel_v3(Dummy* result, double x) { *result = __double2int_rd(x); }
__global__ void double2int_rn_kernel_v1(int* result, double* x) { *result = __double2int_rn(x); }
__global__ void double2int_rn_kernel_v2(int* result, Dummy x) { *result = __double2int_rn(x); }
__global__ void double2int_rn_kernel_v3(Dummy* result, double x) { *result = __double2int_rn(x); }
__global__ void double2int_ru_kernel_v1(int* result, double* x) { *result = __double2int_ru(x); }
__global__ void double2int_ru_kernel_v2(int* result, Dummy x) { *result = __double2int_ru(x); }
__global__ void double2int_ru_kernel_v3(Dummy* result, double x) { *result = __double2int_ru(x); }
__global__ void double2int_rz_kernel_v1(int* result, double* x) { *result = __double2int_rz(x); }
__global__ void double2int_rz_kernel_v2(int* result, Dummy x) { *result = __double2int_rz(x); }
__global__ void double2int_rz_kernel_v3(Dummy* result, double x) { *result = __double2int_rz(x); }
)"};
static constexpr auto kDouble2Uint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void double2uint_rd_kernel_v1(unsigned int* result, double* x) { *result = __double2uint_rd(x); }
__global__ void double2uint_rd_kernel_v2(unsigned int* result, Dummy x) { *result = __double2uint_rd(x); }
__global__ void double2uint_rd_kernel_v3(Dummy* result, double x) { *result = __double2uint_rd(x); }
__global__ void double2uint_rn_kernel_v1(unsigned int* result, double* x) { *result = __double2uint_rn(x); }
__global__ void double2uint_rn_kernel_v2(unsigned int* result, Dummy x) { *result = __double2uint_rn(x); }
__global__ void double2uint_rn_kernel_v3(Dummy* result, double x) { *result = __double2uint_rn(x); }
__global__ void double2uint_ru_kernel_v1(unsigned int* result, double* x) { *result = __double2uint_ru(x); }
__global__ void double2uint_ru_kernel_v2(unsigned int* result, Dummy x) { *result = __double2uint_ru(x); }
__global__ void double2uint_ru_kernel_v3(Dummy* result, double x) { *result = __double2uint_ru(x); }
__global__ void double2uint_rz_kernel_v1(unsigned int* result, double* x) { *result = __double2uint_rz(x); }
__global__ void double2uint_rz_kernel_v2(unsigned int* result, Dummy x) { *result = __double2uint_rz(x); }
__global__ void double2uint_rz_kernel_v3(Dummy* result, double x) { *result = __double2uint_rz(x); }
)"};
static constexpr auto kDouble2LL{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void double2ll_rd_kernel_v1(long long int* result, double* x) { *result = __double2ll_rd(x); }
__global__ void double2ll_rd_kernel_v2(long long int* result, Dummy x) { *result = __double2ll_rd(x); }
__global__ void double2ll_rd_kernel_v3(Dummy* result, double x) { *result = __double2ll_rd(x); }
__global__ void double2ll_rn_kernel_v1(long long int* result, double* x) { *result = __double2ll_rn(x); }
__global__ void double2ll_rn_kernel_v2(long long int* result, Dummy x) { *result = __double2ll_rn(x); }
__global__ void double2ll_rn_kernel_v3(Dummy* result, double x) { *result = __double2ll_rn(x); }
__global__ void double2ll_ru_kernel_v1(long long int* result, double* x) { *result = __double2ll_ru(x); }
__global__ void double2ll_ru_kernel_v2(long long int* result, Dummy x) { *result = __double2ll_ru(x); }
__global__ void double2ll_ru_kernel_v3(Dummy* result, double x) { *result = __double2ll_ru(x); }
__global__ void double2ll_rz_kernel_v1(long long int* result, double* x) { *result = __double2ll_rz(x); }
__global__ void double2ll_rz_kernel_v2(long long int* result, Dummy x) { *result = __double2ll_rz(x); }
__global__ void double2ll_rz_kernel_v3(Dummy* result, double x) { *result = __double2ll_rz(x); }
)"};
static constexpr auto kDouble2ULL{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void double2ull_rd_kernel_v1(unsigned long long int* result, double* x) { *result = __double2ull_rd(x); }
__global__ void double2ull_rd_kernel_v2(unsigned long long int* result, Dummy x) { *result = __double2ull_rd(x); }
__global__ void double2ull_rd_kernel_v3(Dummy* result, double x) { *result = __double2ull_rd(x); }
__global__ void double2ull_rn_kernel_v1(unsigned long long int* result, double* x) { *result = __double2ull_rn(x); }
__global__ void double2ull_rn_kernel_v2(unsigned long long int* result, Dummy x) { *result = __double2ull_rn(x); }
__global__ void double2ull_rn_kernel_v3(Dummy* result, double x) { *result = __double2ull_rn(x); }
__global__ void double2ull_ru_kernel_v1(unsigned long long int* result, double* x) { *result = __double2ull_ru(x); }
__global__ void double2ull_ru_kernel_v2(unsigned long long int* result, Dummy x) { *result = __double2ull_ru(x); }
__global__ void double2ull_ru_kernel_v3(Dummy* result, double x) { *result = __double2ull_ru(x); }
__global__ void double2ull_rz_kernel_v1(unsigned long long int* result, double* x) { *result = __double2ull_rz(x); }
__global__ void double2ull_rz_kernel_v2(unsigned long long int* result, Dummy x) { *result = __double2ull_rz(x); }
__global__ void double2ull_rz_kernel_v3(Dummy* result, double x) { *result = __double2ull_rz(x); }
)"};
static constexpr auto kDouble2Float{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void double2float_rd_kernel_v1(float* result, double* x) { *result = __double2float_rd(x); }
__global__ void double2float_rd_kernel_v2(float* result, Dummy x) { *result = __double2float_rd(x); }
__global__ void double2float_rd_kernel_v3(Dummy* result, double x) { *result = __double2float_rd(x); }
__global__ void double2float_rn_kernel_v1(float* result, double* x) { *result = __double2float_rn(x); }
__global__ void double2float_rn_kernel_v2(float* result, Dummy x) { *result = __double2float_rn(x); }
__global__ void double2float_rn_kernel_v3(Dummy* result, double x) { *result = __double2float_rn(x); }
__global__ void double2float_ru_kernel_v1(float* result, double* x) { *result = __double2float_ru(x); }
__global__ void double2float_ru_kernel_v2(float* result, Dummy x) { *result = __double2float_ru(x); }
__global__ void double2float_ru_kernel_v3(Dummy* result, double x) { *result = __double2float_ru(x); }
__global__ void double2float_rz_kernel_v1(float* result, double* x) { *result = __double2float_rz(x); }
__global__ void double2float_rz_kernel_v2(float* result, Dummy x) { *result = __double2float_rz(x); }
__global__ void double2float_rz_kernel_v3(Dummy* result, double x) { *result = __double2float_rz(x); }
)"};
static constexpr auto kDouble2Hiint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void double2hiint_kernel_v1(int* result, double* x) { *result = __double2hiint(x); }
__global__ void double2hiint_kernel_v2(int* result, Dummy x) { *result = __double2hiint(x); }
__global__ void double2hiint_kernel_v3(Dummy* result, double x) { *result = __double2hiint(x); }
)"};
static constexpr auto kDouble2Loint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void double2loint_kernel_v1(int* result, double* x) { *result = __double2loint(x); }
__global__ void double2loint_kernel_v2(int* result, Dummy x) { *result = __double2loint(x); }
__global__ void double2loint_kernel_v3(Dummy* result, double x) { *result = __double2loint(x); }
)"};
static constexpr auto kDoubleAsLonglong{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void double_as_longlong_kernel_v1(long long int* result, double* x) { *result = __double_as_longlong(x); }
__global__ void double_as_longlong_kernel_v2(long long int* result, Dummy x) { *result = __double_as_longlong(x); }
__global__ void double_as_longlong_kernel_v3(Dummy* result, double x) { *result = __double_as_longlong(x); }
)"};
+440
Wyświetl plik
@@ -0,0 +1,440 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "casting_common.hh"
#include "casting_float_negative_kernels_rtc.hh"
/**
* @addtogroup CastingFloatType CastingFloatType
* @{
* @ingroup MathTest
*/
#define CAST_FLOAT2INT_TEST_DEF(kern_name, T, ref_func) \
CAST_KERNEL_DEF(kern_name, T, float) \
CAST_F2I_REF_DEF(kern_name, T, float, ref_func) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T (*ref)(float) = kern_name##_ref; \
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
std::numeric_limits<float>::lowest(), \
std::numeric_limits<float>::max()); \
}
#define CAST_FLOAT2INT_RZ_TEST_DEF(kern_name, T) \
CAST_KERNEL_DEF(kern_name, T, float) \
CAST_F2I_RZ_REF_DEF(kern_name, T, float) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T (*ref)(float) = kern_name##_ref; \
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
std::numeric_limits<float>::lowest(), \
std::numeric_limits<float>::max()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__float2int_rd` for all possible inputs. The results are compared against
* reference function `std::floor`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2INT_TEST_DEF(__float2int_rd, int, std::floor)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2int_rn` for all possible inputs. The results are compared against
* reference function `std::rint`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2INT_TEST_DEF(__float2int_rn, int, std::rint)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2int_ru` for all possible inputs. The results are compared against
* reference function `std::ceil`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2INT_TEST_DEF(__float2int_ru, int, std::ceil)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2int_rz` for all possible inputs. The results are compared against
* reference function `std::trunc`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2INT_TEST_DEF(__float2int_rz, int, std::trunc)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __float2int_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float2int_Negative_RTC") { NegativeTestRTCWrapper<12>(kFloat2Int); }
/**
* Test Description
* ------------------------
* - Tests that checks `__float2uint_rd` for all possible inputs. The results are compared
* against reference function `std::floor`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2INT_TEST_DEF(__float2uint_rd, unsigned int, std::floor)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2uint_rn` for all possible inputs. The results are compared
* against reference function `std::rint`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2INT_TEST_DEF(__float2uint_rn, unsigned int, std::rint)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2uint_ru` for all possible inputs. The results are compared
* against reference function `std::ceil`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2INT_TEST_DEF(__float2uint_ru, unsigned int, std::ceil)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2uint_rz` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* performs cast to unsigned int.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2INT_RZ_TEST_DEF(__float2uint_rz, unsigned int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __float2uint_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float2uint_Negative_RTC") { NegativeTestRTCWrapper<12>(kFloat2Uint); }
#define CAST_FLOAT2LL_TEST_DEF(kern_name, T, ref_func) \
CAST_KERNEL_DEF(kern_name, T, float) \
CAST_F2I_REF_DEF(kern_name, T, float, ref_func) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T (*ref)(float) = kern_name##_ref; \
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
static_cast<float>(std::numeric_limits<T>::min()), \
static_cast<float>(std::numeric_limits<T>::max())); \
}
#define CAST_FLOAT2LL_RZ_TEST_DEF(kern_name, T) \
CAST_KERNEL_DEF(kern_name, T, float) \
CAST_F2I_RZ_REF_DEF(kern_name, T, float) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T (*ref)(float) = kern_name##_ref; \
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>(), \
static_cast<float>(std::numeric_limits<T>::min()), \
static_cast<float>(std::numeric_limits<T>::max())); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__float2ll_rd` for all possible inputs. The results are compared against
* reference function `std::floor`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2LL_TEST_DEF(__float2ll_rd, long long int, std::floor)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2ll_rn` for all possible inputs between lowest and maximal long
* long int value. The results are compared against reference function `std::rint`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2LL_TEST_DEF(__float2ll_rn, long long int, std::rint)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2ll_ru` for all possible inputs between lowest and maximal long
* long int value. The results are compared against reference function `std::ceil`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2LL_TEST_DEF(__float2ll_ru, long long int, std::ceil)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2ll_rz` for all possible inputs between lowest and maximal long
* long int value. The results are compared against reference function which performs cast to long
* long int.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2LL_RZ_TEST_DEF(__float2ll_rz, long long int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __float2ll_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float2ll_Negative_RTC") { NegativeTestRTCWrapper<12>(kFloat2LL); }
/**
* Test Description
* ------------------------
* - Tests that checks `__float2ull_rd` for all possible inputs between lowest and maximal
* unsigned long long int value. The results are compared against reference function `std::floor`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2LL_TEST_DEF(__float2ull_rd, unsigned long long int, std::floor)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2ull_rn` for all possible inputs between lowest and maximal
* unsigned long long int value. The results are compared against reference function `std::rint`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2LL_TEST_DEF(__float2ull_rn, unsigned long long int, std::rint)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2ull_ru` for all possible inputs between lowest and maximal
* unsigned long long int value. The results are compared against reference function `std::ceil`.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2LL_TEST_DEF(__float2ull_ru, unsigned long long int, std::ceil)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2ll_rz` for all possible inputs between lowest and maximal
* unsigned long long int value. The results are compared against reference function which performs
* cast to unsigned long long int.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2LL_RZ_TEST_DEF(__float2ull_rz, unsigned long long int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __float2ull_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float2ull_Negative_RTC") { NegativeTestRTCWrapper<12>(kFloat2ULL); }
CAST_KERNEL_DEF(__float_as_int, int, float)
/**
* Test Description
* ------------------------
* - Tests that checks `__float_as_int` for all possible inputs. The results are compared against
* reference function which performs copy of float value to int variable.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float_as_int_Positive") {
int (*ref)(float) = type2_as_type1_ref<int, float>;
UnarySinglePrecisionTest(__float_as_int_kernel, ref, EqValidatorBuilderFactory<int>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __float_as_int.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float_as_int_Negative_RTC") { NegativeTestRTCWrapper<3>(kFloatAsInt); }
CAST_KERNEL_DEF(__float_as_uint, unsigned int, float)
/**
* Test Description
* ------------------------
* - Tests that checks `__float_as_uint` for all possible inputs. The results are compared
* against reference function which performs copy of float value to unsigned int variable.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float_as_uint_Positive") {
unsigned int (*ref)(float) = type2_as_type1_ref<unsigned int, float>;
UnarySinglePrecisionTest(__float_as_uint_kernel, ref, EqValidatorBuilderFactory<unsigned int>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __float_as_uint.
*
* Test source
* ------------------------
* - unit/math/casting_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float_as_uint_Negative_RTC") { NegativeTestRTCWrapper<3>(kFloatAsUint); }
@@ -0,0 +1,50 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL(func_name, T) \
__global__ void func_name##_kernel_v1(T* result, float* x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v2(T* result, Dummy x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v3(Dummy* result, float x) { *result = func_name(x); }
NEGATIVE_KERNELS_SHELL(__float2int_rd, int)
NEGATIVE_KERNELS_SHELL(__float2int_rn, int)
NEGATIVE_KERNELS_SHELL(__float2int_ru, int)
NEGATIVE_KERNELS_SHELL(__float2int_rz, int)
NEGATIVE_KERNELS_SHELL(__float2uint_rd, unsigned int)
NEGATIVE_KERNELS_SHELL(__float2uint_rn, unsigned int)
NEGATIVE_KERNELS_SHELL(__float2uint_ru, unsigned int)
NEGATIVE_KERNELS_SHELL(__float2uint_rz, unsigned int)
NEGATIVE_KERNELS_SHELL(__float2ll_rd, long long int)
NEGATIVE_KERNELS_SHELL(__float2ll_rn, long long int)
NEGATIVE_KERNELS_SHELL(__float2ll_ru, long long int)
NEGATIVE_KERNELS_SHELL(__float2ll_rz, long long int)
NEGATIVE_KERNELS_SHELL(__float2ull_rd, unsigned long long int)
NEGATIVE_KERNELS_SHELL(__float2ull_rn, unsigned long long int)
NEGATIVE_KERNELS_SHELL(__float2ull_ru, unsigned long long int)
NEGATIVE_KERNELS_SHELL(__float2ull_rz, unsigned long long int)
NEGATIVE_KERNELS_SHELL(__float_as_int, int)
NEGATIVE_KERNELS_SHELL(__float_as_uint, unsigned int)
@@ -0,0 +1,126 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
/*
Negative kernels used for the float type casting negative Test Cases that are using RTC.
*/
static constexpr auto kFloat2Int{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void float2int_rd_kernel_v1(int* result, float* x) { *result = __float2int_rd(x); }
__global__ void float2int_rd_kernel_v2(int* result, Dummy x) { *result = __float2int_rd(x); }
__global__ void float2int_rd_kernel_v3(Dummy* result, float x) { *result = __float2int_rd(x); }
__global__ void float2int_rn_kernel_v1(int* result, float* x) { *result = __float2int_rn(x); }
__global__ void float2int_rn_kernel_v2(int* result, Dummy x) { *result = __float2int_rn(x); }
__global__ void float2int_rn_kernel_v3(Dummy* result, float x) { *result = __float2int_rn(x); }
__global__ void float2int_ru_kernel_v1(int* result, float* x) { *result = __float2int_ru(x); }
__global__ void float2int_ru_kernel_v2(int* result, Dummy x) { *result = __float2int_ru(x); }
__global__ void float2int_ru_kernel_v3(Dummy* result, float x) { *result = __float2int_ru(x); }
__global__ void float2int_rz_kernel_v1(int* result, float* x) { *result = __float2int_rz(x); }
__global__ void float2int_rz_kernel_v2(int* result, Dummy x) { *result = __float2int_rz(x); }
__global__ void float2int_rz_kernel_v3(Dummy* result, float x) { *result = __float2int_rz(x); }
)"};
static constexpr auto kFloat2Uint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void float2uint_rd_kernel_v1(unsigned int* result, float* x) { *result = __float2uint_rd(x); }
__global__ void float2uint_rd_kernel_v2(unsigned int* result, Dummy x) { *result = __float2uint_rd(x); }
__global__ void float2uint_rd_kernel_v3(Dummy* result, float x) { *result = __float2uint_rd(x); }
__global__ void float2uint_rn_kernel_v1(unsigned int* result, float* x) { *result = __float2uint_rn(x); }
__global__ void float2uint_rn_kernel_v2(unsigned int* result, Dummy x) { *result = __float2uint_rn(x); }
__global__ void float2uint_rn_kernel_v3(Dummy* result, float x) { *result = __float2uint_rn(x); }
__global__ void float2uint_ru_kernel_v1(unsigned int* result, float* x) { *result = __float2uint_ru(x); }
__global__ void float2uint_ru_kernel_v2(unsigned int* result, Dummy x) { *result = __float2uint_ru(x); }
__global__ void float2uint_ru_kernel_v3(Dummy* result, float x) { *result = __float2uint_ru(x); }
__global__ void float2uint_rz_kernel_v1(unsigned int* result, float* x) { *result = __float2uint_rz(x); }
__global__ void float2uint_rz_kernel_v2(unsigned int* result, Dummy x) { *result = __float2uint_rz(x); }
__global__ void float2uint_rz_kernel_v3(Dummy* result, float x) { *result = __float2uint_rz(x); }
)"};
static constexpr auto kFloat2LL{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void float2ll_rd_kernel_v1(long long int* result, float* x) { *result = __float2ll_rd(x); }
__global__ void float2ll_rd_kernel_v2(long long int* result, Dummy x) { *result = __float2ll_rd(x); }
__global__ void float2ll_rd_kernel_v3(Dummy* result, float x) { *result = __float2ll_rd(x); }
__global__ void float2ll_rn_kernel_v1(long long int* result, float* x) { *result = __float2ll_rn(x); }
__global__ void float2ll_rn_kernel_v2(long long int* result, Dummy x) { *result = __float2ll_rn(x); }
__global__ void float2ll_rn_kernel_v3(Dummy* result, float x) { *result = __float2ll_rn(x); }
__global__ void float2ll_ru_kernel_v1(long long int* result, float* x) { *result = __float2ll_ru(x); }
__global__ void float2ll_ru_kernel_v2(long long int* result, Dummy x) { *result = __float2ll_ru(x); }
__global__ void float2ll_ru_kernel_v3(Dummy* result, float x) { *result = __float2ll_ru(x); }
__global__ void float2ll_rz_kernel_v1(long long int* result, float* x) { *result = __float2ll_rz(x); }
__global__ void float2ll_rz_kernel_v2(long long int* result, Dummy x) { *result = __float2ll_rz(x); }
__global__ void float2ll_rz_kernel_v3(Dummy* result, float x) { *result = __float2ll_rz(x); }
)"};
static constexpr auto kFloat2ULL{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void float2ull_rd_kernel_v1(unsigned long long int* result, float* x) { *result = __float2ull_rd(x); }
__global__ void float2ull_rd_kernel_v2(unsigned long long int* result, Dummy x) { *result = __float2ull_rd(x); }
__global__ void float2ull_rd_kernel_v3(Dummy* result, float x) { *result = __float2ull_rd(x); }
__global__ void float2ull_rn_kernel_v1(unsigned long long int* result, float* x) { *result = __float2ull_rn(x); }
__global__ void float2ull_rn_kernel_v2(unsigned long long int* result, Dummy x) { *result = __float2ull_rn(x); }
__global__ void float2ull_rn_kernel_v3(Dummy* result, float x) { *result = __float2ull_rn(x); }
__global__ void float2ull_ru_kernel_v1(unsigned long long int* result, float* x) { *result = __float2ull_ru(x); }
__global__ void float2ull_ru_kernel_v2(unsigned long long int* result, Dummy x) { *result = __float2ull_ru(x); }
__global__ void float2ull_ru_kernel_v3(Dummy* result, float x) { *result = __float2ull_ru(x); }
__global__ void float2ull_rz_kernel_v1(unsigned long long int* result, float* x) { *result = __float2ull_rz(x); }
__global__ void float2ull_rz_kernel_v2(unsigned long long int* result, Dummy x) { *result = __float2ull_rz(x); }
__global__ void float2ull_rz_kernel_v3(Dummy* result, float x) { *result = __float2ull_rz(x); }
)"};
static constexpr auto kFloatAsInt{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void float_as_int_kernel_v1(int* result, float* x) { *result = __float_as_int(x); }
__global__ void float_as_int_kernel_v2(int* result, Dummy x) { *result = __float_as_int(x); }
__global__ void float_as_int_kernel_v3(Dummy* result, float x) { *result = __float_as_int(x); }
)"};
static constexpr auto kFloatAsUint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void float_as_uint_kernel_v1(unsigned int* result, float* x) { *result = __float_as_uint(x); }
__global__ void float_as_uint_kernel_v2(unsigned int* result, Dummy x) { *result = __float_as_uint(x); }
__global__ void float_as_uint_kernel_v3(Dummy* result, float x) { *result = __float_as_uint(x); }
)"};
@@ -0,0 +1,97 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "math_common.hh"
#include "validators.hh"
namespace cg = cooperative_groups;
#define CAST_HALF2_KERNEL_DEF(func_name, T) \
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, Float16* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(__half2{xs[i], -xs[i]}); \
} \
}
#define CAST_BINARY_HALF2_KERNEL_DEF(func_name, T) \
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, Float16* const x1s, \
Float16* const x2s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(__half2{x1s[i], -x1s[i]}, __half2{x2s[i], -x2s[i]}); \
} \
}
template <typename VB> class Float2Validator : public MatcherBase<float2> {
public:
Float2Validator(const float2& target, const VB& vb)
: first_matcher_{vb(target.x)}, second_matcher_{vb(target.y)} {}
bool match(const float2& val) const override {
return first_matcher_->match(val.x) && second_matcher_->match(val.y);
}
std::string describe() const override {
return "<" + first_matcher_->describe() + ", " + second_matcher_->describe() + ">";
}
private:
decltype(std::declval<VB>()(float())) first_matcher_;
decltype(std::declval<VB>()(float())) second_matcher_;
};
template <typename ValidatorBuilder>
auto Float2ValidatorBuilderFactory(const ValidatorBuilder& vb) {
return [=](const float2& t, auto&&...) {
return std::make_unique<Float2Validator<ValidatorBuilder>>(t, vb);
};
}
template <typename VB> class Half2Validator : public MatcherBase<__half2> {
public:
Half2Validator(const __half2& target, const VB& vb)
: first_matcher_{vb(target.data.x)}, second_matcher_{vb(target.data.y)} {}
bool match(const __half2& val) const override {
return first_matcher_->match(val.data.x) && second_matcher_->match(val.data.y);
}
std::string describe() const override {
return "<" + first_matcher_->describe() + ", " + second_matcher_->describe() + ">";
}
private:
decltype(std::declval<VB>()(Float16())) first_matcher_;
decltype(std::declval<VB>()(Float16())) second_matcher_;
};
template <typename ValidatorBuilder> auto Half2ValidatorBuilderFactory(const ValidatorBuilder& vb) {
return [=](const __half2& t, auto&&...) {
return std::make_unique<Half2Validator<ValidatorBuilder>>(t, vb);
};
}
+419
Wyświetl plik
@@ -0,0 +1,419 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "half_precision_common.hh"
#include "casting_common.hh"
#include "casting_half2_common.hh"
/**
* @addtogroup HalfPrecisionCastingHalf2 HalfPrecisionCastingHalf2
* @{
* @ingroup MathTest
*/
/********** half -> half2 **********/
CAST_KERNEL_DEF(__half2half2, __half2, Float16)
static __half2 __half2half2_ref(Float16 x) { return __half2{x, x}; }
/**
* Test Description
* ------------------------
* - Tests that checks `__half2half2` for all possible inputs. The results are compared against
* reference function which returns __half2 value created from one __half value.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___half2half2_Accuracy_Positive") {
UnaryHalfPrecisionTest(__half2half2_kernel, __half2half2_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
CAST_BINARY_KERNEL_DEF(make_half2, __half2, Float16)
static __half2 make_half2_ref(Float16 x, Float16 y) { return __half2{x, y}; }
/**
* Test Description
* ------------------------
* - Tests that checks `make_half2` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* returns __half2 value created from two __half values.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_make_half2_Accuracy_Positive") {
BinaryFloatingPointTest(make_half2_kernel, make_half2_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
CAST_BINARY_KERNEL_DEF(__halves2half2, __half2, Float16)
static __half2 __halves2half2_ref(Float16 x, Float16 y) { return __half2{x, y}; }
/**
* Test Description
* ------------------------
* - Tests that checks `__halves2half2` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* returns __half2 value created from two __half values.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___halves2half2_Accuracy_Positive") {
BinaryFloatingPointTest(__halves2half2_kernel, __halves2half2_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
/********** half2 -> half **********/
CAST_HALF2_KERNEL_DEF(__low2half, Float16)
static Float16 __low2half_ref(Float16 x) { return x; }
/**
* Test Description
* ------------------------
* - Tests that checks `__low2half` for all possible inputs. The results are compared against
* reference function which returns __half value created from lower __half2 element.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___low2half_Accuracy_Positive") {
UnaryHalfPrecisionTest(__low2half_kernel, __low2half_ref, EqValidatorBuilderFactory<Float16>());
}
CAST_HALF2_KERNEL_DEF(__high2half, Float16)
static Float16 __high2half_ref(Float16 x) { return -x; }
/**
* Test Description
* ------------------------
* - Tests that checks `__high2half` for all possible inputs. The results are compared against
* reference function which returns __half value created from higher __half2 element.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___high2half_Accuracy_Positive") {
UnaryHalfPrecisionTest(__high2half_kernel, __high2half_ref, EqValidatorBuilderFactory<Float16>());
}
/********** half2 -> half2 **********/
CAST_HALF2_KERNEL_DEF(__low2half2, __half2)
static __half2 __low2half2_ref(Float16 x) { return __half2{x, x}; }
/**
* Test Description
* ------------------------
* - Tests that checks `__low2half2` for all possible inputs. The results are compared against
* reference function which returns __half2 value created from two lower __half2 elements.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___low2half2_Accuracy_Positive") {
UnaryHalfPrecisionTest(__low2half2_kernel, __low2half2_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
CAST_HALF2_KERNEL_DEF(__high2half2, __half2)
static __half2 __high2half2_ref(Float16 x) { return __half2{-x, -x}; }
/**
* Test Description
* ------------------------
* - Tests that checks `__high2half2` for all possible inputs. The results are compared against
* reference function which returns __half2 value created from two higher __half2 elements.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___high2half2_Accuracy_Positive") {
UnaryHalfPrecisionTest(__high2half2_kernel, __high2half2_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
CAST_HALF2_KERNEL_DEF(__lowhigh2highlow, __half2)
static __half2 __lowhigh2highlow_ref(Float16 x) { return __half2{-x, x}; }
/**
* Test Description
* ------------------------
* - Tests that checks `__lowhigh2highlow` for all possible inputs. The results are compared
* against reference function which returns __half2 value created from higher and lower __half2
* elements.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___lowhigh2highlow_Accuracy_Positive") {
UnaryHalfPrecisionTest(__lowhigh2highlow_kernel, __lowhigh2highlow_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
CAST_BINARY_HALF2_KERNEL_DEF(__lows2half2, __half2)
static __half2 __lows2half2_ref(Float16 x, Float16 y) { return __half2{x, y}; }
/**
* Test Description
* ------------------------
* - Tests that checks `__lows2half2` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* returns __half2 value created from lower elements of two __half2 values.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___lows2half2_Accuracy_Positive") {
BinaryFloatingPointTest(__lows2half2_kernel, __lows2half2_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
CAST_BINARY_HALF2_KERNEL_DEF(__highs2half2, __half2)
static __half2 __highs2half2_ref(Float16 x, Float16 y) { return __half2{-x, -y}; }
/**
* Test Description
* ------------------------
* - Tests that checks `__highs2half2` against a table of difficult values, followed by a large
* number of randomly generated values. The results are compared against reference function which
* returns __half2 value created from higher elements of two __half2 values.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___highs2half2_Accuracy_Positive") {
BinaryFloatingPointTest(__highs2half2_kernel, __highs2half2_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
/********** float -> half2 **********/
CAST_KERNEL_DEF(__float2half2_rn, __half2, float)
static __half2 __float2half2_rn_ref(float x) {
return __half2{static_cast<Float16>(x), static_cast<Float16>(x)};
}
/**
* Test Description
* ------------------------
* - Tests that checks `__float2half2_rn` for all possible inputs. The results are compared
* against reference function which returns __half2 value created from one casted float value.
* elements.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float2half2_rn_Accuracy_Positive") {
UnarySinglePrecisionTest(__float2half2_rn_kernel, __float2half2_rn_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
CAST_BINARY_KERNEL_DEF(__floats2half2_rn, __half2, float)
static __half2 __floats2half2_rn_ref(float x, float y) {
return __half2{static_cast<Float16>(x), static_cast<Float16>(y)};
}
/**
* Test Description
* ------------------------
* - Tests that checks `__floats2half2_rn` against a table of difficult values, followed by a
* large number of randomly generated values. The results are compared against reference function
* which returns __half2 value created from two casted float values.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___floats2half2_rn_Accuracy_Positive") {
BinaryFloatingPointTest(__floats2half2_rn_kernel, __floats2half2_rn_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
/********** float2 -> half2 **********/
__global__ void __float22half2_rn_kernel(__half2* const ys, const size_t num_xs, float* const xs) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
ys[i] = __float22half2_rn(make_float2(xs[i], -xs[i]));
}
}
static __half2 __float22half2_rn_ref(float x) {
return __half2{static_cast<Float16>(x), static_cast<Float16>(-x)};
}
/**
* Test Description
* ------------------------
* - Tests that checks `__float22half2_rn` for all possible inputs. The results are compared
* against reference function which returns __half2 value created from two casted float values.
* elements.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float22half2_rn_Accuracy_Positive") {
UnarySinglePrecisionTest(__float22half2_rn_kernel, __float22half2_rn_ref,
Half2ValidatorBuilderFactory(EqValidatorBuilderFactory<Float16>()));
}
/********** half2 -> float **********/
CAST_HALF2_KERNEL_DEF(__low2float, float)
static float __low2float_ref(Float16 x) { return static_cast<float>(x); }
/**
* Test Description
* ------------------------
* - Tests that checks `__low2float` for all possible inputs. The results are compared
* against reference function which returns float value created from lower __half2 element.
* elements.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___low2float_Accuracy_Positive") {
UnaryHalfPrecisionTest(__low2float_kernel, __low2float_ref, EqValidatorBuilderFactory<float>());
}
CAST_HALF2_KERNEL_DEF(__high2float, float)
static float __high2float_ref(Float16 x) { return static_cast<float>(-x); }
/**
* Test Description
* ------------------------
* - Tests that checks `__high2float` for all possible inputs. The results are compared
* against reference function which returns float value created from higher __half2 element.
* elements.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___high2float_Accuracy_Positive") {
UnaryHalfPrecisionTest(__high2float_kernel, __high2float_ref, EqValidatorBuilderFactory<float>());
}
/********** half2 -> float2 **********/
CAST_HALF2_KERNEL_DEF(__half22float2, float2)
static float2 __half22float2_ref(Float16 x) {
return make_float2(static_cast<float>(x), static_cast<float>(-x));
}
/**
* Test Description
* ------------------------
* - Tests that checks `__half22float2` for all possible inputs. The results are compared against
* reference function which returns float2 value created from casted elements of one __half2 value.
*
* Test source
* ------------------------
* - unit/math/casting_half2_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___half22float2_Accuracy_Positive") {
UnaryHalfPrecisionTest(__half22float2_kernel, __half22float2_ref,
Float2ValidatorBuilderFactory(EqValidatorBuilderFactory<float>()));
}
@@ -0,0 +1,57 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <hip/hip_fp16.h>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_UNARY_KERNELS_SHELL(func_name, T1, T2) \
__global__ void func_name##_kernel_v1(T1* result, T2* x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v2(T1* result, Dummy x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v3(Dummy* result, T2 x) { *result = func_name(x); }
#define NEGATIVE_BINARY_KERNELS_SHELL(func_name, T1, T2) \
__global__ void func_name##_kernel_v1(T2* x, T2 y) { T1 result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(T2 x, T2* y) { T1 result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, T2 y) { T1 result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(T2 x, Dummy y) { T1 result = func_name(x, y); }
NEGATIVE_UNARY_KERNELS_SHELL(__half2half2, __half2, __half)
NEGATIVE_UNARY_KERNELS_SHELL(__low2half, __half, __half2)
NEGATIVE_UNARY_KERNELS_SHELL(__high2half, __half, __half2)
NEGATIVE_UNARY_KERNELS_SHELL(__low2half2, __half2, __half2)
NEGATIVE_UNARY_KERNELS_SHELL(__high2half2, __half2, __half2)
NEGATIVE_UNARY_KERNELS_SHELL(__lowhigh2highlow, __half2, __half2)
NEGATIVE_UNARY_KERNELS_SHELL(__float2half2_rn, __half2, float)
NEGATIVE_UNARY_KERNELS_SHELL(__float22half2_rn, __half2, float2)
NEGATIVE_UNARY_KERNELS_SHELL(__low2float, float, __half2)
NEGATIVE_UNARY_KERNELS_SHELL(__high2float, float, __half2)
NEGATIVE_UNARY_KERNELS_SHELL(__half22float2, float2, __half2)
NEGATIVE_BINARY_KERNELS_SHELL(make_half2, __half2, __half)
NEGATIVE_BINARY_KERNELS_SHELL(__halves2half2, __half2, __half)
NEGATIVE_BINARY_KERNELS_SHELL(__lows2half2, __half2, __half2)
NEGATIVE_BINARY_KERNELS_SHELL(__highs2half2, __half2, __half2)
NEGATIVE_BINARY_KERNELS_SHELL(__floats2half2_rn, __half2, float)
@@ -0,0 +1,440 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "half_precision_common.hh"
#include "casting_common.hh"
/**
* @addtogroup HalfPrecisionCastingIntTypes HalfPrecisionCastingIntTypes
* @{
* @ingroup MathTest
*/
#define CAST_HALF2INT_RN_TEST_DEF(kern_name, T) \
CAST_KERNEL_DEF(kern_name, T, Float16) \
CAST_F2I_RZ_REF_DEF(kern_name, T, Float16) \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive") { \
T (*ref)(Float16) = kern_name##_ref; \
CastUnaryHalfPrecisionTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T>()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__half2int_rn` for all possible inputs. The results are compared against
* reference function which performs __half cast to int.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2int_rn, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2int_rz` for all possible inputs. The results are compared against
* reference function which performs __half cast to int.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2int_rz, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2int_rd` for all possible inputs. The results are compared against
* reference function which performs __half cast to int.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2int_rd, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2int_ru` for all possible inputs. The results are compared against
* reference function which performs __half cast to int.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2int_ru, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2uint_rn` for all possible inputs. The results are compared against
* reference function which performs __half cast to unsigned int.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2uint_rn, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2uint_rz` for all possible inputs. The results are compared against
* reference function which performs __half cast to unsigned int.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2uint_rz, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2uint_rd` for all possible inputs. The results are compared against
* reference function which performs __half cast to unsigned int.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2uint_rd, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2uint_ru` for all possible inputs. The results are compared against
* reference function which performs __half cast to unsigned int.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2uint_ru, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2short_rn` for all possible inputs. The results are compared
* against reference function which performs __half cast to short.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2short_rn, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2short_rz` for all possible inputs. The results are compared
* against reference function which performs __half cast to short.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2short_rz, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2short_rd` for all possible inputs. The results are compared
* against reference function which performs __half cast to short.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2short_rd, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2short_ru` for all possible inputs. The results are compared
* against reference function which performs __half cast to short.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2short_ru, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ushort_rn` for all possible inputs. The results are compared
* against reference function which performs __half cast to unsigned short.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ushort_rn, unsigned short)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ushort_rz` for all possible inputs. The results are compared
* against reference function which performs __half cast to unsigned short.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ushort_rz, unsigned short)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ushort_rd` for all possible inputs. The results are compared
* against reference function which performs __half cast to unsigned short.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ushort_rd, unsigned short)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ushort_ru` for all possible inputs. The results are compared
* against reference function which performs __half cast to unsigned short.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ushort_ru, unsigned short)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ll_rn` for all possible inputs. The results are compared against
* reference function which performs __half cast to long long.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ll_rn, long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ll_rz` for all possible inputs. The results are compared against
* reference function which performs __half cast to long long.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ll_rz, long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ll_rd` for all possible inputs. The results are compared against
* reference function which performs __half cast to long long.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ll_rd, long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ll_ru` for all possible inputs. The results are compared against
* reference function which performs __half cast to long long.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ll_ru, long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ull_rn` for all possible inputs. The results are compared against
* reference function which performs __half cast to unsigned long long.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ull_rn, unsigned long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ull_rz` for all possible inputs. The results are compared against
* reference function which performs __half cast to unsigned long long.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ull_rz, unsigned long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ull_rd` for all possible inputs. The results are compared against
* reference function which performs __half cast to unsigned long long.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ull_rd, unsigned long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2ull_ru` for all possible inputs. The results are compared against
* reference function which performs __half cast to unsigned long long.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_HALF2INT_RN_TEST_DEF(__half2ull_ru, unsigned long long)
CAST_KERNEL_DEF(__half_as_short, short, Float16)
/**
* Test Description
* ------------------------
* - Tests that checks `__half_as_short` for all possible inputs. The results are compared
* against reference function which performs copy of __half value to short variable.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___half_as_short_Accuracy_Positive") {
short (*ref)(Float16) = type2_as_type1_ref<short, Float16>;
CastUnaryHalfPrecisionTest(__half_as_short_kernel, ref, EqValidatorBuilderFactory<short>());
}
CAST_KERNEL_DEF(__half_as_ushort, unsigned short, Float16)
/**
* Test Description
* ------------------------
* - Tests that checks `__half_as_ushort` for all possible inputs. The results are compared
* against reference function which performs copy of __half value to unsigned short variable.
*
* Test source
* ------------------------
* - unit/math/casting_half2int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___half_as_ushort_Accuracy_Positive") {
unsigned short (*ref)(Float16) = type2_as_type1_ref<unsigned short, Float16>;
CastUnaryHalfPrecisionTest(__half_as_ushort_kernel, ref,
EqValidatorBuilderFactory<unsigned short>());
}
@@ -0,0 +1,59 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <hip/hip_fp16.h>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL(func_name, T) \
__global__ void func_name##_kernel_v1(T* result, __half* x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v2(T* result, Dummy x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v3(Dummy* result, __half x) { *result = unc_name(x); }
NEGATIVE_KERNELS_SHELL(__half2int_rn, int)
NEGATIVE_KERNELS_SHELL(__half2int_rz, int)
NEGATIVE_KERNELS_SHELL(__half2int_rd, int)
NEGATIVE_KERNELS_SHELL(__half2int_ru, int)
NEGATIVE_KERNELS_SHELL(__half2uint_rn, unsigned int)
NEGATIVE_KERNELS_SHELL(__half2uint_rz, unsigned int)
NEGATIVE_KERNELS_SHELL(__half2uint_rd, unsigned int)
NEGATIVE_KERNELS_SHELL(__half2uint_ru, unsigned int)
NEGATIVE_KERNELS_SHELL(__half2short_rn, short)
NEGATIVE_KERNELS_SHELL(__half2short_rz, short)
NEGATIVE_KERNELS_SHELL(__half2short_rd, short)
NEGATIVE_KERNELS_SHELL(__half2short_ru, short)
NEGATIVE_KERNELS_SHELL(__half_as_short, short)
NEGATIVE_KERNELS_SHELL(__half2ushort_rn, unsigned short)
NEGATIVE_KERNELS_SHELL(__half2ushort_rz, unsigned short)
NEGATIVE_KERNELS_SHELL(__half2ushort_rd, unsigned short)
NEGATIVE_KERNELS_SHELL(__half2ushort_ru, unsigned short)
NEGATIVE_KERNELS_SHELL(__half_as_ushort, unsigned short)
NEGATIVE_KERNELS_SHELL(__half2ll_rn, long long)
NEGATIVE_KERNELS_SHELL(__half2ll_rz, long long)
NEGATIVE_KERNELS_SHELL(__half2ll_rd, long long)
NEGATIVE_KERNELS_SHELL(__half2ll_ru, long long)
NEGATIVE_KERNELS_SHELL(__half2ull_rn, unsigned long long)
NEGATIVE_KERNELS_SHELL(__half2ull_rz, unsigned long long)
NEGATIVE_KERNELS_SHELL(__half2ull_rd, unsigned long long)
NEGATIVE_KERNELS_SHELL(__half2ull_ru, unsigned long long)
@@ -0,0 +1,247 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "half_precision_common.hh"
#include "casting_common.hh"
/**
* @addtogroup HalfPrecisionCastingFloat HalfPrecisionCastingFloat
* @{
* @ingroup MathTest
*/
#define CAST_FLOAT2HALF_TEST_DEF(kern_name, round_dir) \
CAST_KERNEL_DEF(kern_name, Float16, float) \
CAST_RND_REF_DEF(kern_name, Float16, float, round_dir) \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Limited_Positive") { \
Float16 (*ref)(float) = kern_name##_ref; \
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>(), \
std::numeric_limits<float>::min(), 0.f); \
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>(), \
0.0001f, std::numeric_limits<float>::max()); \
}
#define CAST_FLOAT2HALF_RN_TEST_DEF(kern_name) \
CAST_KERNEL_DEF(kern_name, Float16, float) \
CAST_REF_DEF(kern_name, Float16, float) \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive") { \
Float16 (*ref)(float) = kern_name##_ref; \
UnarySinglePrecisionRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>(), \
std::numeric_limits<float>::min(), \
std::numeric_limits<float>::max()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__float2half_rd` for all possible inputs apart from very small positive
* values. Rounding behaviour is not correct for host functions for this range. The results are
* compared against reference function which performs float cast to __half with FE_DOWNWARD rounding
* mode.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2HALF_TEST_DEF(__float2half_rd, FE_DOWNWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2half_rn` for all possible inputs. The results are compared against
* reference function which performs float cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2HALF_RN_TEST_DEF(__float2half_rn)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2half` for all possible inputs. The results are compared against
* reference function which performs float cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2HALF_RN_TEST_DEF(__float2half)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2half_ru` for all possible inputs apart from very small positive
* values. Rounding behaviour is not correct for host functions for this range. The results are
* compared against reference function which performs float cast to __half with FE_UPWARD rounding
* mode.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2HALF_TEST_DEF(__float2half_ru, FE_UPWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__float2half_rz` for all possible inputs apart from very small positive
* values. Rounding behaviour is not correct for host functions for this range. The results are
* compared against reference function which performs float cast to __half with FE_TOWARDZERO rounding
* mode.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_FLOAT2HALF_TEST_DEF(__float2half_rz, FE_TOWARDZERO)
/**
* Test Description
* ------------------------
* - Sanity test that checks `__float2half_rd` for very small positive values.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float2half_rd_SmallVals_Sanity_Positive") {
const float input[] = {0.8859e-06f, 1.5454e-07f, 6.5955e-08f, 2.7955e-08f,
3.7956e-09f, 4.8995e-10f, 5.7997e-15f, 6.2117e-20f,
7.4999e-25f, 8.9999e-30f, 9.0001e-35f};
const Float16 reference[] = {8.34465e-07, 1.19209e-07, 5.96046e-08, 0, 0, 0, 0, 0, 0, 0, 0};
LinearAllocGuard<float> input_dev{LinearAllocs::hipMalloc, sizeof(float)};
LinearAllocGuard<Float16> out(LinearAllocs::hipMallocManaged, sizeof(Float16));
for (int i = 0; i < 11; ++i) {
HIP_CHECK(hipMemcpy(input_dev.ptr(), input + i, sizeof(float), hipMemcpyHostToDevice));
__float2half_rd_kernel<<<1, 1>>>(out.ptr(), 1, input_dev.ptr());
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(out.ptr()[0] == reference[i]);
}
}
/**
* Test Description
* ------------------------
* - Sanity test that checks `__float2half_ru` for very small positive values.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float2half_ru_SmallVals_Sanity_Positive") {
const float input[] = {0.8859e-06f, 1.5454e-07f, 6.5955e-08f, 2.7955e-08f,
3.7956e-09f, 4.8995e-10f, 5.7997e-15f, 6.2117e-20f,
7.4999e-25f, 8.9999e-30f, 9.0001e-35f};
const Float16 reference[] = {8.9407e-07, 1.78814e-07, 1.19209e-07, 5.96046e-08,
5.96046e-08, 5.96046e-08, 5.96046e-08, 5.96046e-08,
5.96046e-08, 5.96046e-08, 5.96046e-08};
LinearAllocGuard<float> input_dev{LinearAllocs::hipMalloc, sizeof(float)};
LinearAllocGuard<Float16> out(LinearAllocs::hipMallocManaged, sizeof(Float16));
for (int i = 0; i < 11; ++i) {
HIP_CHECK(hipMemcpy(input_dev.ptr(), input + i, sizeof(float), hipMemcpyHostToDevice));
__float2half_ru_kernel<<<1, 1>>>(out.ptr(), 1, input_dev.ptr());
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(out.ptr()[0] == reference[i]);
}
}
/**
* Test Description
* ------------------------
* - Sanity test that checks `__float2half_rz` for very small positive values.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___float2half_rz_SmallVals_Sanity_Positive") {
const float input[] = {0.8859e-06f, 1.5454e-07f, 6.5955e-08f, 2.7955e-08f,
3.7956e-09f, 4.8995e-10f, 5.7997e-15f, 6.2117e-20f,
7.4999e-25f, 8.9999e-30f, 9.0001e-35f};
const Float16 reference[] = {8.34465e-07, 1.19209e-07, 5.96046e-08, 0, 0, 0, 0, 0, 0, 0, 0};
LinearAllocGuard<float> input_dev{LinearAllocs::hipMalloc, sizeof(float)};
LinearAllocGuard<Float16> out(LinearAllocs::hipMallocManaged, sizeof(Float16));
for (int i = 0; i < 11; ++i) {
HIP_CHECK(hipMemcpy(input_dev.ptr(), input + i, sizeof(float), hipMemcpyHostToDevice));
__float2half_rz_kernel<<<1, 1>>>(out.ptr(), 1, input_dev.ptr());
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(out.ptr()[0] == reference[i]);
}
}
CAST_KERNEL_DEF(__half2float, float, Float16)
CAST_REF_DEF(__half2float, float, Float16)
/**
* Test Description
* ------------------------
* - Tests that checks `__half2float` for all possible inputs. The results are compared against
* reference function which performs __half cast to float.
*
* Test source
* ------------------------
* - unit/math/casting_half_float_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___half2float_Accuracy_Positive") {
float (*ref)(Float16) = __half2float_ref;
UnaryHalfPrecisionTest(__half2float_kernel, ref, EqValidatorBuilderFactory<float>());
}
@@ -0,0 +1,45 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <hip/hip_fp16.h>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_F2H_KERNELS_SHELL(func_name) \
__global__ void func_name##_kernel_v1(__half* result, float* x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v2(__half* result, Dummy x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v3(Dummy* result, float x) { *result = func_name(x); }
#define NEGATIVE_H2F_KERNELS_SHELL(func_name) \
__global__ void func_name##_kernel_v1(float* result, __half* x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v2(float* result, Dummy x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v3(Dummy* result, __half x) { *result = func_name(x); }
NEGATIVE_F2H_KERNELS_SHELL(__float2half_rd)
NEGATIVE_F2H_KERNELS_SHELL(__float2half_rn)
NEGATIVE_F2H_KERNELS_SHELL(__float2half_ru)
NEGATIVE_F2H_KERNELS_SHELL(__float2half_rz)
NEGATIVE_F2H_KERNELS_SHELL(__float2half)
NEGATIVE_H2F_KERNELS_SHELL(__half2float)
@@ -0,0 +1,448 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "half_precision_common.hh"
#include "casting_common.hh"
/**
* @addtogroup HalfPrecisionCastingIntTypes HalfPrecisionCastingIntTypes
* @{
* @ingroup MathTest
*/
#define CAST_INT2HALF_RN_TEST_DEF(kern_name, T) \
CAST_KERNEL_DEF(kern_name, Float16, T) \
CAST_REF_DEF(kern_name, Float16, T) \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive") { \
Float16 (*ref)(T) = kern_name##_ref; \
CastIntRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__int2half_rn` for all possible inputs. The results are compared against
* reference function which performs int cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__int2half_rn, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__int2half_rz` for all possible inputs. The results are compared against
* reference function which performs int cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__int2half_rz, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__int2half_rd` for all possible inputs. The results are compared against
* reference function which performs int cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__int2half_rd, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__int2half_ru` for all possible inputs. The results are compared against
* reference function which performs int cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__int2half_ru, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2half_rn` for all possible inputs. The results are compared against
* reference function which performs unsigned int cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__uint2half_rn, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2half_rz` for all possible inputs. The results are compared against
* reference function which performs unsigned int cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__uint2half_rz, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2half_rd` for all possible inputs. The results are compared against
* reference function which performs unsigned int cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__uint2half_rd, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2half_ru` for all possible inputs. The results are compared against
* reference function which performs unsigned int cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__uint2half_ru, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__short2half_rn` for all possible inputs. The results are compared
* against reference function which performs short cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__short2half_rn, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__short2half_rz` for all possible inputs. The results are compared
* against reference function which performs short cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__short2half_rz, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__short2half_rd` for all possible inputs. The results are compared
* against reference function which performs short cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__short2half_rd, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__short2half_ru` for all possible inputs. The results are compared
* against reference function which performs short cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__short2half_ru, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__ushort2half_rn` for all possible inputs. The results are compared
* against reference function which performs unsigned short cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__ushort2half_rn, unsigned short)
/**
* Test Description
* ------------------------
* - Tests that checks `__ushort2half_rz` for all possible inputs. The results are compared
* against reference function which performs unsigned short cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__ushort2half_rz, unsigned short)
/**
* Test Description
* ------------------------
* - Tests that checks `__ushort2half_rd` for all possible inputs. The results are compared
* against reference function which performs unsigned short cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__ushort2half_rd, unsigned short)
/**
* Test Description
* ------------------------
* - Tests that checks `__ushort2half_ru` for all possible inputs. The results are compared
* against reference function which performs unsigned short cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2HALF_RN_TEST_DEF(__ushort2half_ru, unsigned short)
#define CAST_LL2HALF_TEST_DEF(kern_name, T) \
CAST_KERNEL_DEF(kern_name, Float16, T) \
CAST_REF_DEF(kern_name, Float16, T) \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive") { \
Float16 (*ref)(T) = kern_name##_ref; \
CastIntBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<Float16>()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2half_rn` against a large number of randomly generated values. The
* results are compared against reference function which performs long long cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2HALF_TEST_DEF(__ll2half_rn, long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2half_rz` against a large number of randomly generated values. The
* results are compared against reference function which performs long long cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2HALF_TEST_DEF(__ll2half_rz, long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2half_rd` against a large number of randomly generated values. The
* results are compared against reference function which performs long long cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2HALF_TEST_DEF(__ll2half_rd, long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2half_ru` against a large number of randomly generated values. The
* results are compared against reference function which performs long long cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2HALF_TEST_DEF(__ll2half_ru, long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2half_rn` against a large number of randomly generated values. The
* results are compared against reference function which performs unsigned long long cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2HALF_TEST_DEF(__ull2half_rn, unsigned long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2half_rz` against a large number of randomly generated values. The
* results are compared against reference function which performs unsigned long long cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2HALF_TEST_DEF(__ull2half_rz, unsigned long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2half_rd` against a large number of randomly generated values. The
* results are compared against reference function which performs unsigned long long cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2HALF_TEST_DEF(__ull2half_rd, unsigned long long)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2half_ru` against a large number of randomly generated values. The
* results are compared against reference function which performs unsigned long long cast to __half.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2HALF_TEST_DEF(__ull2half_ru, unsigned long long)
CAST_KERNEL_DEF(__short_as_half, Float16, short)
/**
* Test Description
* ------------------------
* - Tests that checks `__short_as_half` for all possible inputs. The results are compared
* against reference function which performs copy of short value to __half variable.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___short_as_half_Accuracy_Positive") {
Float16 (*ref)(short) = type2_as_type1_ref<Float16, short>;
CastIntBruteForceTest(__short_as_half_kernel, ref, EqValidatorBuilderFactory<Float16>());
}
CAST_KERNEL_DEF(__ushort_as_half, Float16, unsigned short)
/**
* Test Description
* ------------------------
* - Tests that checks `__ushort_as_half` for all possible inputs. The results are compared
* against reference function which performs copy of unsigned short value to __half variable.
*
* Test source
* ------------------------
* - unit/math/casting_int2half_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___ushort_as_half_Accuracy_Positive") {
Float16 (*ref)(unsigned short) = type2_as_type1_ref<Float16, unsigned short>;
CastIntBruteForceTest(__ushort_as_half_kernel, ref, EqValidatorBuilderFactory<Float16>());
}
@@ -0,0 +1,59 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <hip/hip_fp16.h>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL(func_name, T) \
__global__ void func_name##_kernel_v1(__half* result, T* x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v2(__half* result, Dummy x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v3(Dummy* result, T x) { *result = func_name(x); }
NEGATIVE_KERNELS_SHELL(__int2half_rn, int)
NEGATIVE_KERNELS_SHELL(__int2half_rz, int)
NEGATIVE_KERNELS_SHELL(__int2half_rd, int)
NEGATIVE_KERNELS_SHELL(__int2half_ru, int)
NEGATIVE_KERNELS_SHELL(__uint2half_rn, unsigned int)
NEGATIVE_KERNELS_SHELL(__uint2half_rz, unsigned int)
NEGATIVE_KERNELS_SHELL(__uint2half_rd, unsigned int)
NEGATIVE_KERNELS_SHELL(__uint2half_ru, unsigned int)
NEGATIVE_KERNELS_SHELL(__short2half_rn, short)
NEGATIVE_KERNELS_SHELL(__short2half_rz, short)
NEGATIVE_KERNELS_SHELL(__short2half_rd, short)
NEGATIVE_KERNELS_SHELL(__short2half_ru, short)
NEGATIVE_KERNELS_SHELL(__short_as_half, short)
NEGATIVE_KERNELS_SHELL(__ushort2half_rn, unsigned short)
NEGATIVE_KERNELS_SHELL(__ushort2half_rz, unsigned short)
NEGATIVE_KERNELS_SHELL(__ushort2half_rd, unsigned short)
NEGATIVE_KERNELS_SHELL(__ushort2half_ru, unsigned short)
NEGATIVE_KERNELS_SHELL(__ushort_as_half, unsigned short)
NEGATIVE_KERNELS_SHELL(__ll2half_rn, long long)
NEGATIVE_KERNELS_SHELL(__ll2half_rz, long long)
NEGATIVE_KERNELS_SHELL(__ll2half_rd, long long)
NEGATIVE_KERNELS_SHELL(__ll2half_ru, long long)
NEGATIVE_KERNELS_SHELL(__ull2half_rn, unsigned long long)
NEGATIVE_KERNELS_SHELL(__ull2half_rz, unsigned long long)
NEGATIVE_KERNELS_SHELL(__ull2half_rd, unsigned long long)
NEGATIVE_KERNELS_SHELL(__ull2half_ru, unsigned long long)
+735
Wyświetl plik
@@ -0,0 +1,735 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "casting_common.hh"
#include "casting_int_negative_kernels_rtc.hh"
/**
* @addtogroup CastingIntTypes CastingIntTypes
* @{
* @ingroup MathTest
*/
#define CAST_INT2FLOAT_TEST_DEF(kern_name, T1, T2, round_dir) \
CAST_KERNEL_DEF(kern_name, T1, T2) \
CAST_RND_REF_DEF(kern_name, T1, T2, round_dir) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T1 (*ref)(T2) = kern_name##_ref; \
CastIntRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T1>()); \
}
#define CAST_INT2FLOAT_RN_TEST_DEF(kern_name, T1, T2) \
CAST_KERNEL_DEF(kern_name, T1, T2) \
CAST_REF_DEF(kern_name, T1, T2) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T1 (*ref)(T2) = kern_name##_ref; \
CastIntRangeTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T1>()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__int2float_rd` for all possible inputs. The results are compared against
* reference function which performs cast to float with FE_DOWNWARD rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_TEST_DEF(__int2float_rd, float, int, FE_DOWNWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__int2float_rn` for all possible inputs. The results are compared against
* reference function which performs cast to float.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_RN_TEST_DEF(__int2float_rn, float, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__int2float_ru` for all possible inputs. The results are compared against
* reference function which performs cast to float with FE_UPWARD rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_TEST_DEF(__int2float_ru, float, int, FE_UPWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__int2float_rz` for all possible inputs. The results are compared against
* reference function which performs cast to float with FE_TOWARDZERO rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_TEST_DEF(__int2float_rz, float, int, FE_TOWARDZERO)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __int2float_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_int2float___Negative_RTC") { NegativeTestRTCWrapper<12>(kInt2Float); }
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2float_rd` for all possible inputs. The results are compared
* against reference function which performs cast to float with FE_DOWNWARD rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_TEST_DEF(__uint2float_rd, float, unsigned int, FE_DOWNWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2float_rn` for all possible inputs. The results are compared
* against reference function which performs cast to float.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_RN_TEST_DEF(__uint2float_rn, float, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2float_ru` for all possible inputs. The results are compared
* against reference function which performs cast to float with FE_UPWARD rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_TEST_DEF(__uint2float_ru, float, unsigned int, FE_UPWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2float_rz` for all possible inputs. The results are compared
* against reference function which performs cast to float with FE_TOWARDZERO rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_TEST_DEF(__uint2float_rz, float, unsigned int, FE_TOWARDZERO)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __uint2float_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___uint2float_Negative_RTC") { NegativeTestRTCWrapper<12>(kUint2Float); }
/**
* Test Description
* ------------------------
* - Tests that checks `__int2double_rn` for all possible inputs. The results are compared
* against reference function which performs cast to double.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_RN_TEST_DEF(__int2double_rn, double, int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __int2double_rn.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___int2double_Negative_RTC") { NegativeTestRTCWrapper<3>(kInt2Double); }
/**
* Test Description
* ------------------------
* - Tests that checks `__uint2double_rn` for all possible inputs. The results are compared
* against reference function which performs cast to double.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_INT2FLOAT_RN_TEST_DEF(__uint2double_rn, double, unsigned int)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __uint2double_rn.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___uint2double_Negative_RTC") { NegativeTestRTCWrapper<3>(kUint2Double); }
#define CAST_LL2FLOAT_TEST_DEF(kern_name, T1, T2, round_dir) \
CAST_KERNEL_DEF(kern_name, T1, T2) \
CAST_RND_REF_DEF(kern_name, T1, T2, round_dir) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T1 (*ref)(T2) = kern_name##_ref; \
CastIntBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T1>()); \
}
#define CAST_LL2FLOAT_RN_TEST_DEF(kern_name, T1, T2) \
CAST_KERNEL_DEF(kern_name, T1, T2) \
CAST_REF_DEF(kern_name, T1, T2) \
\
TEST_CASE("Unit_Device_" #kern_name "_Positive") { \
T1 (*ref)(T2) = kern_name##_ref; \
CastIntBruteForceTest(kern_name##_kernel, ref, EqValidatorBuilderFactory<T1>()); \
}
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2float_rd` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to float with FE_DOWNWARD
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ll2float_rd, float, long long int, FE_DOWNWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2float_rn` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to float.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_RN_TEST_DEF(__ll2float_rn, float, long long int)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2float_ru` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to float with FE_UPWARD
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ll2float_ru, float, long long int, FE_UPWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2float_rz` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to float with FE_TOWARDZERO
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ll2float_rz, float, long long int, FE_TOWARDZERO)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __ll2float_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___ll2float_Negative_RTC") { NegativeTestRTCWrapper<12>(kLL2Float); }
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2float_rd` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to float with FE_DOWNWARD
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ull2float_rd, float, unsigned long long int, FE_DOWNWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2float_rn` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to float.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_RN_TEST_DEF(__ull2float_rn, float, unsigned long long int)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2float_ru` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to float with FE_UPWARD
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ull2float_ru, float, unsigned long long int, FE_UPWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2float_rz` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to float with FE_TOWARDZERO
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ull2float_rz, float, unsigned long long int, FE_TOWARDZERO)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __ull2float_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___ull2float_Negative_RTC") { NegativeTestRTCWrapper<12>(kULL2Float); }
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2double_rd` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to double with FE_DOWNWARD
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ll2double_rd, double, long long int, FE_DOWNWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2double_rn` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to double.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_RN_TEST_DEF(__ll2double_rn, double, long long int)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2double_ru` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to double with FE_UPWARD
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ll2double_ru, double, long long int, FE_UPWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__ll2double_rz` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to double with FE_TOWARDZERO
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ll2double_rz, double, long long int, FE_TOWARDZERO)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __ll2double_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___ll2double_Negative_RTC") { NegativeTestRTCWrapper<12>(kLL2Double); }
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2double_rd` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to double with FE_DOWNWARD
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ull2double_rd, double, unsigned long long int, FE_DOWNWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2double_rn` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to double.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_RN_TEST_DEF(__ull2double_rn, double, unsigned long long int)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2double_ru` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to double with FE_UPWARD
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ull2double_ru, double, unsigned long long int, FE_UPWARD)
/**
* Test Description
* ------------------------
* - Tests that checks `__ull2double_rz` against a large number of randomly generated values. The
* results are compared against reference function which performs cast to double with FE_TOWARDZERO
* rounding mode.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
CAST_LL2FLOAT_TEST_DEF(__ull2double_rz, double, unsigned long long int, FE_TOWARDZERO)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __ull2double_[rd,rn,ru,rz].
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___ull2double_Negative_RTC") { NegativeTestRTCWrapper<12>(kULL2Double); }
CAST_KERNEL_DEF(__int_as_float, float, int)
/**
* Test Description
* ------------------------
* - Tests that checks `__int_as_float` for all possible inputs. The results are compared against
* reference function which performs copy of int value to float variable.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___int_as_float_Positive") {
float (*ref)(int) = type2_as_type1_ref<float, int>;
CastIntRangeTest(__int_as_float_kernel, ref, EqValidatorBuilderFactory<float>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __int_as_float.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___int_as_float_Negative_RTC") { NegativeTestRTCWrapper<3>(kIntAsFloat); }
CAST_KERNEL_DEF(__uint_as_float, float, unsigned int)
/**
* Test Description
* ------------------------
* - Tests that checks `__uint_as_float` for all possible inputs. The results are compared
* against reference function which performs copy of unsigned int value to float variable.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___uint_as_float_Positive") {
float (*ref)(unsigned int) = type2_as_type1_ref<float, unsigned int>;
CastIntRangeTest(__uint_as_float_kernel, ref, EqValidatorBuilderFactory<float>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __uint_as_float.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___uint_as_float_Negative_RTC") { NegativeTestRTCWrapper<3>(kUintAsFloat); }
CAST_KERNEL_DEF(__longlong_as_double, double, long long int)
/**
* Test Description
* ------------------------
* - Tests that checks `__longlong_as_double` against a large number of randomly generated
* values. The results are compared against reference function which performs copy of long long int
* value to double variable.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___longlong_as_double_Positive") {
double (*ref)(long long int) = type2_as_type1_ref<double, long long int>;
CastIntBruteForceTest(__longlong_as_double_kernel, ref, EqValidatorBuilderFactory<double>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __longlong_as_double.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___longlong_as_double_Negative_RTC") {
NegativeTestRTCWrapper<3>(kLonglongAsDouble);
}
__global__ void __hiloint2double_kernel(double* const ys, const size_t num_xs, int* const x1s,
int* const x2s) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
ys[i] = __hiloint2double(x1s[i], x2s[i]);
}
}
double __hiloint2double_ref(int hi, int lo) {
uint64_t tmp0 = (static_cast<uint64_t>(hi) << 32ull) | static_cast<uint32_t>(lo);
double tmp1;
memcpy(&tmp1, &tmp0, sizeof(tmp0));
return tmp1;
}
/**
* Test Description
* ------------------------
* - Tests that checks `__hiloint2double` for all possible inputs for hi value. The results are
* compared against reference function which performs copy of hi int value to higher part of double
* variable and copy of lo int value to lower part of double variable.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___hiloint2double_Positive") {
double (*ref)(int, int) = __hiloint2double_ref;
CastBinaryIntRangeTest(__hiloint2double_kernel, ref, EqValidatorBuilderFactory<double>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for __hiloint2double.
*
* Test source
* ------------------------
* - unit/math/casting_int_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___hiloint2double_Negative_RTC") { NegativeTestRTCWrapper<5>(kHilo2Double); }
@@ -0,0 +1,79 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL_ONE_ARG(func_name, T1, T2) \
__global__ void func_name##_kernel_v1(T1* result, T2* x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v2(T1* result, Dummy x) { *result = func_name(x); } \
__global__ void func_name##_kernel_v3(Dummy* result, T2 x) { *result = func_name(x); }
#define NEGATIVE_KERNELS_SHELL_TWO_ARGS(func_name, T1, T2) \
__global__ void func_name##_kernel_v1(T1* result, T2* x, T2 y) { \
*result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v2(T1* result, T2 x, T2* y) { \
*result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v3(T1* result, Dummy x, T2 y) { \
*result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v4(T1* result, T2 x, Dummy y) { \
*result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v5(Dummy* result, T2 x, T2 y) { \
*result = func_name(x, y); \
}
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2float_rd, float, int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2float_rn, float, int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2float_ru, float, int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2float_rz, float, int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2float_rd, float, unsigned int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2float_rn, float, unsigned int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2float_ru, float, unsigned int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2float_rz, float, unsigned int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2float_rd, float, long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2float_rn, float, long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2float_ru, float, long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2float_rz, float, long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2float_rd, float, unsigned long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2float_rn, float, unsigned long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2float_ru, float, unsigned long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2float_rz, float, unsigned long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int2double_rn, double, int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint2double_rn, double, unsigned int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2double_rd, double, long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2double_rn, double, long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2double_ru, double, long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ll2double_rz, double, long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2double_rd, double, unsigned long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2double_rn, double, unsigned long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2double_ru, double, unsigned long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__ull2double_rz, double, unsigned long long int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__int_as_float, float, int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__uint_as_float, float, unsigned int)
NEGATIVE_KERNELS_SHELL_ONE_ARG(__longlong_as_double, double, long long int)
NEGATIVE_KERNELS_SHELL_TWO_ARGS(__hiloint2double, double, int)
@@ -0,0 +1,215 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
/*
Negative kernels used for the <unsigned> int/long long type casting negative Test Cases that are using RTC.
*/
static constexpr auto kInt2Float{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void int2float_rd_kernel_v1(float* result, int* x) { *result = __int2float_rd(x); }
__global__ void int2float_rd_kernel_v2(float* result, Dummy x) { *result = __int2float_rd(x); }
__global__ void int2float_rd_kernel_v3(Dummy* result, int x) { *result = __int2float_rd(x); }
__global__ void int2float_rn_kernel_v1(float* result, int* x) { *result = __int2float_rn(x); }
__global__ void int2float_rn_kernel_v2(float* result, Dummy x) { *result = __int2float_rn(x); }
__global__ void int2float_rn_kernel_v3(Dummy* result, int x) { *result = __int2float_rn(x); }
__global__ void int2float_ru_kernel_v1(float* result, int* x) { *result = __int2float_ru(x); }
__global__ void int2float_ru_kernel_v2(float* result, Dummy x) { *result = __int2float_ru(x); }
__global__ void int2float_ru_kernel_v3(Dummy* result, int x) { *result = __int2float_ru(x); }
__global__ void int2float_rz_kernel_v1(float* result, int* x) { *result = __int2float_rz(x); }
__global__ void int2float_rz_kernel_v2(float* result, Dummy x) { *result = __int2float_rz(x); }
__global__ void int2float_rz_kernel_v3(Dummy* result, int x) { *result = __int2float_rz(x); }
)"};
static constexpr auto kUint2Float{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void uint2float_rd_kernel_v1(float* result, unsigned int* x) { *result = __uint2float_rd(x); }
__global__ void uint2float_rd_kernel_v2(float* result, Dummy x) { *result = __uint2float_rd(x); }
__global__ void uint2float_rd_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2float_rd(x); }
__global__ void uint2float_rn_kernel_v1(float* result, unsigned int* x) { *result = __uint2float_rn(x); }
__global__ void uint2float_rn_kernel_v2(float* result, Dummy x) { *result = __uint2float_rn(x); }
__global__ void uint2float_rn_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2float_rn(x); }
__global__ void uint2float_ru_kernel_v1(float* result, unsigned int* x) { *result = __uint2float_ru(x); }
__global__ void uint2float_ru_kernel_v2(float* result, Dummy x) { *result = __uint2float_ru(x); }
__global__ void uint2float_ru_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2float_ru(x); }
__global__ void uint2float_rz_kernel_v1(float* result, unsigned int* x) { *result = __uint2float_rz(x); }
__global__ void uint2float_rz_kernel_v2(float* result, Dummy x) { *result = __uint2float_rz(x); }
__global__ void uint2float_rz_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2float_rz(x); }
)"};
static constexpr auto kLL2Float{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void ll2float_rd_kernel_v1(float* result, long long int* x) { *result = __ll2float_rd(x); }
__global__ void ll2float_rd_kernel_v2(float* result, Dummy x) { *result = __ll2float_rd(x); }
__global__ void ll2float_rd_kernel_v3(Dummy* result, long long int x) { *result = __ll2float_rd(x); }
__global__ void ll2float_rn_kernel_v1(float* result, long long int* x) { *result = __ll2float_rn(x); }
__global__ void ll2float_rn_kernel_v2(float* result, Dummy x) { *result = __ll2float_rn(x); }
__global__ void ll2float_rn_kernel_v3(Dummy* result, long long int x) { *result = __ll2float_rn(x); }
__global__ void ll2float_ru_kernel_v1(float* result, long long int* x) { *result = __ll2float_ru(x); }
__global__ void ll2float_ru_kernel_v2(float* result, Dummy x) { *result = __ll2float_ru(x); }
__global__ void ll2float_ru_kernel_v3(Dummy* result, long long int x) { *result = __ll2float_ru(x); }
__global__ void ll2float_rz_kernel_v1(float* result, long long int* x) { *result = __ll2float_rz(x); }
__global__ void ll2float_rz_kernel_v2(float* result, Dummy x) { *result = __ll2float_rz(x); }
__global__ void ll2float_rz_kernel_v3(Dummy* result, long long int x) { *result = __ll2float_rz(x); }
)"};
static constexpr auto kULL2Float{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void ull2float_rd_kernel_v1(float* result, unsigned long long int* x) { *result = __ull2float_rd(x); }
__global__ void ull2float_rd_kernel_v2(float* result, Dummy x) { *result = __ull2float_rd(x); }
__global__ void ull2float_rd_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2float_rd(x); }
__global__ void ull2float_rn_kernel_v1(float* result, unsigned long long int* x) { *result = __ull2float_rn(x); }
__global__ void ull2float_rn_kernel_v2(float* result, Dummy x) { *result = __ull2float_rn(x); }
__global__ void ull2float_rn_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2float_rn(x); }
__global__ void ull2float_ru_kernel_v1(float* result, unsigned long long int* x) { *result = __ull2float_ru(x); }
__global__ void ull2float_ru_kernel_v2(float* result, Dummy x) { *result = __ull2float_ru(x); }
__global__ void ull2float_ru_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2float_ru(x); }
__global__ void ull2float_rz_kernel_v1(float* result, unsigned long long int* x) { *result = __ull2float_rz(x); }
__global__ void ull2float_rz_kernel_v2(float* result, Dummy x) { *result = __ull2float_rz(x); }
__global__ void ull2float_rz_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2float_rz(x); }
)"};
static constexpr auto kIntAsFloat{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void int_as_float_kernel_v1(float* result, int* x) { *result = __int_as_float(x); }
__global__ void int_as_float_kernel_v2(float* result, Dummy x) { *result = __int_as_float(x); }
__global__ void int_as_float_kernel_v3(Dummy* result, int x) { *result = __int_as_float(x); }
)"};
static constexpr auto kUintAsFloat{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void uint_as_float_kernel_v1(float* result, unsigned int* x) { *result = __uint_as_float(x); }
__global__ void uint_as_float_kernel_v2(float* result, Dummy x) { *result = __uint_as_float(x); }
__global__ void uint_as_float_kernel_v3(Dummy* result, unsigned int x) { *result = __uint_as_float(x); }
)"};
static constexpr auto kInt2Double{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void int2double_rn_kernel_v1(double* result, int* x) { *result = __int2double_rn(x); }
__global__ void int2double_rn_kernel_v2(double* result, Dummy x) { *result = __int2double_rn(x); }
__global__ void int2double_rn_kernel_v3(Dummy* result, int x) { *result = __int2double_rn(x); }
)"};
static constexpr auto kUint2Double{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void uint2double_rn_kernel_v1(double* result, unsigned int* x) { *result = __uint2double_rn(x); }
__global__ void uint2double_rn_kernel_v2(double* result, Dummy x) { *result = __uint2double_rn(x); }
__global__ void uint2double_rn_kernel_v3(Dummy* result, unsigned int x) { *result = __uint2double_rn(x); }
)"};
static constexpr auto kLL2Double{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void ll2double_rd_kernel_v1(double* result, long long int* x) { *result = __ll2double_rd(x); }
__global__ void ll2double_rd_kernel_v2(double* result, Dummy x) { *result = __ll2double_rd(x); }
__global__ void ll2double_rd_kernel_v3(Dummy* result, long long int x) { *result = __ll2double_rd(x); }
__global__ void ll2double_rn_kernel_v1(double* result, long long int* x) { *result = __ll2double_rn(x); }
__global__ void ll2double_rn_kernel_v2(double* result, Dummy x) { *result = __ll2double_rn(x); }
__global__ void ll2double_rn_kernel_v3(Dummy* result, long long int x) { *result = __ll2double_rn(x); }
__global__ void ll2double_ru_kernel_v1(double* result, long long int* x) { *result = __ll2double_ru(x); }
__global__ void ll2double_ru_kernel_v2(double* result, Dummy x) { *result = __ll2double_ru(x); }
__global__ void ll2double_ru_kernel_v3(Dummy* result, long long int x) { *result = __ll2double_ru(x); }
__global__ void ll2double_rz_kernel_v1(double* result, long long int* x) { *result = __ll2double_rz(x); }
__global__ void ll2double_rz_kernel_v2(double* result, Dummy x) { *result = __ll2double_rz(x); }
__global__ void ll2double_rz_kernel_v3(Dummy* result, long long int x) { *result = __ll2double_rz(x); }
)"};
static constexpr auto kULL2Double{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void ull2double_rd_kernel_v1(double* result, unsigned long long int* x) { *result = __ull2double_rd(x); }
__global__ void ull2double_rd_kernel_v2(double* result, Dummy x) { *result = __ull2double_rd(x); }
__global__ void ull2double_rd_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2double_rd(x); }
__global__ void ull2double_rn_kernel_v1(double* result, unsigned long long int* x) { *result = __ull2double_rn(x); }
__global__ void ull2double_rn_kernel_v2(double* result, Dummy x) { *result = __ull2double_rn(x); }
__global__ void ull2double_rn_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2double_rn(x); }
__global__ void ull2double_ru_kernel_v1(double* result, unsigned long long int* x) { *result = __ull2double_ru(x); }
__global__ void ull2double_ru_kernel_v2(double* result, Dummy x) { *result = __ull2double_ru(x); }
__global__ void ull2double_ru_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2double_ru(x); }
__global__ void ull2double_rz_kernel_v1(double* result, unsigned long long int* x) { *result = __ull2double_rz(x); }
__global__ void ull2double_rz_kernel_v2(double* result, Dummy x) { *result = __ull2double_rz(x); }
__global__ void ull2double_rz_kernel_v3(Dummy* result, unsigned long long int x) { *result = __ull2double_rz(x); }
)"};
static constexpr auto kLonglongAsDouble{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void longlong_as_double_kernel_v1(double* result, long long int* x) { *result = __longlong_as_double(x); }
__global__ void longlong_as_double_kernel_v2(double* result, Dummy x) { *result = __longlong_as_double(x); }
__global__ void longlong_as_double_kernel_v3(Dummy* result, long long int x) { *result = __longlong_as_double(x); }
)"};
static constexpr auto kHilo2Double{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void hiloint2double_kernel_v1(double* result, int* x, int y) { *result = __hiloint2double(x, y); }
__global__ void hiloint2double_kernel_v2(double* result, int x, int* y) { *result = __hiloint2double(x, y); }
__global__ void hiloint2double_kernel_v3(double* result, Dummy x, int y) { *result = __hiloint2double(x, y); }
__global__ void hiloint2double_kernel_v4(double* result, int x, Dummy y) { *result = __hiloint2double(x, y); }
__global__ void hiloint2double_kernel_v5(Dummy* result, int x, int y) { *result = __hiloint2double(x, y); }
)"};
@@ -0,0 +1,243 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include "unary_common.hh"
#include "binary_common.hh"
#include "ternary_common.hh"
/********** Unary Functions **********/
#define MATH_UNARY_DP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(double* const ys, const size_t num_xs, double* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(xs[i]); \
} \
}
#define MATH_UNARY_DP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
UnaryDoublePrecisionTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_UNARY_DP_TEST_DEF(func_name, ref_func) \
MATH_UNARY_DP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_UNARY_DP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<double>> func_name##_validator_builder(double target, double x)
static double __drcp_rn_ref(double x) { return 1.0 / x; }
MATH_UNARY_DP_KERNEL_DEF(__drcp_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__drcp_rn(x)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are
* IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/double_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_DP_TEST_DEF_IMPL(__drcp_rn, __drcp_rn_ref, EqValidatorBuilderFactory<double>());
MATH_UNARY_DP_KERNEL_DEF(__dsqrt_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__dsqrt_rn(x)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are
* compared against reference function `double std::sqrt(double)`. The error bounds are
* IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/double_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_DP_TEST_DEF_IMPL(__dsqrt_rn, static_cast<double (*)(double)>(std::sqrt),
EqValidatorBuilderFactory<double>());
/********** Binary Functions **********/
#define MATH_BINARY_DP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(double* const ys, const size_t num_xs, double* const x1s, \
double* const x2s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(x1s[i], x2s[i]); \
} \
}
#define MATH_BINARY_DP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
BinaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_BINARY_DP_TEST_DEF(func_name, ref_func) \
MATH_BINARY_DP_TEST_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_BINARY_DP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<double>> func_name##_validator_builder(double target, \
double x1, double x2)
static double __dadd_rn_ref(double x1, double x2) { return x1 + x2; }
MATH_BINARY_DP_KERNEL_DEF(__dadd_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__dadd_rn(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/double_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_DP_TEST_DEF_IMPL(__dadd_rn, __dadd_rn_ref, EqValidatorBuilderFactory<double>());
static double __dsub_rn_ref(double x1, double x2) { return x1 - x2; }
MATH_BINARY_DP_KERNEL_DEF(__dsub_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__dsub_rn(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/double_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_DP_TEST_DEF_IMPL(__dsub_rn, __dsub_rn_ref, EqValidatorBuilderFactory<double>());
static double __dmul_rn_ref(double x1, double x2) { return x1 * x2; }
MATH_BINARY_DP_KERNEL_DEF(__dmul_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__dmul_rn(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/double_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_DP_TEST_DEF_IMPL(__dmul_rn, __dmul_rn_ref, EqValidatorBuilderFactory<double>());
static double __ddiv_rn_ref(double x1, double x2) { return x1 / x2; }
MATH_BINARY_DP_KERNEL_DEF(__ddiv_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__ddiv_rn(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/double_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_DP_TEST_DEF_IMPL(__ddiv_rn, __ddiv_rn_ref, EqValidatorBuilderFactory<double>());
/********** Ternary Functions **********/
#define MATH_TERNARY_DP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(double* const ys, const size_t num_xs, double* const x1s, \
double* const x2s, double* const x3s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(x1s[i], x2s[i], x3s[i]); \
} \
}
#define MATH_TERNARY_DP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
TernaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_TERNARY_DP_TEST_DEF(func_name, ref_func, validator_builder) \
MATH_TERNARY_DP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_TERNARY_DP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<double>> func_name##_validator_builder( \
double target, double x1, double x2, double x3)
MATH_TERNARY_DP_KERNEL_DEF(__fma_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__fma(x,y,z)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/double_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_TERNARY_DP_TEST_DEF_IMPL(__fma_rn, static_cast<double (*)(double, double, double)>(std::fma),
EqValidatorBuilderFactory<double>());
@@ -0,0 +1,46 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define INTRINSIC_UNARY_DOUBLE_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); }
#define INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(double* x, double y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(double x, double* y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, double y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(double x, Dummy y) { double result = func_name(x, y); }
INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(__dadd_rn)
INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(__dsub_rn)
INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(__dmul_rn)
INTRINSIC_BINARY_DOUBLE_NEGATIVE_KERNELS(__ddiv_rn)
INTRINSIC_UNARY_DOUBLE_NEGATIVE_KERNELS(__dsqrt_rn)
@@ -0,0 +1,441 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "half_precision_common.hh"
/**
* @addtogroup HalfPrecisionArithmetic HalfPrecisionArithmetic
* @{
* @ingroup MathTest
*/
MATH_UNARY_HP_KERNEL_DEF(__habs);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__habs(x)` for all possible inputs. The results are
* compared against reference function `float std::abs(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(__habs, static_cast<float (*)(float)>(std::abs),
EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(__habs2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__habs2(x)` for all possible inputs. The results are
* compared against reference function `float std::abs(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(__habs2, static_cast<float (*)(float)>(std::abs),
EqValidatorBuilderFactory<float>());
static float __hneg_ref(float x) { return -x; }
MATH_UNARY_HP_KERNEL_DEF(__hneg);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hneg(x)` for all possible inputs. The error bounds are
* IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(__hneg, __hneg_ref, EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(__hneg2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hneg2(x)` for all possible inputs. The error bounds are
* IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(__hneg2, __hneg_ref, EqValidatorBuilderFactory<float>());
// Wrapper to avoid ambiguity error with __hadd(int, int)
__device__ __half __hadd_wrapper(__half x1, __half x2) { return __hadd(x1, x2); }
static float __hadd_ref(float x1, float x2) { return x1 + x2; }
MATH_BINARY_HP_KERNEL_DEF(__hadd_wrapper);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hadd(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hadd_wrapper, __hadd_ref, EqValidatorBuilderFactory<float>());
MATH_BINARY_HP_KERNEL_DEF(__hadd2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hadd2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hadd2, __hadd_ref, EqValidatorBuilderFactory<float>());
static float __hadd_sat_ref(float x1, float x2) { return std::clamp(x1 + x2, 0.0f, 1.0f); }
MATH_BINARY_HP_KERNEL_DEF(__hadd_sat);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hadd_sat(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hadd_sat, __hadd_sat_ref, EqValidatorBuilderFactory<float>());
MATH_BINARY_HP_KERNEL_DEF(__hadd2_sat);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hadd2_sat(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hadd2_sat, __hadd_sat_ref, EqValidatorBuilderFactory<float>());
static float __hsub_ref(float x1, float x2) { return x1 - x2; }
MATH_BINARY_HP_KERNEL_DEF(__hsub);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hsub(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hsub, __hsub_ref, EqValidatorBuilderFactory<float>());
MATH_BINARY_HP_KERNEL_DEF(__hsub2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hsub2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hsub2, __hsub_ref, EqValidatorBuilderFactory<float>());
static float __hsub_sat_ref(float x1, float x2) { return std::clamp(x1 - x2, 0.0f, 1.0f); }
MATH_BINARY_HP_KERNEL_DEF(__hsub_sat);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hsub_sat(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hsub_sat, __hsub_sat_ref, EqValidatorBuilderFactory<float>());
MATH_BINARY_HP_KERNEL_DEF(__hsub2_sat);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hsub2_sat(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hsub2_sat, __hsub_sat_ref, EqValidatorBuilderFactory<float>());
static float __hmul_ref(float x1, float x2) { return x1 * x2; }
MATH_BINARY_HP_KERNEL_DEF(__hmul);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hmul(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hmul, __hmul_ref, EqValidatorBuilderFactory<float>());
MATH_BINARY_HP_KERNEL_DEF(__hmul2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hmul2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hmul2, __hmul_ref, EqValidatorBuilderFactory<float>());
static float __hmul_sat_ref(float x1, float x2) { return std::clamp(x1 * x2, 0.0f, 1.0f); }
MATH_BINARY_HP_KERNEL_DEF(__hmul_sat);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hmul_sat(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hmul_sat, __hmul_sat_ref, EqValidatorBuilderFactory<float>());
MATH_BINARY_HP_KERNEL_DEF(__hmul2_sat);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hmul2_sat(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hmul2_sat, __hmul_sat_ref, EqValidatorBuilderFactory<float>());
static float __hdiv_ref(float x1, float x2) { return x1 / x2; }
MATH_BINARY_HP_KERNEL_DEF(__hdiv);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hdiv(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hdiv, __hdiv_ref, EqValidatorBuilderFactory<float>());
MATH_BINARY_HP_KERNEL_DEF(__h2div);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__h2div(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__h2div, __hdiv_ref, EqValidatorBuilderFactory<float>());
MATH_TERNARY_HP_KERNEL_DEF(__hfma);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hfma(x,y,z)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_TERNARY_HP_TEST_DEF_IMPL(__hfma, static_cast<float (*)(float, float, float)>(std::fma),
EqValidatorBuilderFactory<float>());
MATH_TERNARY_HP_KERNEL_DEF(__hfma2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hfma2(x,y,z)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_TERNARY_HP_TEST_DEF_IMPL(__hfma2, static_cast<float (*)(float, float, float)>(std::fma),
EqValidatorBuilderFactory<float>());
static float __hfma_sat_ref(float x1, float x2, float x3) {
return std::clamp(std::fma(x1, x2, x3), 0.0f, 1.0f);
}
MATH_TERNARY_HP_KERNEL_DEF(__hfma_sat);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hfma_sat(x,y,z)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_TERNARY_HP_TEST_DEF_IMPL(__hfma_sat, __hfma_sat_ref, EqValidatorBuilderFactory<float>());
MATH_TERNARY_HP_KERNEL_DEF(__hfma2_sat);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hfma2_sat(x,y,z)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/half_precision_arithmetic.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_TERNARY_HP_TEST_DEF_IMPL(__hfma2_sat, __hfma_sat_ref, EqValidatorBuilderFactory<float>());
@@ -0,0 +1,124 @@
/*
Copyright (c) 2022 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <hip/hip_fp16.h>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define UNARY_HALF_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half* x) { __half result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { __half result = func_name(x); }
#define BINARY_HALF_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half* x, __half y) { __half result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(__half x, __half* y) { __half result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, __half y) { __half result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(__half x, Dummy y) { __half result = func_name(x, y); }
#define TERNARY_HALF_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half* x, __half y, __half z) { \
__half result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v2(__half x, __half* y, __half z) { \
__half result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v3(__half x, __half y, __half* z) { \
__half result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v4(Dummy x, __half y, __half z) { \
__half result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v5(__half x, Dummy y, __half z) { \
__half result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v6(__half x, __half y, Dummy z) { \
__half result = func_name(x, y, z); \
}
UNARY_HALF_NEGATIVE_KERNELS(__habs)
UNARY_HALF_NEGATIVE_KERNELS(__hneg)
BINARY_HALF_NEGATIVE_KERNELS(__hadd)
BINARY_HALF_NEGATIVE_KERNELS(__hadd_sat)
BINARY_HALF_NEGATIVE_KERNELS(__hsub)
BINARY_HALF_NEGATIVE_KERNELS(__hsub_sat)
BINARY_HALF_NEGATIVE_KERNELS(__hmul)
BINARY_HALF_NEGATIVE_KERNELS(__hmul_sat)
BINARY_HALF_NEGATIVE_KERNELS(__hdiv)
TERNARY_HALF_NEGATIVE_KERNELS(__hfma)
TERNARY_HALF_NEGATIVE_KERNELS(__hfma_sat)
#define UNARY_HALF2_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half2* x) { __half2 result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { __half2 result = func_name(x); }
#define BINARY_HALF2_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half2* x, __half2 y) { \
__half2 result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v2(__half2 x, __half2* y) { \
__half2 result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v3(Dummy x, __half2 y) { __half2 result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(__half2 x, Dummy y) { __half2 result = func_name(x, y); }
#define TERNARY_HALF2_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half2* x, __half2 y, __half2 z) { \
__half2 result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v2(__half2 x, __half2* y, __half2 z) { \
__half2 result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v3(__half2 x, __half2 y, __half2* z) { \
__half2 result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v4(Dummy x, __half2 y, __half2 z) { \
__half2 result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v5(__half2 x, Dummy y, __half2 z) { \
__half2 result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v6(__half2 x, __half2 y, Dummy z) { \
__half2 result = func_name(x, y, z); \
}
UNARY_HALF2_NEGATIVE_KERNELS(__habs2)
UNARY_HALF2_NEGATIVE_KERNELS(__hneg2)
BINARY_HALF2_NEGATIVE_KERNELS(__hadd2)
BINARY_HALF2_NEGATIVE_KERNELS(__hadd2_sat)
BINARY_HALF2_NEGATIVE_KERNELS(__hsub2)
BINARY_HALF2_NEGATIVE_KERNELS(__hsub2_sat)
BINARY_HALF2_NEGATIVE_KERNELS(__hmul2)
BINARY_HALF2_NEGATIVE_KERNELS(__hmul2_sat)
BINARY_HALF2_NEGATIVE_KERNELS(__h2div)
TERNARY_HALF2_NEGATIVE_KERNELS(__hfma2)
TERNARY_HALF2_NEGATIVE_KERNELS(__hfma2_sat)
@@ -0,0 +1,103 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "unary_common.hh"
#include "binary_common.hh"
#include "ternary_common.hh"
/********** Unary **********/
#define MATH_UNARY_HP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(Float16* const ys, const size_t num_xs, Float16* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(xs[i]); \
} \
}
#define MATH_UNARY_HP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
UnaryHalfPrecisionTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_UNARY_HP_TEST_DEF(func_name, ref_func) \
MATH_UNARY_HP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_UNARY_HP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x)
/********** Binary **********/
#define MATH_BINARY_HP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(Float16* const ys, const size_t num_xs, Float16* const x1s, \
Float16* const x2s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(x1s[i], x2s[i]); \
} \
}
#define MATH_BINARY_HP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
BinaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_BINARY_HP_TEST_DEF(func_name, ref_func) \
MATH_BINARY_HP_TEST_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_BINARY_HP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x1, \
float x2)
/********** Ternary **********/
#define MATH_TERNARY_HP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(Float16* const ys, const size_t num_xs, Float16* const x1s, \
Float16* const x2s, Float16* const x3s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(x1s[i], x2s[i], x3s[i]); \
} \
}
#define MATH_TERNARY_HP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
TernaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_TERNARY_HP_TEST_DEF(func_name, ref_func, validator_builder) \
MATH_TERNARY_HP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_TERNARY_HP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x1, \
float x2, float x3)
@@ -0,0 +1,847 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "half_precision_common.hh"
/**
* @addtogroup HalfPrecisionComparison HalfPrecisionComparison
* @{
* @ingroup MathTest
*/
/********** Unary Functions **********/
#define MATH_BOOL_UNARY_HP_TEST_DEF(func_name, ref_func) \
__global__ void func_name##_kernel(bool* const ys, const size_t num_xs, Float16* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(xs[i]); \
} \
} \
\
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
UnaryHalfPrecisionTest(func_name##_kernel, ref_func, EqValidatorBuilderFactory<bool>()); \
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hisinf(x)` for all possible inputs. The results are
* compared against reference function `bool std::isinf(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BOOL_UNARY_HP_TEST_DEF(__hisinf, static_cast<bool (*)(float)>(std::isinf))
static float __hisinf2_ref(float x) { return static_cast<float>(std::isinf(x)); }
MATH_UNARY_HP_KERNEL_DEF(__hisinf2)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hisinf2(x)` for all possible inputs. The results are
* compared against reference function `float std::isinf(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(__hisinf2, __hisinf2_ref, EqValidatorBuilderFactory<float>());
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hisnan(x)` for all possible inputs. The results are
* compared against reference function `bool std::isnan(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BOOL_UNARY_HP_TEST_DEF(__hisnan, static_cast<bool (*)(float)>(std::isnan))
static float __hisnan2_ref(float x) { return static_cast<float>(std::isnan(x)); }
MATH_UNARY_HP_KERNEL_DEF(__hisnan2)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hisnan2(x)` for all possible inputs. The results are
* compared against reference function `float std::isnan(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(__hisnan2, __hisnan2_ref, EqValidatorBuilderFactory<float>());
/********** Binary Functions **********/
#define MATH_COMPARISON_HP_TEST_DEF(func_name, ref_func, T, RT, nan_value) \
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, Float16* const x1s, \
Float16* const x2s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(x1s[i], x2s[i]); \
} \
} \
\
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
BinaryFloatingPointTest(func_name##_kernel, ref_func<nan_value, RT>, \
EqValidatorBuilderFactory<RT>()); \
}
template <bool nan_value, typename T> static T __heq_ref(float x1, float x2) {
if (std::isnan(x1) || std::isnan(x2)) {
return static_cast<T>(nan_value);
}
return x1 == x2;
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__heq(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'equal
* to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__heq, __heq_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbeq2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'equal
* to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbeq2, __heq_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hequ(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'equal
* to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hequ, __heq_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbequ2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are compared against result
* of 'equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbequ2, __heq_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__heq2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'equal
* to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__heq2, __heq_ref, Float16, float, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hequ2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'equal
* to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hequ2, __heq_ref, Float16, float, true)
template <bool nan_value, typename T> static T __hne_ref(float x1, float x2) {
if (std::isnan(x1) || std::isnan(x2)) {
return static_cast<T>(nan_value);
}
return x1 != x2;
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hne(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'not
* equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hne, __hne_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbne2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'not
* equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbne2, __hne_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hneu(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'not
* equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hneu, __hne_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbneu2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are compared against result
* of 'not equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbneu2, __hne_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hne2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'not
* equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hne2, __hne_ref, Float16, float, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hneu2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'not
* equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hneu2, __hne_ref, Float16, float, true)
template <bool nan_value, typename T> static T __hge_ref(float x1, float x2) {
if (std::isnan(x1) || std::isnan(x2)) {
return static_cast<T>(nan_value);
}
return x1 >= x2;
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hge(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hge, __hge_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbge2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbge2, __hge_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hgeu(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hgeu, __hge_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbgeu2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are compared against result
* of 'greater than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbgeu2, __hge_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hge2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hge2, __hge_ref, Float16, float, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hgeu2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hgeu2, __hge_ref, Float16, float, true)
template <bool nan_value, typename T> static T __hgt_ref(float x1, float x2) {
if (std::isnan(x1) || std::isnan(x2)) {
return static_cast<T>(nan_value);
}
return x1 > x2;
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hgt(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hgt, __hgt_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbgt2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbgt2, __hgt_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hgtu(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hgtu, __hgt_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbgtu2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are compared against result
* of 'greater than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbgtu2, __hgt_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hgt2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hgt2, __hgt_ref, Float16, float, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hgtu2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of
* 'greater than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hgtu2, __hgt_ref, Float16, float, true)
template <bool nan_value, typename T> static T __hle_ref(float x1, float x2) {
if (std::isnan(x1) || std::isnan(x2)) {
return static_cast<T>(nan_value);
}
return x1 <= x2;
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hle(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hle, __hle_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hble2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hble2, __hle_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hleu(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hleu, __hle_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbleu2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are compared against result
* of 'less than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbleu2, __hle_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hle2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hle2, __hle_ref, Float16, float, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hleu2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than equal to' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hleu2, __hle_ref, Float16, float, true)
template <bool nan_value, typename T> static T __hlt_ref(float x1, float x2) {
if (std::isnan(x1) || std::isnan(x2)) {
return static_cast<T>(nan_value);
}
return x1 < x2;
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hlt(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hlt, __hlt_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hblt2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hblt2, __hlt_ref, bool, bool, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hltu(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hltu, __hlt_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hbltu2(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are compared against result
* of 'less than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hbltu2, __hlt_ref, bool, bool, true)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hlt2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hlt2, __hlt_ref, Float16, float, false)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hltu2(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against result of 'less
* than' relational operator for float operands.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_COMPARISON_HP_TEST_DEF(__hltu2, __hlt_ref, Float16, float, true)
MATH_BINARY_HP_KERNEL_DEF(__hmax)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hmax(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against reference
* function `float std::fmax(float, float)`
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hmax, static_cast<float (*)(float, float)>(std::fmax),
EqValidatorBuilderFactory<float>())
MATH_BINARY_HP_KERNEL_DEF(__hmin)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hmin(x,y)` against a table of difficult values, followed
* by a large number of randomly generated values. The results are compared against reference
* function `float std::fmin(float, float)`
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hmin, static_cast<float (*)(float, float)>(std::fmin),
EqValidatorBuilderFactory<float>())
static float __hmax_nan_ref(float x1, float x2) {
if (std::isnan(x1))
return x1;
else if (std::isnan(x2))
return x2;
else
return std::fmax(x1, x2);
}
MATH_BINARY_HP_KERNEL_DEF(__hmax_nan)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hmax_nan(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are compared against
* reference function `float std::fmax(float, float)` with modified result when an operand is nan.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hmax_nan, __hmax_nan_ref, EqValidatorBuilderFactory<float>())
static float __hmin_nan_ref(float x1, float x2) {
if (std::isnan(x1))
return x1;
else if (std::isnan(x2))
return x2;
else
return std::fmin(x1, x2);
}
MATH_BINARY_HP_KERNEL_DEF(__hmin_nan)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__hmin_nan(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are compared against
* reference function `float std::fmin(float, float)` with modified result when an operand is nan.
*
* Test source
* ------------------------
* - unit/math/half_precision_comparison.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_HP_TEST_DEF_IMPL(__hmin_nan, __hmin_nan_ref, EqValidatorBuilderFactory<float>())
@@ -0,0 +1,120 @@
/*
Copyright (c) 2022 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <hip/hip_fp16.h>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define UNARY_BOOL_HALF_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half* x) { bool result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { bool result = func_name(x); }
#define BINARY_BOOL_HALF_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half* x, __half y) { bool result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(__half x, __half* y) { bool result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, __half y) { bool result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(__half x, Dummy y) { bool result = func_name(x, y); }
#define BINARY_HALF_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half* x, __half y) { __half result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(__half x, __half* y) { __half result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, __half y) { __half result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(__half x, Dummy y) { __half result = func_name(x, y); }
UNARY_BOOL_HALF_NEGATIVE_KERNELS(__hisinf)
UNARY_BOOL_HALF_NEGATIVE_KERNELS(__hisnan)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__heq)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hequ)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hne)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hneu)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hge)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hgeu)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hgt)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hgtu)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hle)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hleu)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hlt)
BINARY_BOOL_HALF_NEGATIVE_KERNELS(__hltu)
BINARY_HALF_NEGATIVE_KERNELS(__hmax)
BINARY_HALF_NEGATIVE_KERNELS(__hmax_nan)
BINARY_HALF_NEGATIVE_KERNELS(__hmin)
BINARY_HALF_NEGATIVE_KERNELS(__hmin_nan)
#define UNARY_HALF2_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half2* x) { __half2 result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { __half2 result = func_name(x); }
#define BINARY_HALF2_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half2* x, __half2 y) { \
__half2 result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v2(__half2 x, __half2* y) { \
__half2 result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v3(Dummy x, __half2 y) { __half2 result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(__half2 x, Dummy y) { __half2 result = func_name(x, y); }
#define BINARY_BOOL_HALF2_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half2* x, __half2 y) { bool result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(__half2 x, __half2* y) { bool result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, __half2 y) { bool result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(__half2 x, Dummy y) { bool result = func_name(x, y); }
UNARY_HALF2_NEGATIVE_KERNELS(__hisinf2)
UNARY_HALF2_NEGATIVE_KERNELS(__hisnan2)
BINARY_HALF2_NEGATIVE_KERNELS(__heq2)
BINARY_HALF2_NEGATIVE_KERNELS(__hequ2)
BINARY_HALF2_NEGATIVE_KERNELS(__hne2)
BINARY_HALF2_NEGATIVE_KERNELS(__hneu2)
BINARY_HALF2_NEGATIVE_KERNELS(__hge2)
BINARY_HALF2_NEGATIVE_KERNELS(__hgeu2)
BINARY_HALF2_NEGATIVE_KERNELS(__hgt2)
BINARY_HALF2_NEGATIVE_KERNELS(__hgtu2)
BINARY_HALF2_NEGATIVE_KERNELS(__hle2)
BINARY_HALF2_NEGATIVE_KERNELS(__hleu2)
BINARY_HALF2_NEGATIVE_KERNELS(__hlt2)
BINARY_HALF2_NEGATIVE_KERNELS(__hltu2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbeq2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbequ2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbne2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbneu2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbge2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbgeu2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbgt2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbgtu2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hble2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbleu2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hblt2)
BINARY_BOOL_HALF2_NEGATIVE_KERNELS(__hbltu2)
+580
Wyświetl plik
@@ -0,0 +1,580 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "half_precision_common.hh"
/**
* @addtogroup HalfPrecisionMath HalfPrecisionMath
* @{
* @ingroup MathTest
*/
MATH_UNARY_HP_KERNEL_DEF(hcos);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hcos(x)` for all possible inputs. The results are
* compared against reference function `float std::cos(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hcos, static_cast<float (*)(float)>(std::cos),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(h2cos);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2cos(x)` for all possible inputs. The results are
* compared against reference function `float std::cos(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2cos, static_cast<float (*)(float)>(std::cos),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(hsin);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hsin(x)` for all possible inputs. The results are
* compared against reference function `float std::sin(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hsin, static_cast<float (*)(float)>(std::sin),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(h2sin);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2sin(x)` for all possible inputs. The results are
* compared against reference function `float std::sin(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2sin, static_cast<float (*)(float)>(std::sin),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(hexp);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hexp(x)` for all possible inputs. The results are
* compared against reference function `float std::exp(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hexp, static_cast<float (*)(float)>(std::exp),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(h2exp);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2exp(x)` for all possible inputs. The results are
* compared against reference function `float std::exp(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2exp, static_cast<float (*)(float)>(std::exp),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(hexp10);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hexp10(x)` for all possible inputs. The results are
* compared against reference function `float exp10(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hexp10, static_cast<float (*)(float)>(exp10f),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(h2exp10);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2exp10(x)` for all possible inputs. The results are
* compared against reference function `float exp10(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2exp10, static_cast<float (*)(float)>(exp10f),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(hexp2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hexp2(x)` for all possible inputs. The results are
* compared against reference function `float std::exp2(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hexp2, static_cast<float (*)(float)>(std::exp2),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(h2exp2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2exp2(x)` for all possible inputs. The results are
* compared against reference function `float std::exp2(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2exp2, static_cast<float (*)(float)>(std::exp2),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(hlog);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hlog(x)` for all possible inputs. The results are
* compared against reference function `float std::log(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hlog, static_cast<float (*)(float)>(std::log),
ULPValidatorBuilderFactory<float>(1));
MATH_UNARY_HP_KERNEL_DEF(h2log);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2log(x)` for all possible inputs. The results are
* compared against reference function `float std::log(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2log, static_cast<float (*)(float)>(std::log),
ULPValidatorBuilderFactory<float>(1));
MATH_UNARY_HP_KERNEL_DEF(hlog10);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hlog10(x)` for all possible inputs. The results are
* compared against reference function `float std::log10(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hlog10, static_cast<float (*)(float)>(std::log10),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(h2log10);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2log10(x)` for all possible inputs. The results are
* compared against reference function `float std::log10(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2log10, static_cast<float (*)(float)>(std::log10),
ULPValidatorBuilderFactory<float>(2));
MATH_UNARY_HP_KERNEL_DEF(hlog2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hlog2(x)` for all possible inputs. The results are
* compared against reference function `float std::log2(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hlog2, static_cast<float (*)(float)>(std::log2),
ULPValidatorBuilderFactory<float>(1));
MATH_UNARY_HP_KERNEL_DEF(h2log2);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2log2(x)` for all possible inputs. The results are
* compared against reference function `float std::log2(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2log2, static_cast<float (*)(float)>(std::log2),
ULPValidatorBuilderFactory<float>(1));
MATH_UNARY_HP_KERNEL_DEF(hsqrt);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hsqrt(x)` for all possible inputs. The results are
* compared against reference function `float std::sqrt(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hsqrt, static_cast<float (*)(float)>(std::sqrt),
ULPValidatorBuilderFactory<float>(1));
MATH_UNARY_HP_KERNEL_DEF(h2sqrt);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2sqrt(x)` for all possible inputs. The results are
* compared against reference function `float std::sqrt(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2sqrt, static_cast<float (*)(float)>(std::sqrt),
ULPValidatorBuilderFactory<float>(1));
MATH_UNARY_HP_KERNEL_DEF(hceil);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hceil(x)` for all possible inputs. The results are
* compared against reference function `float std::ceil(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hceil, static_cast<float (*)(float)>(std::ceil),
EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(h2ceil);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2ceil(x)` for all possible inputs. The results are
* compared against reference function `float std::ceil(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2ceil, static_cast<float (*)(float)>(std::ceil),
EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(hfloor);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hfloor(x)` for all possible inputs. The results are
* compared against reference function `float std::floor(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hfloor, static_cast<float (*)(float)>(std::floor),
EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(h2floor);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2floor(x)` for all possible inputs. The results are
* compared against reference function `float std::floor(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2floor, static_cast<float (*)(float)>(std::floor),
EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(htrunc);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `htrunc(x)` for all possible inputs. The results are
* compared against reference function `float std::trunc(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(htrunc, static_cast<float (*)(float)>(std::trunc),
EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(h2trunc);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2trunc(x)` for all possible inputs. The results are
* compared against reference function `float std::trunc(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2trunc, static_cast<float (*)(float)>(std::trunc),
EqValidatorBuilderFactory<float>());
static float hrcp_ref(float x) { return 1.0f / x; }
MATH_UNARY_HP_KERNEL_DEF(hrcp);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hrcp(x)` for all possible inputs.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hrcp, hrcp_ref, EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(h2rcp);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2rcp(x)` for all possible inputs.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2rcp, hrcp_ref, EqValidatorBuilderFactory<float>());
static float hrsqrt_ref(float x) { return 1.0f / std::sqrt(x); }
MATH_UNARY_HP_KERNEL_DEF(hrsqrt);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hrsqrt(x)` for all possible inputs.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hrsqrt, hrsqrt_ref, EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(h2rsqrt);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2rsqrt(x)` for all possible inputs.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2rsqrt, hrsqrt_ref, EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(hrint);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hrint(x)` for all possible inputs. The results are
* compared against reference function `float std::rint(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(hrint, static_cast<float (*)(float)>(std::rint),
EqValidatorBuilderFactory<float>());
MATH_UNARY_HP_KERNEL_DEF(h2rint);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `h2rint(x)` for all possible inputs. The results are
* compared against reference function `float std::rint(float)`.
*
* Test source
* ------------------------
* - unit/math/half_precision_math.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_HP_TEST_DEF_IMPL(h2rint, static_cast<float (*)(float)>(std::rint),
EqValidatorBuilderFactory<float>());
@@ -0,0 +1,72 @@
/*
Copyright (c) 2022 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <hip/hip_fp16.h>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define UNARY_HALF_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half* x) { __half result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { __half result = func_name(x); }
UNARY_HALF_NEGATIVE_KERNELS(hcos)
UNARY_HALF_NEGATIVE_KERNELS(hsin)
UNARY_HALF_NEGATIVE_KERNELS(hexp)
UNARY_HALF_NEGATIVE_KERNELS(hexp10)
UNARY_HALF_NEGATIVE_KERNELS(hexp2)
UNARY_HALF_NEGATIVE_KERNELS(hlog)
UNARY_HALF_NEGATIVE_KERNELS(hlog10)
UNARY_HALF_NEGATIVE_KERNELS(hlog2)
UNARY_HALF_NEGATIVE_KERNELS(hsqrt)
UNARY_HALF_NEGATIVE_KERNELS(hceil)
UNARY_HALF_NEGATIVE_KERNELS(hfloor)
UNARY_HALF_NEGATIVE_KERNELS(htrunc)
UNARY_HALF_NEGATIVE_KERNELS(hrcp)
UNARY_HALF_NEGATIVE_KERNELS(hrsqrt)
UNARY_HALF_NEGATIVE_KERNELS(hrint)
#define UNARY_HALF2_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(__half2* x) { __half2 result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { __half2 result = func_name(x); }
UNARY_HALF2_NEGATIVE_KERNELS(h2cos)
UNARY_HALF2_NEGATIVE_KERNELS(h2sin)
UNARY_HALF2_NEGATIVE_KERNELS(h2exp)
UNARY_HALF2_NEGATIVE_KERNELS(h2exp10)
UNARY_HALF2_NEGATIVE_KERNELS(h2exp2)
UNARY_HALF2_NEGATIVE_KERNELS(h2log)
UNARY_HALF2_NEGATIVE_KERNELS(h2log10)
UNARY_HALF2_NEGATIVE_KERNELS(h2log2)
UNARY_HALF2_NEGATIVE_KERNELS(h2sqrt)
UNARY_HALF2_NEGATIVE_KERNELS(h2ceil)
UNARY_HALF2_NEGATIVE_KERNELS(h2floor)
UNARY_HALF2_NEGATIVE_KERNELS(h2trunc)
UNARY_HALF2_NEGATIVE_KERNELS(h2rcp)
UNARY_HALF2_NEGATIVE_KERNELS(h2rsqrt)
UNARY_HALF2_NEGATIVE_KERNELS(h2rint)
+794
Wyświetl plik
@@ -0,0 +1,794 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include <resource_guards.hh>
__global__ void __brev_kernel(unsigned int* y, unsigned int x) { y[0] = __brev(x); }
/**
* Test Description
* ------------------------
* - Sanity test for `__brev(x)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___brev_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
__brev_kernel<<<1, 1>>>(y.ptr(), 0xAAAAAAAA);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 0x55555555);
}
__global__ void __brevll_kernel(unsigned long long int* y, unsigned long long int x) {
y[0] = __brevll(x);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__brevll(x)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___brevll_Sanity_Positive") {
LinearAllocGuard<unsigned long long int> y(LinearAllocs::hipMallocManaged,
sizeof(unsigned long long int));
__brevll_kernel<<<1, 1>>>(y.ptr(), 0xAAAAAAAAAAAAAAAA);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 0x5555555555555555);
}
template <typename T> __global__ void __clz_kernel(T* y, T x) { y[0] = __clz(x); }
/**
* Test Description
* ------------------------
* - Sanity test for `__clz(x)`. Run for `int` and `unsigned int` overloads.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device___clz_Sanity_Positive", "", int, unsigned int) {
LinearAllocGuard<TestType> y(LinearAllocs::hipMallocManaged, sizeof(TestType));
__clz_kernel<<<1, 1>>>(y.ptr(), static_cast<TestType>(0));
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 32);
TestType x = 1;
for (int i = 0; i < 32; ++i) {
__clz_kernel<<<1, 1>>>(y.ptr(), x << i);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 31 - i);
}
}
template <typename T> __global__ void __clzll_kernel(T* y, T x) { y[0] = __clzll(x); }
/**
* Test Description
* ------------------------
* - Sanity test for `__clzll(x)`. Run for `long long int` and `unsigned long long int`
* overloads.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device___clzll_Sanity_Positive", "", long long int,
unsigned long long int) {
LinearAllocGuard<TestType> y(LinearAllocs::hipMallocManaged, sizeof(TestType));
__clzll_kernel<<<1, 1>>>(y.ptr(), static_cast<TestType>(0));
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 64);
TestType x = 1;
for (int i = 0; i < 64; ++i) {
__clzll_kernel<<<1, 1>>>(y.ptr(), x << i);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 63 - i);
}
}
template <typename T> __global__ void __ffs_kernel(T* y, T x) { y[0] = __ffs(x); }
/**
* Test Description
* ------------------------
* - Sanity test for `__ffs(x)`. Run for `int` and `unsigned int` overloads.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device___ffs_Sanity_Positive", "", int, unsigned int) {
LinearAllocGuard<TestType> y(LinearAllocs::hipMallocManaged, sizeof(TestType));
__ffs_kernel<<<1, 1>>>(y.ptr(), static_cast<TestType>(0));
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 0);
TestType x = 1;
for (int i = 0; i < 32; ++i) {
__ffs_kernel<<<1, 1>>>(y.ptr(), x << i);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == i + 1);
}
}
template <typename T> __global__ void __ffsll_kernel(T* y, T x) { y[0] = __ffsll(x); }
/**
* Test Description
* ------------------------
* - Sanity test for `__ffsll(x)`. Run for `long long int` and `unsigned long long int`
* overloads.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device___ffsll_Sanity_Positive", "", long long int,
unsigned long long int) {
LinearAllocGuard<TestType> y(LinearAllocs::hipMallocManaged, sizeof(TestType));
__ffsll_kernel<<<1, 1>>>(y.ptr(), static_cast<TestType>(0));
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 0);
TestType x = 1;
for (int i = 0; i < 64; ++i) {
__ffsll_kernel<<<1, 1>>>(y.ptr(), x << i);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == i + 1);
}
}
__global__ void __popc_kernel(unsigned int* y, unsigned int x) { y[0] = __popc(x); }
/**
* Test Description
* ------------------------
* - Sanity test for `__popc(x)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___popc_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
__popc_kernel<<<1, 1>>>(y.ptr(), 0);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 0);
unsigned int x = 0;
for (int i = 0; i < 32; ++i) {
__popc_kernel<<<1, 1>>>(y.ptr(), x |= (1u << i));
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == i + 1);
}
}
__global__ void __popcll_kernel(unsigned long long int* y, unsigned long long int x) {
y[0] = __popcll(x);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__popcll(x)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___popcll_Sanity_Positive") {
LinearAllocGuard<unsigned long long int> y(LinearAllocs::hipMallocManaged,
sizeof(unsigned long long int));
__popcll_kernel<<<1, 1>>>(y.ptr(), 0);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == 0);
unsigned long long int x = 0;
for (int i = 0; i < 64; ++i) {
__popcll_kernel<<<1, 1>>>(y.ptr(), x |= (1ull << i));
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == i + 1);
}
}
__global__ void __mul24_kernel(int* y, int x1, int x2) { y[0] = __mul24(x1, x2); }
/**
* Test Description
* ------------------------
* - Sanity test for `__mul24(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___mul24_Sanity_Positive") {
LinearAllocGuard<int> y(LinearAllocs::hipMallocManaged, sizeof(int));
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
__mul24_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == x1 * x2);
}
__global__ void __umul24_kernel(unsigned int* y, unsigned int x1, unsigned int x2) {
y[0] = __umul24(x1, x2);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__umul24(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___umul24_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
unsigned int x1 = GENERATE(0, 42, 0xFFFFFF);
unsigned int x2 = GENERATE(0, 42, 0xFFFFFF);
__umul24_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
REQUIRE(y.ptr()[0] == x1 * x2);
}
__global__ void __funnelshift_l_kernel(unsigned int* y, unsigned int lo, unsigned int hi,
unsigned int shift) {
y[0] = __funnelshift_l(lo, hi, shift);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__funnelshift_l(lo,hi,shift)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___funnelshift_l_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
const unsigned int lo = 0xAAAAAAAA, hi = 0xBBBBBBBB;
const unsigned long long hi_lo = (static_cast<unsigned long long>(hi) << 32) | lo;
for (unsigned int shift = 0; shift < 64; ++shift) {
__funnelshift_l_kernel<<<1, 1>>>(y.ptr(), lo, hi, shift);
HIP_CHECK(hipDeviceSynchronize());
INFO("shift: " << shift);
REQUIRE(y.ptr()[0] == static_cast<unsigned int>((hi_lo << (shift & 31)) >> 32));
}
}
__global__ void __funnelshift_lc_kernel(unsigned int* y, unsigned int lo, unsigned int hi,
unsigned int shift) {
y[0] = __funnelshift_lc(lo, hi, shift);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__funnelshift_lc(lo,hi,shift)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___funnelshift_lc_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
const unsigned int lo = 0xAAAAAAAA, hi = 0xBBBBBBBB;
const unsigned long long hi_lo = (static_cast<unsigned long long>(hi) << 32) | lo;
for (unsigned int shift = 0; shift < 64; ++shift) {
__funnelshift_lc_kernel<<<1, 1>>>(y.ptr(), lo, hi, shift);
HIP_CHECK(hipDeviceSynchronize());
INFO("shift: " << shift);
REQUIRE(y.ptr()[0] == static_cast<unsigned int>((hi_lo << std::min(shift, 32u)) >> 32));
}
}
__global__ void __funnelshift_r_kernel(unsigned int* y, unsigned int lo, unsigned int hi,
unsigned int shift) {
y[0] = __funnelshift_r(lo, hi, shift);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__funnelshift_r(lo,hi,shift)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___funnelshift_r_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
const unsigned int lo = 0xAAAAAAAA, hi = 0xBBBBBBBB;
const unsigned long long hi_lo = (static_cast<unsigned long long>(hi) << 32) | lo;
for (unsigned int shift = 0; shift < 64; ++shift) {
__funnelshift_r_kernel<<<1, 1>>>(y.ptr(), lo, hi, shift);
HIP_CHECK(hipDeviceSynchronize());
INFO("shift: " << shift);
REQUIRE(y.ptr()[0] == static_cast<unsigned int>(hi_lo >> (shift & 31)));
}
}
__global__ void __funnelshift_rc_kernel(unsigned int* y, unsigned int lo, unsigned int hi,
unsigned int shift) {
y[0] = __funnelshift_rc(lo, hi, shift);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__funnelshift_rc(lo,hi,shift)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___funnelshift_rc_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
const unsigned int lo = 0xAAAAAAAA, hi = 0xBBBBBBBB;
const unsigned long long hi_lo = (static_cast<unsigned long long>(hi) << 32) | lo;
for (unsigned int shift = 0; shift < 64; ++shift) {
__funnelshift_rc_kernel<<<1, 1>>>(y.ptr(), lo, hi, shift);
HIP_CHECK(hipDeviceSynchronize());
INFO("shift: " << shift);
REQUIRE(y.ptr()[0] == static_cast<unsigned int>(hi_lo >> std::min(shift, 32u)));
}
}
__global__ void __hadd_kernel(int* y, int x1, int x2) { y[0] = __hadd(x1, x2); }
/**
* Test Description
* ------------------------
* - Sanity test for `__hadd(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___hadd_Sanity_Positive") {
LinearAllocGuard<int> y(LinearAllocs::hipMallocManaged, sizeof(int));
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
__hadd_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] == static_cast<int>((static_cast<long long>(x1) + x2) >> 1));
}
__global__ void __uhadd_kernel(unsigned int* y, unsigned int x1, unsigned int x2) {
y[0] = __uhadd(x1, x2);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__uhadd(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___uhadd_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
unsigned int x1 = GENERATE(0, 42, 0xFFFFFFFF);
unsigned int x2 = GENERATE(0, 42, 0xFFFFFFFF);
__uhadd_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] == static_cast<unsigned int>((static_cast<unsigned long long>(x1) + x2) >> 1));
}
__global__ void __rhadd_kernel(int* y, int x1, int x2) { y[0] = __rhadd(x1, x2); }
/**
* Test Description
* ------------------------
* - Sanity test for `__rhadd(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___rhadd_Sanity_Positive") {
LinearAllocGuard<int> y(LinearAllocs::hipMallocManaged, sizeof(int));
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
__rhadd_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] == static_cast<int>((static_cast<long long>(x1) + x2 + 1) >> 1));
}
__global__ void __urhadd_kernel(unsigned int* y, unsigned int x1, unsigned int x2) {
y[0] = __urhadd(x1, x2);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__urhadd(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___urhadd_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
unsigned int x1 = GENERATE(0, 42, 0xFFFFFFFF);
unsigned int x2 = GENERATE(0, 42, 0xFFFFFFFF);
__urhadd_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] ==
static_cast<unsigned int>((static_cast<unsigned long long>(x1) + x2 + 1) >> 1));
}
__global__ void __mulhi_kernel(int* y, int x1, int x2) { y[0] = __mulhi(x1, x2); }
/**
* Test Description
* ------------------------
* - Sanity test for `__mulhi(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___mulhi_Sanity_Positive") {
LinearAllocGuard<int> y(LinearAllocs::hipMallocManaged, sizeof(int));
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
__mulhi_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] ==
static_cast<int>((static_cast<long long>(x1) * static_cast<long long>(x2)) >> 32));
}
__global__ void __umulhi_kernel(unsigned int* y, unsigned int x1, unsigned int x2) {
y[0] = __umulhi(x1, x2);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__umulhi(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___umulhi_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
unsigned int x1 = GENERATE(0, 42, 0xFFFFFFFF);
unsigned int x2 = GENERATE(0, 42, 0xFFFFFFFF);
__umulhi_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] ==
static_cast<unsigned int>((static_cast<unsigned long long>(x1) * x2) >> 32));
}
__global__ void __mul64hi_kernel(long long* y, long long x1, long long x2) {
y[0] = __mul64hi(x1, x2);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__mul64hi(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___mul64hi_Sanity_Positive") {
LinearAllocGuard<long long> y(LinearAllocs::hipMallocManaged, sizeof(long long));
long long x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
long long x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
__mul64hi_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(
y.ptr()[0] ==
static_cast<long long>((static_cast<__int128_t>(x1) * static_cast<__int128_t>(x2)) >> 64));
}
__global__ void __umul64hi_kernel(unsigned long long* y, unsigned long long x1,
unsigned long long x2) {
y[0] = __umul64hi(x1, x2);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__umul64hi(x,y)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___umul64hi_Sanity_Positive") {
LinearAllocGuard<unsigned long long> y(LinearAllocs::hipMallocManaged,
sizeof(unsigned long long));
unsigned long long x1 = GENERATE(0, 42, 0xFFFFFFFF);
unsigned long long x2 = GENERATE(0, 42, 0xFFFFFFFF);
__umul64hi_kernel<<<1, 1>>>(y.ptr(), x1, x2);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] ==
static_cast<unsigned long long>(
(static_cast<__uint128_t>(x1) * static_cast<__uint128_t>(x2)) >> 64));
}
__global__ void __sad_kernel(unsigned int* y, int x1, int x2, unsigned int x3) {
y[0] = __sad(x1, x2, x3);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__sad(x,y,z)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___sad_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
int x1 = GENERATE(0, -42, 42, 0xFFFFFFFF);
int x2 = GENERATE(0, -42, 42, 0xFFFFFFFF);
unsigned int x3 = GENERATE(0, 42, 0xFFFFFFFF);
__sad_kernel<<<1, 1>>>(y.ptr(), x1, x2, x3);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] == (static_cast<unsigned int>(std::abs(x1 - x2)) + x3));
}
__global__ void __usad_kernel(unsigned int* y, unsigned int x1, unsigned int x2, unsigned int x3) {
y[0] = __usad(x1, x2, x3);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__usad(x,y,z)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___usad_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
unsigned int x1 = GENERATE(0, 42, 0xFFFFFFFF);
unsigned int x2 = GENERATE(0, 42, 0xFFFFFFFF);
unsigned int x3 = GENERATE(0, 42, 0xFFFFFFFF);
__usad_kernel<<<1, 1>>>(y.ptr(), x1, x2, x3);
HIP_CHECK(hipDeviceSynchronize());
INFO("x1: " << x1);
INFO("x2: " << x2);
REQUIRE(y.ptr()[0] ==
(static_cast<unsigned int>(
std::abs(static_cast<long long>(x1) - static_cast<long long>(x2))) +
x3));
}
__global__ void __byte_perm(unsigned int* y, unsigned int x1, unsigned int x2, unsigned int s) {
y[0] = __byte_perm(x1, x2, s);
}
/**
* Test Description
* ------------------------
* - Sanity test for `__byte_perm(x,y,s)`.
*
* Test source
* ------------------------
* - unit/math/integer_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device___byte_perm_Sanity_Positive") {
LinearAllocGuard<unsigned int> y(LinearAllocs::hipMallocManaged, sizeof(unsigned int));
unsigned int bytes[] = {0x88, 0x99, 0xAA, 0xBB, 0xCC, 0xDD, 0xEE, 0xFF};
unsigned int x1 = (bytes[3] << 24) | (bytes[2] << 16) | (bytes[1] << 8) | bytes[0];
unsigned int x2 = (bytes[7] << 24) | (bytes[6] << 16) | (bytes[5] << 8) | bytes[4];
unsigned int s0 = GENERATE(0, 1);
unsigned int s1 = GENERATE(2, 3);
unsigned int s2 = GENERATE(4, 5);
unsigned int s3 = GENERATE(6, 7);
unsigned int s = (s3 << 12) | (s2 << 8) | (s1 << 4) | s0;
__byte_perm<<<1, 1>>>(y.ptr(), x1, x2, s);
HIP_CHECK(hipDeviceSynchronize());
unsigned int expected = (bytes[s3] << 24) | (bytes[s2] << 16) | (bytes[s1] << 8) | bytes[s0];
REQUIRE(y.ptr()[0] == expected);
}
@@ -0,0 +1,67 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(int* x) { int result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { int result = func_name(x); }
#define INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(long long int* x) { long long int result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { long long int result = func_name(x); }
#define INTRINSIC_BINARY_INT_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(int* x, int y) { int result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(int x, int* y) { int result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, int y) { int result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(int x, Dummy y) { int result = func_name(x, y); }
#define INTRINSIC_BINARY_LONGLONG_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(long long int* x, long long int y) { \
long long int result = func_name(x, y); \
} \
__global__ void func_name##_kernel_v2(long long int x, long long int* y) { \
long long int result = func_name##(x, y); \
} \
__global__ void func_name##_kernel_v3(Dummy x, long long int y) { \
long long int result = func_name##(x, y); \
} \
__global__ void func_name##_kernel_v4(long long int x, Dummy y) { \
long long int result = func_name##(x, y); \
}
INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(__brev)
INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(__clz)
INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(__ffs)
INTRINSIC_UNARY_INT_NEGATIVE_KERNELS(__popc)
INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(__brevll)
INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(__clzll)
INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(__ffsll)
INTRINSIC_UNARY_LONGLONG_NEGATIVE_KERNELS(__popcll)
INTRINSIC_BINARY_INT_NEGATIVE_KERNELS(__mul24)
+260
Wyświetl plik
@@ -0,0 +1,260 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "unary_common.hh"
#include "math_log_negative_kernels_rtc.hh"
/**
* @addtogroup LogMathFuncs LogMathFuncs
* @{
* @ingroup MathTest
*/
/********** Unary Functions **********/
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `logf(x)` for all possible inputs and `log(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::log(T)`. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(log, 1, 1)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for logf and log.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_log_logf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLog); }
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `log2f(x)` for all possible inputs and `log2(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::log2(T)`. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(log2, 1, 1)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for log2f and log2.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_log2_log2f_Negative_RTC") { NegativeTestRTCWrapper<4>(kLog2); }
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `log10f(x)` for all possible inputs and `log10(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::log10(T)`. The maximum ulp error for single
* precision is 2 and for double precision is 1.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(log10, 2, 1)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for log10f and log10.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_log10_log10f_Negative_RTC") { NegativeTestRTCWrapper<4>(kLog10); }
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `log1pf(x)` for all possible inputs and `log1p(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::log1p(T)`. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(log1p, 1, 1)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for log1pf and log1p.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_log1p_log1pf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLog1p); }
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `logb(x)` for all possible inputs and `logb(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::logb(T)`. The maximum ulp error is 0.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(logb, 0, 0)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for logbf and logb.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_logb_logbf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLogb); }
template <typename T>
__global__ void ilogb_kernel(int* const ys, const size_t num_xs, T* const xs) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
if constexpr (std::is_same_v<float, T>) {
ys[i] = ilogbf(xs[i]);
} else if constexpr (std::is_same_v<double, T>) {
ys[i] = ilogb(xs[i]);
}
}
}
template <typename T> int ilogb_ref(T arg) {
if (arg == 0) {
return std::numeric_limits<int>::min();
} else if (std::isnan(arg)) {
return std::numeric_limits<int>::min();
} else if (std::isinf(arg)) {
return std::numeric_limits<int>::max();
} else {
return std::ilogb(arg);
}
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `ilogbf(x)` for all possible inputs. The results are
* compared against reference function `int std::ilogb(double)`. The maximum ulp error is 0.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_ilogbf_Accuracy_Positive") {
UnarySinglePrecisionTest(ilogb_kernel<float>, ilogb_ref<double>,
EqValidatorBuilderFactory<int>());
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `ilogb(x)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are
* compared against reference function `int std::ilogb(long double)`. The maximum ulp error is 0.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_ilogb_Accuracy_Positive") {
UnaryDoublePrecisionTest(ilogb_kernel<double>, ilogb_ref<long double>,
EqValidatorBuilderFactory<int>());
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for ilogbf and ilogb.
*
* Test source
* ------------------------
* - unit/math/log_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_ilogb_ilogbf_Negative_RTC") { NegativeTestRTCWrapper<4>(kIlogb); }
+256
Wyświetl plik
@@ -0,0 +1,256 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include <cmd_options.hh>
#include <hip_test_common.hh>
#include <resource_guards.hh>
#include <hip/hip_cooperative_groups.h>
#include "Float16.hh"
#include "thread_pool.hh"
#include "validators.hh"
namespace cg = cooperative_groups;
template <typename T, typename U>
std::enable_if_t<std::conjunction_v<std::is_arithmetic<T>, std::is_arithmetic<U>>, std::ostream&>
operator<<(std::ostream& os, const std::pair<T, U>& p) {
const auto default_prec = os.precision();
return os << "<" << std::setprecision(std::numeric_limits<T>::max_digits10 - 1) << p.first << ", "
<< std::setprecision(std::numeric_limits<U>::max_digits10 - 1) << p.second << ">"
<< std::setprecision(default_prec);
}
template <typename T>
std::enable_if_t<sizeof(T) / sizeof(decltype(T().x)) == 2 && !std::is_same_v<T, __half2>, std::ostream&>
operator<<(std::ostream& os, const T& p) {
const auto default_prec = os.precision();
return os << "<" << std::setprecision(std::numeric_limits<decltype(T().x)>::max_digits10 - 1) << p.x << ", "
<< std::setprecision(std::numeric_limits<decltype(T().x)>::max_digits10 - 1) << p.y << ">"
<< std::setprecision(default_prec);
}
// This class represents a generic numerical accuracy math test. Template parameter T is the output
// type of the function being tested, and template parameter pack Ts represents the input types. The
// constructor takes a kernel with the signature void(T*, const size_t, Ts*...). The first kernel
// parameter is the output array, the second parameter is the number of outputs, and the rest of the
// parameters are arrays containing input values. The number of input arrays depends on the arity of
// the function being tested e.g. one input array for unary functions, two input arrays for binary
// functions, etc. The kernel threads take one element from each input array at the index
// corresponding to that thread, feed the input elements to the testee function, and store the
// result in the output array at the corresponding index.
//
// E.g. for a binary function the kernel would have the following signature:
// void kernel(float* y, const size_t n, float* x1, float* x2)
//
// The outputs would be calculated in parallel the following way:
// y[0] = testee(x1[0], x2[0])
// y[1] = testee(x1[1], x2[1])
// y[2] = testee(x1[2], x2[2])
// ...
//
// The constructor also takes max_num_args, which represents the maximum number of input values used
// for one kernel launch. The device memory for the input and output arrays is allocated based on
// that number.
template <typename T, typename... Ts> class MathTest {
public:
MathTest(void (*kernel)(T*, const size_t, Ts*...), const size_t max_num_args)
: kernel_{kernel},
xss_dev_(LinearAllocGuard<Ts>(LinearAllocs::hipMalloc, max_num_args * sizeof(Ts))...),
y_dev_{LinearAllocs::hipMalloc, max_num_args * sizeof(T)},
y_{LinearAllocs::hipHostMalloc, max_num_args * sizeof(T)} {}
// This method runs the test with the following steps:
// 1. Copy the values from the input arrays provided in the parameter pack xss to device memory
// 2. Launch the kernel using the configuration provided in grid_dims and block_dims
// 3. Copy the outputs back to host memory
// 4. Generate the reference values using ref_func and compare against the outputs using the
// validator provided by validator_builder
// 5. If non-type template parameter parallel is true, then step 4 is broken up into chunks of
// work that are done in parallel on the host.
template <bool parallel = true, typename RT, typename ValidatorBuilder, typename... RTs>
void Run(const ValidatorBuilder& validator_builder, const size_t grid_dims,
const size_t block_dims, RT (*const ref_func)(RTs...), const size_t num_args,
const Ts*... xss) {
fail_flag_.store(false);
error_info_.clear();
RunImpl<parallel>(validator_builder, grid_dims, block_dims, ref_func, num_args,
std::index_sequence_for<Ts...>{}, xss...);
}
private:
void (*kernel_)(T*, const size_t, Ts*...);
std::tuple<LinearAllocGuard<Ts>...> xss_dev_;
LinearAllocGuard<T> y_dev_;
LinearAllocGuard<T> y_;
std::atomic<bool> fail_flag_{false};
std::mutex mtx_;
std::string error_info_;
template <bool parallel, typename RT, typename ValidatorBuilder, typename... RTs, size_t... I>
void RunImpl(const ValidatorBuilder& validator_builder, const size_t grid_dim,
const size_t block_dim, RT (*const ref_func)(RTs...), const size_t num_args,
std::index_sequence<I...>, const Ts*... xss) {
const auto xss_tup = std::make_tuple(xss...);
constexpr auto f = [](auto dst, auto src, size_t size) {
HIP_CHECK(hipMemcpy(dst, src, size, hipMemcpyHostToDevice))
};
((f(std::get<I>(xss_dev_).ptr(), std::get<I>(xss_tup),
num_args * sizeof(*std::get<I>(xss_tup)))),
...);
kernel_<<<grid_dim, block_dim>>>(y_dev_.ptr(), num_args, std::get<I>(xss_dev_).ptr()...);
HIP_CHECK(hipGetLastError());
HIP_CHECK(hipMemcpy(y_.ptr(), y_dev_.ptr(), num_args * sizeof(T), hipMemcpyDeviceToHost));
HIP_CHECK(hipStreamSynchronize(nullptr));
if constexpr (!parallel) {
for (auto i = 0u; i < num_args; ++i) {
const auto actual_val = y_.ptr()[i];
const auto ref_val = static_cast<T>(ref_func(xss[i]...));
const auto validator = validator_builder(ref_val, xss[i]...);
if (!validator->match(actual_val)) {
const auto log = MakeLogMessage(actual_val, xss[i]...) + validator->describe() + "\n";
INFO(log);
REQUIRE(false);
}
}
return;
}
const auto task = [&, this](size_t iters, size_t base_idx) {
for (auto i = 0u; i < iters; ++i) {
if (fail_flag_.load(std::memory_order_relaxed)) return;
const auto actual_val = y_.ptr()[base_idx + i];
const auto ref_val = static_cast<T>(ref_func(xss[base_idx + i]...));
const auto validator = validator_builder(ref_val, xss[base_idx + i]...);
if (!validator->match(actual_val)) {
fail_flag_.store(true, std::memory_order_relaxed);
// Several threads might have passed the first check, but failed validation. On the
// chance of this happening, access to the string stream must be serialized.
const auto log =
MakeLogMessage(actual_val, xss[base_idx + i]...) + validator->describe() + "\n";
{
std::lock_guard lg{mtx_};
error_info_ += log;
}
return;
}
}
};
const auto task_count = thread_pool.thread_count();
const auto chunk_size = num_args / task_count;
const auto tail = num_args % task_count;
auto base_idx = 0u;
for (auto i = 0u; i < task_count; ++i) {
const auto iters = chunk_size + (i < tail);
thread_pool.Post([=, &task] { task(iters, base_idx); });
base_idx += iters;
}
thread_pool.Wait();
INFO(error_info_);
REQUIRE(!fail_flag_);
}
template <typename... Args> std::string MakeLogMessage(T actual_val, Args... args) {
std::stringstream ss;
ss << "Input value(s): " << std::scientific
<< std::setprecision(std::numeric_limits<T>::max_digits10 - 1);
((ss << " " << args), ...) << "\n" << actual_val << " ";
return ss.str();
}
};
template <typename T> struct RefType {};
template <> struct RefType<Float16> { using type = float; };
template <> struct RefType<float> { using type = double; };
template <> struct RefType<double> { using type = long double; };
template <typename T> using RefType_t = typename RefType<T>::type;
template <typename F> auto GetOccupancyMaxPotentialBlockSize(F kernel) {
int grid_size = 0, block_size = 0;
HIP_CHECK(hipOccupancyMaxPotentialBlockSize(&grid_size, &block_size, kernel, 0, 0));
return std::make_tuple(grid_size, block_size);
}
inline size_t GetMaxAllowedDeviceMemoryUsage() {
hipDeviceProp_t props;
HIP_CHECK(hipGetDeviceProperties(&props, 0));
return props.totalGlobalMem * (cmd_options.accuracy_max_memory * 0.01f);
}
inline uint64_t GetTestIterationCount() { return cmd_options.accuracy_iterations; }
template <typename T, typename... Ts> using kernel_sig = void (*)(T*, const size_t, Ts*...);
template <typename T, typename... Ts> using ref_sig = T (*)(Ts...);
template <int error_num> void NegativeTestRTCWrapper(const char* program_source) {
hiprtcProgram program{};
HIPRTC_CHECK(
hiprtcCreateProgram(&program, program_source, "math_test_rtc.cc", 0, nullptr, nullptr));
#if HT_AMD
std::string args = std::string("-ferror-limit=200");
const char* options[] = {args.c_str()};
hiprtcResult result{hiprtcCompileProgram(program, 1, options)};
#else
hiprtcResult result{hiprtcCompileProgram(program, 0, nullptr)};
#endif
// Get the compile log and count compiler error messages
size_t log_size{};
HIPRTC_CHECK(hiprtcGetProgramLogSize(program, &log_size));
std::string log(log_size, ' ');
HIPRTC_CHECK(hiprtcGetProgramLog(program, log.data()));
int error_count{0};
int expected_error_count{error_num};
std::string error_message{"error:"};
size_t n_pos = log.find(error_message, 0);
while (n_pos != std::string::npos) {
++error_count;
n_pos = log.find(error_message, n_pos + 1);
}
HIPRTC_CHECK(hiprtcDestroyProgram(&program));
HIPRTC_CHECK_ERROR(result, HIPRTC_ERROR_COMPILATION);
REQUIRE(error_count == expected_error_count);
}
@@ -0,0 +1,39 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL(func_name) \
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); } \
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
NEGATIVE_KERNELS_SHELL(log)
NEGATIVE_KERNELS_SHELL(log2)
NEGATIVE_KERNELS_SHELL(log10)
NEGATIVE_KERNELS_SHELL(log1p)
NEGATIVE_KERNELS_SHELL(logb)
NEGATIVE_KERNELS_SHELL(ilogb)
@@ -0,0 +1,96 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
/*
Negative kernels used for the math log negative Test Cases that are using RTC.
*/
static constexpr auto kLog{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void log_kernel_v1(double* x) { double result = log(x); }
__global__ void log_kernel_v2(Dummy x) { double result = log(x); }
__global__ void logf_kernel_v1(float* x) { float result = logf(x); }
__global__ void logf_kernel_v2(Dummy x) { float result = logf(x); }
)"};
static constexpr auto kLog2{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void log2_kernel_v1(double* x) { double result = log2(x); }
__global__ void log2_kernel_v2(Dummy x) { double result = log2(x); }
__global__ void log2f_kernel_v1(float* x) { float result = log2f(x); }
__global__ void log2f_kernel_v2(Dummy x) { float result = log2f(x); }
)"};
static constexpr auto kLog10{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void log10_kernel_v1(double* x) { double result = log10(x); }
__global__ void log10_kernel_v2(Dummy x) { double result = log10(x); }
__global__ void log10f_kernel_v1(float* x) { float result = log10f(x); }
__global__ void log10f_kernel_v2(Dummy x) { float result = log10f(x); }
)"};
static constexpr auto kLog1p{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void log1p_kernel_v1(double* x) { double result = log1p(x); }
__global__ void log1p_kernel_v2(Dummy x) { double result = log1p(x); }
__global__ void log1pf_kernel_v1(float* x) { float result = log1pf(x); }
__global__ void log1pf_kernel_v2(Dummy x) { float result = log1pf(x); }
)"};
static constexpr auto kLogb{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void logb_kernel_v1(double* x) { double result = logb(x); }
__global__ void logb_kernel_v2(Dummy x) { double result = logb(x); }
__global__ void logbf_kernel_v1(float* x) { float result = logbf(x); }
__global__ void logbf_kernel_v2(Dummy x) { float result = logbf(x); }
)"};
static constexpr auto kIlogb{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void ilogb_kernel_v1(double* x) { double result = ilogb(x); }
__global__ void ilogb_kernel_v2(Dummy x) { double result = ilogb(x); }
__global__ void ilogbf_kernel_v1(float* x) { float result = ilogbf(x); }
__global__ void ilogbf_kernel_v2(Dummy x) { float result = ilogbf(x); }
)"};
@@ -0,0 +1,92 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL_EXP(func_name) \
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); } \
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
#define NEGATIVE_KERNELS_SHELL_INT_2ND(func_name) \
__global__ void func_name##_kernel_v1(double* x, int e) { double result = func_name(x, e); } \
__global__ void func_name##_kernel_v2(Dummy x, int e) { double result = func_name(x, e); } \
__global__ void func_name##_kernel_v3(double x, int* e) { double result = func_name(x, e); } \
__global__ void func_name##_kernel_v4(double x, Dummy e) { double result = func_name(x, e); } \
__global__ void func_name##f_kernel_v1(float* x, int e) { float result = func_name##f(x, e); } \
__global__ void func_name##f_kernel_v2(Dummy x, int e) { float result = func_name##f(x, e); } \
__global__ void func_name##f_kernel_v3(float x, int* e) { float result = func_name##f(x, e); } \
__global__ void func_name##f_kernel_v4(float x, Dummy e) { float result = func_name##f(x, e); }
NEGATIVE_KERNELS_SHELL_EXP(exp)
NEGATIVE_KERNELS_SHELL_EXP(exp2)
NEGATIVE_KERNELS_SHELL_EXP(exp10)
NEGATIVE_KERNELS_SHELL_EXP(expm1)
__global__ void frexp_kernel_v1(double* x, int* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v2(Dummy x, int* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v3(double x, char* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v4(double x, short* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v5(double x, long* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v6(double x, long long* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v7(double x, float* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v8(double x, double* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v9(double x, Dummy* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v10(double x, const int* nptr) { double result = frexp(x, nptr); }
__global__ void frexpf_kernel_v1(float* x, int* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v2(Dummy x, int* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v3(float x, char* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v4(float x, short* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v5(float x, long* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v6(float x, long long* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v7(float x, float* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v8(float x, double* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v9(float x, Dummy* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v10(float x, const int* nptr) { float result = frexpf(x, nptr); }
NEGATIVE_KERNELS_SHELL_INT_2ND(ldexp)
__global__ void pow_kernel_v1(double* x, double e) { double result = pow(x, e); }
__global__ void pow_kernel_v2(Dummy x, double e) { double result = pow(x, e); }
__global__ void pow_kernel_v3(double x, double* e) { double result = pow(x, e); }
__global__ void pow_kernel_v4(double x, Dummy e) { double result = pow(x, e); }
__global__ void powf_kernel_v1(float* x, float e) { float result = powf(x, e); }
__global__ void powf_kernel_v2(Dummy x, float e) { float result = powf(x, e); }
__global__ void powf_kernel_v3(float x, float* e) { float result = powf(x, e); }
__global__ void powf_kernel_v4(float x, Dummy e) { float result = powf(x, e); }
NEGATIVE_KERNELS_SHELL_INT_2ND(powi)
NEGATIVE_KERNELS_SHELL_INT_2ND(scalbn)
__global__ void scalbln_kernel_v1(double* x, long int n) { double result = scalbln(x, n); }
__global__ void scalbln_kernel_v2(Dummy x, long int n) { double result = scalbln(x, n); }
__global__ void scalbln_kernel_v3(double x, long int* n) { double result = scalbln(x, n); }
__global__ void scalbln_kernel_v4(double x, Dummy n) { double result = scalbln(x, n); }
__global__ void scalblnf_kernel_v1(float* x, long int n) { float result = scalblnf(x, n); }
__global__ void scalblnf_kernel_v2(Dummy x, long int n) { float result = scalblnf(x, n); }
__global__ void scalblnf_kernel_v3(float x, long int* n) { float result = scalblnf(x, n); }
__global__ void scalblnf_kernel_v4(float x, Dummy n) { float result = scalblnf(x, n); }
@@ -0,0 +1,150 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
/*
Negative kernels used for the math pow negative Test Cases that are using RTC.
*/
static constexpr auto kExp{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void exp_kernel_v1(double* x) { double result = exp(x); }
__global__ void exp_kernel_v2(Dummy x) { double result = exp(x); }
__global__ void expf_kernel_v1(float* x) { float result = expf(x); }
__global__ void expf_kernel_v2(Dummy x) { float result = expf(x); }
)"};
static constexpr auto kExp2{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void exp2_kernel_v1(double* x) { double result = exp2(x); }
__global__ void exp2_kernel_v2(Dummy x) { double result = exp2(x); }
__global__ void exp2f_kernel_v1(float* x) { float result = exp2f(x); }
__global__ void exp2f_kernel_v2(Dummy x) { float result = exp2f(x); }
)"};
static constexpr auto kExp10{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void exp10_kernel_v1(double* x) { double result = exp10(x); }
__global__ void exp10_kernel_v2(Dummy x) { double result = exp10(x); }
__global__ void exp10f_kernel_v1(float* x) { float result = exp10f(x); }
__global__ void exp10f_kernel_v2(Dummy x) { float result = exp10f(x); }
)"};
static constexpr auto kExpm1{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void expm1_kernel_v1(double* x) { double result = expm1(x); }
__global__ void expm1_kernel_v2(Dummy x) { double result = expm1(x); }
__global__ void expm1f_kernel_v1(float* x) { float result = expm1f(x); }
__global__ void expm1f_kernel_v2(Dummy x) { float result = expm1f(x); }
)"};
static constexpr auto kFrexp{R"(
__global__ void frexp_kernel_v1(double* x, int* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v2(Dummy x, int* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v3(double x, char* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v4(double x, short* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v5(double x, long* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v6(double x, long long* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v7(double x, float* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v8(double x, double* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v9(double x, Dummy* nptr) { double result = frexp(x, nptr); }
__global__ void frexp_kernel_v10(double x, const int* nptr) { double result = frexp(x, nptr); }
__global__ void frexpf_kernel_v1(float* x, int* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v2(Dummy x, int* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v3(float x, char* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v4(float x, short* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v5(float x, long* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v6(float x, long long* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v7(float x, float* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v8(float x, double* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v9(float x, Dummy* nptr) { float result = frexpf(x, nptr); }
__global__ void frexpf_kernel_v10(float x, const int* nptr) { float result = frexpf(x, nptr); }
)"};
static constexpr auto kLdexp{R"(
__global__ void ldexp_kernel_v1(double* x, int e) { double result = ldexp(x, e); }
__global__ void ldexp_kernel_v2(Dummy x, int e) { double result = ldexp(x, e); }
__global__ void ldexp_kernel_v3(double x, int* e) { double result = ldexp(x, e); }
__global__ void ldexp_kernel_v4(double x, Dummy e) { double result = ldexp(x, e); }
__global__ void ldexpf_kernel_v1(float* x, int e) { float result = ldexpf(x, e); }
__global__ void ldexpf_kernel_v2(Dummy x, int e) { float result = ldexpf(x, e); }
__global__ void ldexpf_kernel_v3(float x, int* e) { float result = ldexpf(x, e); }
__global__ void ldexpf_kernel_v4(float x, Dummy e) { float result = ldexpf(x, e); }
)"};
static constexpr auto kPow{R"(
__global__ void pow_kernel_v1(double* x, double e) { double result = pow(x, e); }
__global__ void pow_kernel_v2(Dummy x, double e) { double result = pow(x, e); }
__global__ void pow_kernel_v3(double x, double* e) { double result = pow(x, e); }
__global__ void pow_kernel_v4(double x, Dummy e) { double result = pow(x, e); }
__global__ void powf_kernel_v1(float* x, float e) { float result = powf(x, e); }
__global__ void powf_kernel_v2(Dummy x, float e) { float result = powf(x, e); }
__global__ void powf_kernel_v3(float x, float* e) { float result = powf(x, e); }
__global__ void powf_kernel_v4(float x, Dummy e) { float result = powf(x, e); }
)"};
static constexpr auto kPowi{R"(
__global__ void powi_kernel_v1(double* x, int e) { double result = powi(x, e); }
__global__ void powi_kernel_v2(Dummy x, int e) { double result = powi(x, e); }
__global__ void powi_kernel_v3(double x, int* e) { double result = powi(x, e); }
__global__ void powi_kernel_v4(double x, Dummy e) { double result = powi(x, e); }
__global__ void powif_kernel_v1(float* x, int e) { float result = powif(x, e); }
__global__ void powif_kernel_v2(Dummy x, int e) { float result = powif(x, e); }
__global__ void powif_kernel_v3(float x, int* e) { float result = powif(x, e); }
__global__ void powif_kernel_v4(float x, Dummy e) { float result = powif(x, e); }
)"};
static constexpr auto kScalbn{R"(
__global__ void scalbn_kernel_v1(double* x, int e) { double result = scalbn(x, e); }
__global__ void scalbn_kernel_v2(Dummy x, int e) { double result = scalbn(x, e); }
__global__ void scalbn_kernel_v3(double x, int* e) { double result = scalbn(x, e); }
__global__ void scalbn_kernel_v4(double x, Dummy e) { double result = scalbn(x, e); }
__global__ void scalbnf_kernel_v1(float* x, int e) { float result = scalbnf(x, e); }
__global__ void scalbnf_kernel_v2(Dummy x, int e) { float result = scalbnf(x, e); }
__global__ void scalbnf_kernel_v3(float x, int* e) { float result = scalbnf(x, e); }
__global__ void scalbnf_kernel_v4(float x, Dummy e) { float result = scalbnf(x, e); }
)"};
static constexpr auto kScalbln{R"(
__global__ void scalbln_kernel_v1(double* x, long int n) { double result = scalbln(x, n); }
__global__ void scalbln_kernel_v2(Dummy x, long int n) { double result = scalbln(x, n); }
__global__ void scalbln_kernel_v3(double x, long int* n) { double result = scalbln(x, n); }
__global__ void scalbln_kernel_v4(double x, Dummy n) { double result = scalbln(x, n); }
__global__ void scalblnf_kernel_v1(float* x, long int n) { float result = scalblnf(x, n); }
__global__ void scalblnf_kernel_v2(Dummy x, long int n) { float result = scalblnf(x, n); }
__global__ void scalblnf_kernel_v3(float x, long int* n) { float result = scalblnf(x, n); }
__global__ void scalblnf_kernel_v4(float x, Dummy n) { float result = scalblnf(x, n); }
)"};
@@ -0,0 +1,113 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL(func_name) \
__global__ void func_name##_kernel_v1(double* x, double y) { auto result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(double x, double* y) { auto result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, double y) { auto result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(double x, Dummy y) { auto result = func_name(x, y); } \
__global__ void func_name##f_kernel_v1(float* x, float y) { auto result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v2(float x, float* y) { auto result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v3(Dummy x, float y) { auto result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v4(float x, Dummy y) { auto result = func_name##f(x, y); }
NEGATIVE_KERNELS_SHELL(fmod)
NEGATIVE_KERNELS_SHELL(remainder)
__global__ void remquo_kernel_v1(double* x, double y, int* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v2(Dummy x, double y, int* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v3(double x, double* y, int* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v4(double x, Dummy y, int* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v5(double x, double y, char* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v6(double x, double y, short* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v7(double x, double y, long* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v8(double x, double y, long long* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v9(double x, double y, float* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v10(double x, double y, double* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v11(double x, double y, Dummy* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v12(double x, double y, const int* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquof_kernel_v1(float* x, float y, int* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v2(Dummy x, float y, int* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v3(float x, float* y, int* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v4(float x, Dummy y, int* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v5(float x, float y, char* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v6(float x, float y, short* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v7(float x, float y, long* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v8(float x, float y, long long* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v9(float x, float y, float* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v10(float x, float y, double* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v11(float x, float y, Dummy* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v12(float x, float y, const int* quo) {
auto result = remquof(x, y, quo);
}
__global__ void modf_kernel_v1(double* x, double* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v2(Dummy x, double* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v3(double x, int* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v4(double x, char* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v5(double x, short* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v6(double x, long* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v7(double x, long long* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v8(double x, float* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v9(double x, Dummy* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v10(double x, const double* iptr) { auto result = modf(x, iptr); }
__global__ void modff_kernel_v1(float* x, float* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v2(Dummy x, float* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v3(float x, int* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v4(float x, char* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v5(float x, short* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v6(float x, long* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v7(float x, long long* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v8(float x, double* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v9(float x, Dummy* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v10(float x, const float* iptr) { auto result = modff(x, iptr); }
NEGATIVE_KERNELS_SHELL(fdim)
@@ -0,0 +1,276 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
/*
Negative kernels used for the math remainder and rounding negative Test Cases that are using RTC.
*/
static constexpr auto kTrunc{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void trunc_kernel_v1(double* x) { auto result = trunc(x); }
__global__ void trunc_kernel_v2(Dummy x) { auto result = trunc(x); }
__global__ void truncf_kernel_v1(float* x) { auto result = truncf(x); }
__global__ void truncf_kernel_v2(Dummy x) { auto result = truncf(x); }
)"};
static constexpr auto kRound{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void round_kernel_v1(double* x) { auto result = round(x); }
__global__ void round_kernel_v2(Dummy x) { auto result = round(x); }
__global__ void roundf_kernel_v1(float* x) { auto result = roundf(x); }
__global__ void roundf_kernel_v2(Dummy x) { auto result = roundf(x); }
)"};
static constexpr auto kRint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void rint_kernel_v1(double* x) { auto result = rint(x); }
__global__ void rint_kernel_v2(Dummy x) { auto result = rint(x); }
__global__ void rintf_kernel_v1(float* x) { auto result = rintf(x); }
__global__ void rintf_kernel_v2(Dummy x) { auto result = rintf(x); }
)"};
static constexpr auto kNearbyint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void nearbyint_kernel_v1(double* x) { auto result = nearbyint(x); }
__global__ void nearbyint_kernel_v2(Dummy x) { auto result = nearbyint(x); }
__global__ void nearbyintf_kernel_v1(float* x) { auto result = nearbyintf(x); }
__global__ void nearbyintf_kernel_v2(Dummy x) { auto result = nearbyintf(x); }
)"};
static constexpr auto kCeil{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void ceil_kernel_v1(double* x) { auto result = ceil(x); }
__global__ void ceil_kernel_v2(Dummy x) { auto result = ceil(x); }
__global__ void ceilf_kernel_v1(float* x) { auto result = ceilf(x); }
__global__ void ceilf_kernel_v2(Dummy x) { auto result = ceilf(x); }
)"};
static constexpr auto kFloor{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void floor_kernel_v1(double* x) { auto result = floor(x); }
__global__ void floor_kernel_v2(Dummy x) { auto result = floor(x); }
__global__ void floorf_kernel_v1(float* x) { auto result = floorf(x); }
__global__ void floorf_kernel_v2(Dummy x) { auto result = floorf(x); }
)"};
static constexpr auto kLrint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void lrint_kernel_v1(double* x) { auto result = lrint(x); }
__global__ void lrint_kernel_v2(Dummy x) { auto result = lrint(x); }
__global__ void lrintf_kernel_v1(float* x) { auto result = lrintf(x); }
__global__ void lrintf_kernel_v2(Dummy x) { auto result = lrintf(x); }
)"};
static constexpr auto kLround{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void lround_kernel_v1(double* x) { auto result = lround(x); }
__global__ void lround_kernel_v2(Dummy x) { auto result = lround(x); }
__global__ void lroundf_kernel_v1(float* x) { auto result = lroundf(x); }
__global__ void lroundf_kernel_v2(Dummy x) { auto result = lroundf(x); }
)"};
static constexpr auto kLlrint{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void llrint_kernel_v1(double* x) { auto result = llrint(x); }
__global__ void llrint_kernel_v2(Dummy x) { auto result = llrint(x); }
__global__ void llrintf_kernel_v1(float* x) { auto result = llrintf(x); }
__global__ void llrintf_kernel_v2(Dummy x) { auto result = llrintf(x); }
)"};
static constexpr auto kLlround{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void llround_kernel_v1(double* x) { auto result = llround(x); }
__global__ void llround_kernel_v2(Dummy x) { auto result = llround(x); }
__global__ void llroundf_kernel_v1(float* x) { auto result = llroundf(x); }
__global__ void llroundf_kernel_v2(Dummy x) { auto result = llroundf(x); }
)"};
static constexpr auto kFmod{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void fmod_kernel_v1(double* x, double y) { auto result = fmod(x, y); }
__global__ void fmod_kernel_v2(double x, double* y) { auto result = fmod(x, y); }
__global__ void fmod_kernel_v3(Dummy x, double y) { auto result = fmod(x, y); }
__global__ void fmod_kernel_v4(double x, Dummy y) { auto result = fmod(x, y); }
__global__ void fmodf_kernel_v1(float* x, float y) { auto result = fmodf(x, y); }
__global__ void fmodf_kernel_v2(float x, float* y) { auto result = fmodf(x, y); }
__global__ void fmodf_kernel_v3(Dummy x, float y) { auto result = fmodf(x, y); }
__global__ void fmodf_kernel_v4(float x, Dummy y) { auto result = fmodf(x, y); }
)"};
static constexpr auto kRemainder{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void remainder_kernel_v1(double* x, double y) { auto result = remainder(x, y); }
__global__ void remainder_kernel_v2(double x, double* y) { auto result = remainder(x, y); }
__global__ void remainder_kernel_v3(Dummy x, double y) { auto result = remainder(x, y); }
__global__ void remainder_kernel_v4(double x, Dummy y) { auto result = remainder(x, y); }
__global__ void remainderf_kernel_v1(float* x, float y) { auto result = remainderf(x, y); }
__global__ void remainderf_kernel_v2(float x, float* y) { auto result = remainderf(x, y); }
__global__ void remainderf_kernel_v3(Dummy x, float y) { auto result = remainderf(x, y); }
__global__ void remainderf_kernel_v4(float x, Dummy y) { auto result = remainderf(x, y); }
)"};
static constexpr auto kRemquo{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void remquo_kernel_v1(double* x, double y, int* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v2(Dummy x, double y, int* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v3(double x, double* y, int* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v4(double x, Dummy y, int* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v5(double x, double y, char* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v6(double x, double y, short* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v7(double x, double y, long* quo) { auto result = remquo(x, y, quo); }
__global__ void remquo_kernel_v8(double x, double y, long long* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v9(double x, double y, float* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v10(double x, double y, double* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v11(double x, double y, Dummy* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquo_kernel_v12(double x, double y, const int* quo) {
auto result = remquo(x, y, quo);
}
__global__ void remquof_kernel_v1(float* x, float y, int* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v2(Dummy x, float y, int* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v3(float x, float* y, int* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v4(float x, Dummy y, int* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v5(float x, float y, char* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v6(float x, float y, short* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v7(float x, float y, long* quo) { auto result = remquof(x, y, quo); }
__global__ void remquof_kernel_v8(float x, float y, long long* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v9(float x, float y, float* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v10(float x, float y, double* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v11(float x, float y, Dummy* quo) {
auto result = remquof(x, y, quo);
}
__global__ void remquof_kernel_v12(float x, float y, const int* quo) {
auto result = remquof(x, y, quo);
}
)"};
static constexpr auto kModf{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void modf_kernel_v1(double* x, double* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v2(Dummy x, double* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v3(double x, int* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v4(double x, char* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v5(double x, short* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v6(double x, long* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v7(double x, long long* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v8(double x, float* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v9(double x, Dummy* iptr) { auto result = modf(x, iptr); }
__global__ void modf_kernel_v10(double x, const double* iptr) { auto result = modf(x, iptr); }
__global__ void modff_kernel_v1(float* x, float* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v2(Dummy x, float* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v3(float x, int* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v4(float x, char* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v5(float x, short* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v6(float x, long* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v7(float x, long long* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v8(float x, double* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v9(float x, Dummy* iptr) { auto result = modff(x, iptr); }
__global__ void modff_kernel_v10(float x, const float* iptr) { auto result = modff(x, iptr); }
)"};
static constexpr auto kFdim{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void fdim_kernel_v1(double* x, double y) { auto result = fdim(x, y); }
__global__ void fdim_kernel_v2(double x, double* y) { auto result = fdim(x, y); }
__global__ void fdim_kernel_v3(Dummy x, double y) { auto result = fdim(x, y); }
__global__ void fdim_kernel_v4(double x, Dummy y) { auto result = fdim(x, y); }
__global__ void fdimf_kernel_v1(float* x, float y) { auto result = fdimf(x, y); }
__global__ void fdimf_kernel_v2(float x, float* y) { auto result = fdimf(x, y); }
__global__ void fdimf_kernel_v3(Dummy x, float y) { auto result = fdimf(x, y); }
__global__ void fdimf_kernel_v4(float x, Dummy y) { auto result = fdimf(x, y); }
)"};
@@ -0,0 +1,107 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL_ONE_ARG(func_name) \
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); } \
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
#define NEGATIVE_KERNELS_SHELL_TWO_ARGS(func_name) \
__global__ void func_name##_kernel_v1(double* x, double y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(double x, double* y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, double y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(double x, Dummy y) { double result = func_name(x, y); } \
__global__ void func_name##f_kernel_v1(float* x, float y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v2(float x, float* y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v3(Dummy x, float y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v4(float x, Dummy y) { float result = func_name##f(x, y); }
#define NEGATIVE_KERNELS_SHELL_ARRAY_ARG(func_name) \
__global__ void func_name##_kernel_v1(int* dim, const double* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##_kernel_v2(Dummy dim, const double* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##_kernel_v3(int dim, const int* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##_kernel_v4(int dim, const char* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##_kernel_v5(int dim, const short* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##_kernel_v6(int dim, const long* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##_kernel_v7(int dim, const long long* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##_kernel_v8(int dim, const float* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##_kernel_v9(int dim, const Dummy* a) { \
double result = func_name(dim, a); \
} \
__global__ void func_name##f_kernel_v1(int* dim, const float* a) { \
float result = func_name##f(dim, a); \
} \
__global__ void func_name##f_kernel_v2(Dummy dim, const float* a) { \
float result = func_name##f(dim, a); \
} \
__global__ void func_name##f_kernel_v3(int dim, const int* a) { \
float result = func_name##f(dim, a); \
} \
__global__ void func_name##f_kernel_v4(int dim, const char* a) { \
float result = func_name##f(dim, a); \
} \
__global__ void func_name##f_kernel_v5(int dim, const short* a) { \
float result = func_name##f(dim, a); \
} \
__global__ void func_name##f_kernel_v6(int dim, const long* a) { \
float result = func_name##f(dim, a); \
} \
__global__ void func_name##f_kernel_v7(int dim, const long long* a) { \
float result = func_name##f(dim, a); \
} \
__global__ void func_name##f_kernel_v8(int dim, const double* a) { \
float result = func_name##f(dim, a); \
} \
__global__ void func_name##f_kernel_v9(int dim, const Dummy* a) { \
double result = func_name##f(dim, a); \
}
NEGATIVE_KERNELS_SHELL_ONE_ARG(sqrt)
NEGATIVE_KERNELS_SHELL_ONE_ARG(rsqrt)
NEGATIVE_KERNELS_SHELL_ONE_ARG(cbrt)
NEGATIVE_KERNELS_SHELL_ONE_ARG(rcbrt)
NEGATIVE_KERNELS_SHELL_TWO_ARGS(hypot)
NEGATIVE_KERNELS_SHELL_TWO_ARGS(rhypot)
NEGATIVE_KERNELS_SHELL_ARRAY_ARG(norm)
NEGATIVE_KERNELS_SHELL_ARRAY_ARG(rnorm)
@@ -0,0 +1,119 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL_THREE_ARGS(func_name) \
__global__ void func_name##_kernel_v1(double* x, double y, double z) { \
double result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v2(double x, double* y, double z) { \
double result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v3(double x, double y, double* z) { \
double result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v4(Dummy x, double y, double z) { \
double result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v5(double x, Dummy y, double z) { \
double result = func_name(x, y, z); \
} \
__global__ void func_name##_kernel_v6(double x, double y, Dummy z) { \
double result = func_name(x, y, z); \
} \
__global__ void func_name##f_kernel_v1(float* x, float y, float z) { \
float result = func_name##f(x, y, z); \
} \
__global__ void func_name##f_kernel_v2(float x, float* y, float z) { \
float result = func_name##f(x, y, z); \
} \
__global__ void func_name##f_kernel_v3(float x, float y, float* z) { \
float result = func_name##f(x, y, z); \
} \
__global__ void func_name##f_kernel_v4(Dummy x, float y, float z) { \
float result = func_name##f(x, y, z); \
} \
__global__ void func_name##f_kernel_v5(float x, Dummy y, float z) { \
float result = func_name##f(x, y, z); \
} \
__global__ void func_name##f_kernel_v6(float x, float y, Dummy z) { \
float result = func_name##f(x, y, z); \
}
#define NEGATIVE_KERNELS_SHELL_FOUR_ARGS(func_name) \
__global__ void func_name##_kernel_v1(double* x, double y, double z, double w) { \
double result = func_name(x, y, z, w); \
} \
__global__ void func_name##_kernel_v2(double x, double* y, double z, double w) { \
double result = func_name(x, y, z, w); \
} \
__global__ void func_name##_kernel_v3(double x, double y, double* z, double w) { \
double result = func_name(x, y, z, w); \
} \
__global__ void func_name##_kernel_v4(double x, double y, double z, double* w) { \
double result = func_name(x, y, z, w); \
} \
__global__ void func_name##_kernel_v5(Dummy x, double y, double z, double w) { \
double result = func_name(x, y, z, w); \
} \
__global__ void func_name##_kernel_v6(double x, Dummy y, double z, double w) { \
double result = func_name(x, y, z, w); \
} \
__global__ void func_name##_kernel_v7(double x, double y, Dummy z, double w) { \
double result = func_name(x, y, z, w); \
} \
__global__ void func_name##_kernel_v8(double x, double y, double z, Dummy w) { \
double result = func_name(x, y, z, w); \
} \
__global__ void func_name##f_kernel_v1(float* x, float y, float z, float w) { \
float result = func_name##f(x, y, z, w); \
} \
__global__ void func_name##f_kernel_v2(float x, float* y, float z, float w) { \
float result = func_name##f(x, y, z, w); \
} \
__global__ void func_name##f_kernel_v3(float x, float y, float* z, float w) { \
float result = func_name##f(x, y, z, w); \
} \
__global__ void func_name##f_kernel_v4(float x, float y, float z, float* w) { \
float result = func_name##f(x, y, z, w); \
} \
__global__ void func_name##f_kernel_v5(Dummy x, float y, float z, float w) { \
float result = func_name##f(x, y, z, w); \
} \
__global__ void func_name##f_kernel_v6(float x, Dummy y, float z, float w) { \
float result = func_name##f(x, y, z, w); \
} \
__global__ void func_name##f_kernel_v7(float x, float y, Dummy z, float w) { \
float result = func_name##f(x, y, z, w); \
} \
__global__ void func_name##f_kernel_v8(float x, float y, float z, Dummy w) { \
float result = func_name##f(x, y, z, w); \
}
NEGATIVE_KERNELS_SHELL_THREE_ARGS(norm3d)
NEGATIVE_KERNELS_SHELL_THREE_ARGS(rnorm3d)
NEGATIVE_KERNELS_SHELL_FOUR_ARGS(norm4d)
NEGATIVE_KERNELS_SHELL_FOUR_ARGS(rnorm4d)
@@ -0,0 +1,428 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
/*
Negative kernels used for the math root negative Test Cases that are using RTC.
*/
static constexpr auto kSqrt{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void sqrt_kernel_v1(double* x) { double result = sqrt(x); }
__global__ void sqrt_kernel_v2(Dummy x) { double result = sqrt(x); }
__global__ void sqrtf_kernel_v1(float* x) { float result = sqrtf(x); }
__global__ void sqrtf_kernel_v2(Dummy x) { float result = sqrtf(x); }
)"};
static constexpr auto kRsqrt{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void rsqrt_kernel_v1(double* x) { double result = rsqrt(x); }
__global__ void rsqrt_kernel_v2(Dummy x) { double result = rsqrt(x); }
__global__ void rsqrtf_kernel_v1(float* x) { float result = rsqrtf(x); }
__global__ void rsqrtf_kernel_v2(Dummy x) { float result = rsqrtf(x); }
)"};
static constexpr auto kCbrt{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void cbrt_kernel_v1(double* x) { double result = cbrt(x); }
__global__ void cbrt_kernel_v2(Dummy x) { double result = cbrt(x); }
__global__ void cbrtf_kernel_v1(float* x) { float result = cbrtf(x); }
__global__ void cbrtf_kernel_v2(Dummy x) { float result = cbrtf(x); }
)"};
static constexpr auto kRcbrt{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void rcbrt_kernel_v1(double* x) { double result = rcbrt(x); }
__global__ void rcbrt_kernel_v2(Dummy x) { double result = rcbrt(x); }
__global__ void rcbrtf_kernel_v1(float* x) { float result = rcbrtf(x); }
__global__ void rcbrtf_kernel_v2(Dummy x) { float result = rcbrtf(x); }
)"};
static constexpr auto kHypot{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void hypot_kernel_v1(double* x, double y) { double result = hypot(x, y); }
__global__ void hypot_kernel_v2(double x, double* y) { double result = hypot(x, y); }
__global__ void hypot_kernel_v3(Dummy x, double y) { double result = hypot(x, y); }
__global__ void hypot_kernel_v4(double x, Dummy y) { double result = hypot(x, y); }
__global__ void hypotf_kernel_v1(float* x, float y) { float result = hypotf(x, y); }
__global__ void hypotf_kernel_v2(float x, float* y) { float result = hypotf(x, y); }
__global__ void hypotf_kernel_v3(Dummy x, float y) { float result = hypotf(x, y); }
__global__ void hypotf_kernel_v4(float x, Dummy y) { float result = hypotf(x, y); }
)"};
static constexpr auto kRhypot{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void rhypot_kernel_v1(double* x, double y) { double result = rhypot(x, y); }
__global__ void rhypot_kernel_v2(double x, double* y) { double result = rhypot(x, y); }
__global__ void rhypot_kernel_v3(Dummy x, double y) { double result = rhypot(x, y); }
__global__ void rhypot_kernel_v4(double x, Dummy y) { double result = rhypot(x, y); }
__global__ void rhypotf_kernel_v1(float* x, float y) { float result = rhypotf(x, y); }
__global__ void rhypotf_kernel_v2(float x, float* y) { float result = rhypotf(x, y); }
__global__ void rhypotf_kernel_v3(Dummy x, float y) { float result = rhypotf(x, y); }
__global__ void rhypotf_kernel_v4(float x, Dummy y) { float result = rhypotf(x, y); }
)"};
static constexpr auto kNorm3D{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void norm3d_kernel_v1(double* x, double y, double z) {
double result = norm3d(x, y, z);
}
__global__ void norm3d_kernel_v2(double x, double* y, double z) {
double result = norm3d(x, y, z);
}
__global__ void norm3d_kernel_v3(double x, double y, double* z) {
double result = norm3d(x, y, z);
}
__global__ void norm3d_kernel_v4(Dummy x, double y, double z) {
double result = norm3d(x, y, z);
}
__global__ void norm3d_kernel_v5(double x, Dummy y, double z) {
double result = norm3d(x, y, z);
}
__global__ void norm3d_kernel_v6(double x, double y, Dummy z) {
double result = norm3d(x, y, z);
}
__global__ void norm3df_kernel_v1(float* x, float y, float z) {
float result = norm3df(x, y, z);
}
__global__ void norm3df_kernel_v2(float x, float* y, float z) {
float result = norm3df(x, y, z);
}
__global__ void norm3df_kernel_v3(float x, float y, float* z) {
float result = norm3df(x, y, z);
}
__global__ void norm3df_kernel_v4(Dummy x, float y, float z) {
float result = norm3df(x, y, z);
}
__global__ void norm3df_kernel_v5(float x, Dummy y, float z) {
float result = norm3df(x, y, z);
}
__global__ void norm3df_kernel_v6(float x, float y, Dummy z) {
float result = norm3df(x, y, z);
}
)"};
static constexpr auto kRnorm3D{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void rnorm3d_kernel_v1(double* x, double y, double z) {
double result = rnorm3d(x, y, z);
}
__global__ void rnorm3d_kernel_v2(double x, double* y, double z) {
double result = rnorm3d(x, y, z);
}
__global__ void rnorm3d_kernel_v3(double x, double y, double* z) {
double result = rnorm3d(x, y, z);
}
__global__ void rnorm3d_kernel_v4(Dummy x, double y, double z) {
double result = rnorm3d(x, y, z);
}
__global__ void rnorm3d_kernel_v5(double x, Dummy y, double z) {
double result = rnorm3d(x, y, z);
}
__global__ void rnorm3d_kernel_v6(double x, double y, Dummy z) {
double result = rnorm3d(x, y, z);
}
__global__ void rnorm3df_kernel_v1(float* x, float y, float z) {
float result = rnorm3df(x, y, z);
}
__global__ void rnorm3df_kernel_v2(float x, float* y, float z) {
float result = rnorm3df(x, y, z);
}
__global__ void rnorm3df_kernel_v3(float x, float y, float* z) {
float result = rnorm3df(x, y, z);
}
__global__ void rnorm3df_kernel_v4(Dummy x, float y, float z) {
float result = rnorm3df(x, y, z);
}
__global__ void rnorm3df_kernel_v5(float x, Dummy y, float z) {
float result = rnorm3df(x, y, z);
}
__global__ void rnorm3df_kernel_v6(float x, float y, Dummy z) {
float result = rnorm3df(x, y, z);
}
)"};
static constexpr auto kNorm4D{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void norm4d_kernel_v1(double* x, double y, double z, double w) {
double result = norm4d(x, y, z, w);
}
__global__ void norm4d_kernel_v2(double x, double* y, double z, double w) {
double result = norm4d(x, y, z, w);
}
__global__ void norm4d_kernel_v3(double x, double y, double* z, double w) {
double result = norm4d(x, y, z, w);
}
__global__ void norm4d_kernel_v4(double x, double y, double z, double* w) {
double result = norm4d(x, y, z, w);
}
__global__ void norm4d_kernel_v5(Dummy x, double y, double z, double w) {
double result = norm4d(x, y, z, w);
}
__global__ void norm4d_kernel_v6(double x, Dummy y, double z, double w) {
double result = norm4d(x, y, z, w);
}
__global__ void norm4d_kernel_v7(double x, double y, Dummy z, double w) {
double result = norm4d(x, y, z, w);
}
__global__ void norm4d_kernel_v8(double x, double y, double z, Dummy w) {
double result = norm4d(x, y, z, w);
}
__global__ void norm4df_kernel_v1(float* x, float y, float z, float w) {
float result = norm4df(x, y, z, w);
}
__global__ void norm4df_kernel_v2(float x, float* y, float z, float w) {
float result = norm4df(x, y, z, w);
}
__global__ void norm4df_kernel_v3(float x, float y, float* z, float w) {
float result = norm4df(x, y, z, w);
}
__global__ void norm4df_kernel_v4(float x, float y, float z, float* w) {
float result = norm4df(x, y, z, w);
}
__global__ void norm4df_kernel_v5(Dummy x, float y, float z, float w) {
float result = norm4df(x, y, z, w);
}
__global__ void norm4df_kernel_v6(float x, Dummy y, float z, float w) {
float result = norm4df(x, y, z, w);
}
__global__ void norm4df_kernel_v7(float x, float y, Dummy z, float w) {
float result = norm4df(x, y, z, w);
}
__global__ void norm4df_kernel_v8(float x, float y, float z, Dummy w) {
float result = norm4df(x, y, z, w);
}
)"};
static constexpr auto kRnorm4D{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void rnorm4d_kernel_v1(double* x, double y, double z, double w) {
double result = rnorm4d(x, y, z, w);
}
__global__ void rnorm4d_kernel_v2(double x, double* y, double z, double w) {
double result = rnorm4d(x, y, z, w);
}
__global__ void rnorm4d_kernel_v3(double x, double y, double* z, double w) {
double result = rnorm4d(x, y, z, w);
}
__global__ void rnorm4d_kernel_v4(double x, double y, double z, double* w) {
double result = rnorm4d(x, y, z, w);
}
__global__ void rnorm4d_kernel_v5(Dummy x, double y, double z, double w) {
double result = rnorm4d(x, y, z, w);
}
__global__ void rnorm4d_kernel_v6(double x, Dummy y, double z, double w) {
double result = rnorm4d(x, y, z, w);
}
__global__ void rnorm4d_kernel_v7(double x, double y, Dummy z, double w) {
double result = rnorm4d(x, y, z, w);
}
__global__ void rnorm4d_kernel_v8(double x, double y, double z, Dummy w) {
double result = rnorm4d(x, y, z, w);
}
__global__ void rnorm4df_kernel_v1(float* x, float y, float z, float w) {
float result = rnorm4df(x, y, z, w);
}
__global__ void rnorm4df_kernel_v2(float x, float* y, float z, float w) {
float result = rnorm4df(x, y, z, w);
}
__global__ void rnorm4df_kernel_v3(float x, float y, float* z, float w) {
float result = rnorm4df(x, y, z, w);
}
__global__ void rnorm4df_kernel_v4(float x, float y, float z, float* w) {
float result = rnorm4df(x, y, z, w);
}
__global__ void rnorm4df_kernel_v5(Dummy x, float y, float z, float w) {
float result = rnorm4df(x, y, z, w);
}
__global__ void rnorm4df_kernel_v6(float x, Dummy y, float z, float w) {
float result = rnorm4df(x, y, z, w);
}
__global__ void rnorm4df_kernel_v7(float x, float y, Dummy z, float w) {
float result = rnorm4df(x, y, z, w);
}
__global__ void rnorm4df_kernel_v8(float x, float y, float z, Dummy w) {
float result = rnorm4df(x, y, z, w);
}
)"};
static constexpr auto kNorm{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void norm_kernel_v1(int* dim, const double* a) {
double result = norm(dim, a);
}
__global__ void norm_kernel_v2(Dummy dim, const double* a) {
double result = norm(dim, a);
}
__global__ void norm_kernel_v3(int dim, const int* a) {
double result = norm(dim, a);
}
__global__ void norm_kernel_v4(int dim, const char* a) {
double result = norm(dim, a);
}
__global__ void norm_kernel_v5(int dim, const short* a) {
double result = norm(dim, a);
}
__global__ void norm_kernel_v6(int dim, const long* a) {
double result = norm(dim, a);
}
__global__ void norm_kernel_v7(int dim, const long long* a) {
double result = norm(dim, a);
}
__global__ void norm_kernel_v8(int dim, const float* a) {
double result = norm(dim, a);
}
__global__ void norm_kernel_v9(int dim, const Dummy* a) {
double result = norm(dim, a);
}
__global__ void normf_kernel_v1(int* dim, const float* a) {
float result = normf(dim, a);
}
__global__ void normf_kernel_v2(Dummy dim, const float* a) {
float result = normf(dim, a);
}
__global__ void normf_kernel_v3(int dim, const int* a) {
float result = normf(dim, a);
}
__global__ void normf_kernel_v4(int dim, const char* a) {
float result = normf(dim, a);
}
__global__ void normf_kernel_v5(int dim, const short* a) {
float result = normf(dim, a);
}
__global__ void normf_kernel_v6(int dim, const long* a) {
float result = normf(dim, a);
}
__global__ void normf_kernel_v7(int dim, const long long* a) {
float result = normf(dim, a);
}
__global__ void normf_kernel_v8(int dim, const double* a) {
float result = normf(dim, a);
}
__global__ void normf_kernel_v9(int dim, const Dummy* a) {
double result = normf(dim, a);
}
)"};
static constexpr auto kRnorm{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void rnorm_kernel_v1(int* dim, const double* a) {
double result = rnorm(dim, a);
}
__global__ void rnorm_kernel_v2(Dummy dim, const double* a) {
double result = rnorm(dim, a);
}
__global__ void rnorm_kernel_v3(int dim, const int* a) {
double result = rnorm(dim, a);
}
__global__ void rnorm_kernel_v4(int dim, const char* a) {
double result = rnorm(dim, a);
}
__global__ void rnorm_kernel_v5(int dim, const short* a) {
double result = rnorm(dim, a);
}
__global__ void rnorm_kernel_v6(int dim, const long* a) {
double result = rnorm(dim, a);
}
__global__ void rnorm_kernel_v7(int dim, const long long* a) {
double result = rnorm(dim, a);
}
__global__ void rnorm_kernel_v8(int dim, const float* a) {
double result = rnorm(dim, a);
}
__global__ void rnorm_kernel_v9(int dim, const Dummy* a) {
double result = rnorm(dim, a);
}
__global__ void rnormf_kernel_v1(int* dim, const float* a) {
float result = rnormf(dim, a);
}
__global__ void rnormf_kernel_v2(Dummy dim, const float* a) {
float result = rnormf(dim, a);
}
__global__ void rnormf_kernel_v3(int dim, const int* a) {
float result = rnormf(dim, a);
}
__global__ void rnormf_kernel_v4(int dim, const char* a) {
float result = rnormf(dim, a);
}
__global__ void rnormf_kernel_v5(int dim, const short* a) {
float result = rnormf(dim, a);
}
__global__ void rnormf_kernel_v6(int dim, const long* a) {
float result = rnormf(dim, a);
}
__global__ void rnormf_kernel_v7(int dim, const long long* a) {
float result = rnormf(dim, a);
}
__global__ void rnormf_kernel_v8(int dim, const double* a) {
float result = rnormf(dim, a);
}
__global__ void rnormf_kernel_v9(int dim, const Dummy* a) {
double result = rnormf(dim, a);
}
)"};
@@ -0,0 +1,43 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL(func_name) \
__global__ void func_name##_kernel_v1(double* x) { auto result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { auto result = func_name(x); } \
__global__ void func_name##f_kernel_v1(float* x) { auto result = func_name##f(x); } \
__global__ void func_name##f_kernel_v2(Dummy x) { auto result = func_name##f(x); }
NEGATIVE_KERNELS_SHELL(trunc)
NEGATIVE_KERNELS_SHELL(round)
NEGATIVE_KERNELS_SHELL(rint)
NEGATIVE_KERNELS_SHELL(nearbyint)
NEGATIVE_KERNELS_SHELL(ceil)
NEGATIVE_KERNELS_SHELL(floor)
NEGATIVE_KERNELS_SHELL(lrint)
NEGATIVE_KERNELS_SHELL(lround)
NEGATIVE_KERNELS_SHELL(llrint)
NEGATIVE_KERNELS_SHELL(llround)
@@ -0,0 +1,60 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define NEGATIVE_KERNELS_SHELL_ONE_ARG(func_name) \
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); } \
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
#define NEGATIVE_KERNELS_SHELL_TWO_ARGS(func_name) \
__global__ void func_name##_kernel_v1(int* x, double y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(int x, double* y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, double y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(int x, Dummy y) { double result = func_name(x, y); } \
__global__ void func_name##f_kernel_v1(int* x, float y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v2(int x, float* y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v3(Dummy x, float y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v4(int x, Dummy y) { float result = func_name##f(x, y); }
NEGATIVE_KERNELS_SHELL_ONE_ARG(erf)
NEGATIVE_KERNELS_SHELL_ONE_ARG(erfc)
NEGATIVE_KERNELS_SHELL_ONE_ARG(erfinv)
NEGATIVE_KERNELS_SHELL_ONE_ARG(erfcinv)
NEGATIVE_KERNELS_SHELL_ONE_ARG(erfcx)
NEGATIVE_KERNELS_SHELL_ONE_ARG(normcdf)
NEGATIVE_KERNELS_SHELL_ONE_ARG(normcdfinv)
NEGATIVE_KERNELS_SHELL_ONE_ARG(lgamma)
NEGATIVE_KERNELS_SHELL_ONE_ARG(tgamma)
NEGATIVE_KERNELS_SHELL_ONE_ARG(j0)
NEGATIVE_KERNELS_SHELL_ONE_ARG(j1)
NEGATIVE_KERNELS_SHELL_TWO_ARGS(jn)
NEGATIVE_KERNELS_SHELL_ONE_ARG(y0)
NEGATIVE_KERNELS_SHELL_ONE_ARG(y1)
NEGATIVE_KERNELS_SHELL_TWO_ARGS(yn)
NEGATIVE_KERNELS_SHELL_ONE_ARG(cyl_bessel_i0)
NEGATIVE_KERNELS_SHELL_ONE_ARG(cyl_bessel_i1)
@@ -0,0 +1,236 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
/*
Negative kernels used for the math special function negative Test Cases that are using RTC.
*/
static constexpr auto kErf{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void erf_kernel_v1(double* x) { double result = erf(x); }
__global__ void erf_kernel_v2(Dummy x) { double result = erf(x); }
__global__ void erff_kernel_v1(float* x) { float result = erff(x); }
__global__ void erff_kernel_v2(Dummy x) { float result = erff(x); }
)"};
static constexpr auto kErfc{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void erfc_kernel_v1(double* x) { double result = erfc(x); }
__global__ void erfc_kernel_v2(Dummy x) { double result = erfc(x); }
__global__ void erfcf_kernel_v1(float* x) { float result = erfcf(x); }
__global__ void erfcf_kernel_v2(Dummy x) { float result = erfcf(x); }
)"};
static constexpr auto kErfinv{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void erfinv_kernel_v1(double* x) { double result = erfinv(x); }
__global__ void erfinv_kernel_v2(Dummy x) { double result = erfinv(x); }
__global__ void erfinvf_kernel_v1(float* x) { float result = erfinvf(x); }
__global__ void erfinvf_kernel_v2(Dummy x) { float result = erfinvf(x); }
)"};
static constexpr auto kErfcinv{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void erfcinv_kernel_v1(double* x) { double result = erfcinv(x); }
__global__ void erfcinv_kernel_v2(Dummy x) { double result = erfcinv(x); }
__global__ void erfcinvf_kernel_v1(float* x) { float result = erfcinvf(x); }
__global__ void erfcinvf_kernel_v2(Dummy x) { float result = erfcinvf(x); }
)"};
static constexpr auto kErfcx{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void erfcx_kernel_v1(double* x) { double result = erfcx(x); }
__global__ void erfcx_kernel_v2(Dummy x) { double result = erfcx(x); }
__global__ void erfcxf_kernel_v1(float* x) { float result = erfcxf(x); }
__global__ void erfcxf_kernel_v2(Dummy x) { float result = erfcxf(x); }
)"};
static constexpr auto kNormcdf{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void normcdf_kernel_v1(double* x) { double result = normcdf(x); }
__global__ void normcdf_kernel_v2(Dummy x) { double result = normcdf(x); }
__global__ void normcdff_kernel_v1(float* x) { float result = normcdff(x); }
__global__ void normcdff_kernel_v2(Dummy x) { float result = normcdff(x); }
)"};
static constexpr auto kNormcdfinv{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void normcdfinv_kernel_v1(double* x) { double result = normcdfinv(x); }
__global__ void normcdfinv_kernel_v2(Dummy x) { double result = normcdfinv(x); }
__global__ void normcdfinvf_kernel_v1(float* x) { float result = normcdfinvf(x); }
__global__ void normcdfinvf_kernel_v2(Dummy x) { float result = normcdfinvf(x); }
)"};
static constexpr auto kLgamma{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void lgamma_kernel_v1(double* x) { double result = lgamma(x); }
__global__ void lgamma_kernel_v2(Dummy x) { double result = lgamma(x); }
__global__ void lgammaf_kernel_v1(float* x) { float result = lgammaf(x); }
__global__ void lgammaf_kernel_v2(Dummy x) { float result = lgammaf(x); }
)"};
static constexpr auto kTgamma{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void tgamma_kernel_v1(double* x) { double result = tgamma(x); }
__global__ void tgamma_kernel_v2(Dummy x) { double result = tgamma(x); }
__global__ void tgammaf_kernel_v1(float* x) { float result = tgammaf(x); }
__global__ void tgammaf_kernel_v2(Dummy x) { float result = tgammaf(x); }
)"};
static constexpr auto kJ0{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void j0_kernel_v1(double* x) { double result = j0(x); }
__global__ void j0_kernel_v2(Dummy x) { double result = j0(x); }
__global__ void j0f_kernel_v1(float* x) { float result = j0f(x); }
__global__ void j0f_kernel_v2(Dummy x) { float result = j0f(x); }
)"};
static constexpr auto kJ1{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void j1_kernel_v1(double* x) { double result = j1(x); }
__global__ void j1_kernel_v2(Dummy x) { double result = j1(x); }
__global__ void j1f_kernel_v1(float* x) { float result = j1f(x); }
__global__ void j1f_kernel_v2(Dummy x) { float result = j1f(x); }
)"};
static constexpr auto kJn{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void jn_kernel_v1(int* x, double y) { double result = jn(x, y); }
__global__ void jn_kernel_v2(int x, double* y) { double result = jn(x, y); }
__global__ void jn_kernel_v3(Dummy x, double y) { double result = jn(x, y); }
__global__ void jn_kernel_v4(int x, Dummy y) { double result = jn(x, y); }
__global__ void jnf_kernel_v1(int* x, float y) { float result = jnf(x, y); }
__global__ void jnf_kernel_v2(int x, float* y) { float result = jnf(x, y); }
__global__ void jnf_kernel_v3(Dummy x, float y) { float result = jnf(x, y); }
__global__ void jnf_kernel_v4(int x, Dummy y) { float result = jnf(x, y); }
)"};
static constexpr auto kY0{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void y0_kernel_v1(double* x) { double result = y0(x); }
__global__ void y0_kernel_v2(Dummy x) { double result = y0(x); }
__global__ void y0f_kernel_v1(float* x) { float result = y0f(x); }
__global__ void y0f_kernel_v2(Dummy x) { float result = y0f(x); }
)"};
static constexpr auto kY1{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void y1_kernel_v1(double* x) { double result = y1(x); }
__global__ void y1_kernel_v2(Dummy x) { double result = y1(x); }
__global__ void y1f_kernel_v1(float* x) { float result = y1f(x); }
__global__ void y1f_kernel_v2(Dummy x) { float result = y1f(x); }
)"};
static constexpr auto kYn{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void yn_kernel_v1(int* x, double y) { double result = yn(x, y); }
__global__ void yn_kernel_v2(int x, double* y) { double result = yn(x, y); }
__global__ void yn_kernel_v3(Dummy x, double y) { double result = yn(x, y); }
__global__ void yn_kernel_v4(int x, Dummy y) { double result = yn(x, y); }
__global__ void ynf_kernel_v1(int* x, float y) { float result = ynf(x, y); }
__global__ void ynf_kernel_v2(int x, float* y) { float result = ynf(x, y); }
__global__ void ynf_kernel_v3(Dummy x, float y) { float result = ynf(x, y); }
__global__ void ynf_kernel_v4(int x, Dummy y) { float result = ynf(x, y); }
)"};
static constexpr auto kCylBesselI0{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void cyl_bessel_i0_kernel_v1(double* x) { double result = cyl_bessel_i0(x); }
__global__ void cyl_bessel_i0_kernel_v2(Dummy x) { double result = cyl_bessel_i0(x); }
__global__ void cyl_bessel_i0f_kernel_v1(float* x) { float result = cyl_bessel_i0f(x); }
__global__ void cyl_bessel_i0f_kernel_v2(Dummy x) { float result = cyl_bessel_i0f(x); }
)"};
static constexpr auto kCylBesselI1{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void cyl_bessel_i1_kernel_v1(double* x) { double result = cyl_bessel_i1(x); }
__global__ void cyl_bessel_i1_kernel_v2(Dummy x) { double result = cyl_bessel_i1(x); }
__global__ void cyl_bessel_i1f_kernel_v1(float* x) { float result = cyl_bessel_i1f(x); }
__global__ void cyl_bessel_i1f_kernel_v2(Dummy x) { float result = cyl_bessel_i1f(x); }
)"};
+294
Wyświetl plik
@@ -0,0 +1,294 @@
//
// Copyright (c) 2017 The Khronos Group Inc.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//
// Disclaimer:
// This code is based on the work found in OpenCL-CTS authored by The Khronos Group.
// The original code can be found at https://github.com/KhronosGroup/OpenCL-CTS.
// We acknowledge the contributions of The Khronos Group to the development of this code.
#pragma once
#include <array>
#include <limits>
/*-----------------------------------------------------------------------------
HEX_FLT, HEXT_DBL, HEX_LDBL -- Create hex floating point literal of type
float, double, long double respectively. Arguments:
sm -- sign of number,
int -- integer part of mantissa (without `0x' prefix),
fract -- fractional part of mantissa (without decimal point and `L' or
`LL' suffixes),
se -- sign of exponent,
exp -- absolute value of (binary) exponent.
Example:
double yhi = HEX_DBL(+, 1, 5555555555555, -, 2); // 0x1.5555555555555p-2
Note:
We have to pass signs as separate arguments because gcc pass negative
integer values (e. g. `-2') into a macro as two separate tokens, so
`HEX_FLT(1, 0, -2)' produces result `0x1.0p- 2' (note a space between minus
and two) which is not a correct floating point literal.
-----------------------------------------------------------------------------*/
#if defined(_MSC_VER) && !defined(__INTEL_COMPILER)
// If compiler does not support hex floating point literals:
#define HEX_FLT(sm, int, fract, se, exp) \
sm ldexpf((float)(0x##int##fract##UL), \
se exp + ilogbf((float)0x##int) - ilogbf((float)(0x##int##fract##UL)))
#define HEX_DBL(sm, int, fract, se, exp) \
sm ldexp((double)(0x##int##fract##ULL), \
se exp + ilogb((double)0x##int) - ilogb((double)(0x##int##fract##ULL)))
#define HEX_LDBL(sm, int, fract, se, exp) \
sm ldexpl((long double)(0x##int##fract##ULL), \
se exp + ilogbl((long double)0x##int) - ilogbl((long double)(0x##int##fract##ULL)))
#else
// If compiler supports hex floating point literals: just concatenate all the
// parts into a literal.
#define HEX_FLT(sm, int, fract, se, exp) sm 0x##int##.##fract##p##se##exp##F
#define HEX_DBL(sm, int, fract, se, exp) sm 0x##int##.##fract##p##se##exp
#define HEX_LDBL(sm, int, fract, se, exp) sm 0x##int##.##fract##p##se##exp##L
#endif
inline constexpr std::array kSpecialValuesDouble{
-std::numeric_limits<double>::quiet_NaN(),
-std::numeric_limits<double>::infinity(),
-std::numeric_limits<double>::max(),
HEX_DBL(-, 1, 0000000000001, +, 64),
HEX_DBL(-, 1, 0, +, 64),
HEX_DBL(-, 1, fffffffffffff, +, 63),
HEX_DBL(-, 1, 0000000000001, +, 63),
HEX_DBL(-, 1, 0, +, 63),
HEX_DBL(-, 1, fffffffffffff, +, 62),
HEX_DBL(-, 1, 000002, +, 32),
HEX_DBL(-, 1, 0, +, 32),
HEX_DBL(-, 1, fffffffffffff, +, 31),
HEX_DBL(-, 1, 0000000000001, +, 31),
HEX_DBL(-, 1, 0, +, 31),
HEX_DBL(-, 1, fffffffffffff, +, 30),
-1000.0,
-100.0,
-4.0,
-3.5,
-3.0,
HEX_DBL(-, 1, 8000000000001, +, 1),
-2.5,
HEX_DBL(-, 1, 7ffffffffffff, +, 1),
-2.0,
HEX_DBL(-, 1, 8000000000001, +, 0),
-1.5,
HEX_DBL(-, 1, 7ffffffffffff, +, 0),
HEX_DBL(-, 1, 0000000000001, +, 0),
-1.0,
HEX_DBL(-, 1, fffffffffffff, -, 1),
HEX_DBL(-, 1, 0000000000001, -, 1),
-0.5,
HEX_DBL(-, 1, fffffffffffff, -, 2),
HEX_DBL(-, 1, 0000000000001, -, 2),
-0.25,
HEX_DBL(-, 1, fffffffffffff, -, 3),
HEX_DBL(-, 1, 0000000000001, -, 1022),
-std::numeric_limits<double>::min(),
HEX_DBL(-, 0, fffffffffffff, -, 1022),
HEX_DBL(-, 0, 0000000000fff, -, 1022),
HEX_DBL(-, 0, 00000000000fe, -, 1022),
HEX_DBL(-, 0, 000000000000e, -, 1022),
HEX_DBL(-, 0, 000000000000c, -, 1022),
HEX_DBL(-, 0, 000000000000a, -, 1022),
HEX_DBL(-, 0, 0000000000008, -, 1022),
HEX_DBL(-, 0, 0000000000007, -, 1022),
HEX_DBL(-, 0, 0000000000006, -, 1022),
HEX_DBL(-, 0, 0000000000005, -, 1022),
HEX_DBL(-, 0, 0000000000004, -, 1022),
HEX_DBL(-, 0, 0000000000003, -, 1022),
HEX_DBL(-, 0, 0000000000002, -, 1022),
HEX_DBL(-, 0, 0000000000001, -, 1022),
-0.0,
std::numeric_limits<double>::quiet_NaN(),
std::numeric_limits<double>::infinity(),
std::numeric_limits<double>::max(),
HEX_DBL(+, 1, 0000000000001, +, 64),
HEX_DBL(+, 1, 0, +, 64),
HEX_DBL(+, 1, fffffffffffff, +, 63),
HEX_DBL(+, 1, 0000000000001, +, 63),
HEX_DBL(+, 1, 0, +, 63),
HEX_DBL(+, 1, fffffffffffff, +, 62),
HEX_DBL(+, 1, 000002, +, 32),
HEX_DBL(+, 1, 0, +, 32),
HEX_DBL(+, 1, fffffffffffff, +, 31),
HEX_DBL(+, 1, 0000000000001, +, 31),
HEX_DBL(+, 1, 0, +, 31),
HEX_DBL(+, 1, fffffffffffff, +, 30),
+1000.0,
+100.0,
+4.0,
+3.5,
+3.0,
HEX_DBL(+, 1, 8000000000001, +, 1),
+2.5,
HEX_DBL(+, 1, 7ffffffffffff, +, 1),
+2.0,
HEX_DBL(+, 1, 8000000000001, +, 0),
+1.5,
HEX_DBL(+, 1, 7ffffffffffff, +, 0),
HEX_DBL(+, 1, 0000000000001, +, 0),
+1.0,
HEX_DBL(+, 1, fffffffffffff, -, 1),
HEX_DBL(+, 1, 0000000000001, -, 1),
+0.5,
HEX_DBL(+, 1, fffffffffffff, -, 2),
HEX_DBL(+, 1, 0000000000001, -, 2),
+0.25,
HEX_DBL(+, 1, fffffffffffff, -, 3),
HEX_DBL(+, 1, 0000000000001, -, 1022),
+std::numeric_limits<double>::min(),
HEX_DBL(+, 0, fffffffffffff, -, 1022),
HEX_DBL(+, 0, 0000000000fff, -, 1022),
HEX_DBL(+, 0, 00000000000fe, -, 1022),
HEX_DBL(+, 0, 000000000000e, -, 1022),
HEX_DBL(+, 0, 000000000000c, -, 1022),
HEX_DBL(+, 0, 000000000000a, -, 1022),
HEX_DBL(+, 0, 0000000000008, -, 1022),
HEX_DBL(+, 0, 0000000000007, -, 1022),
HEX_DBL(+, 0, 0000000000006, -, 1022),
HEX_DBL(+, 0, 0000000000005, -, 1022),
HEX_DBL(+, 0, 0000000000004, -, 1022),
HEX_DBL(+, 0, 0000000000003, -, 1022),
HEX_DBL(+, 0, 0000000000002, -, 1022),
HEX_DBL(+, 0, 0000000000001, -, 1022),
+0.0,
};
inline constexpr std::array kSpecialValuesFloat{
-std::numeric_limits<float>::quiet_NaN(),
-std::numeric_limits<float>::infinity(),
-std::numeric_limits<float>::max(),
HEX_FLT(-, 1, 000002, +, 64),
HEX_FLT(-, 1, 0, +, 64),
HEX_FLT(-, 1, fffffe, +, 63),
HEX_FLT(-, 1, 000002, +, 63),
HEX_FLT(-, 1, 0, +, 63),
HEX_FLT(-, 1, fffffe, +, 62),
HEX_FLT(-, 1, 000002, +, 32),
HEX_FLT(-, 1, 0, +, 32),
HEX_FLT(-, 1, fffffe, +, 31),
HEX_FLT(-, 1, 000002, +, 31),
HEX_FLT(-, 1, 0, +, 31),
HEX_FLT(-, 1, fffffe, +, 30),
-1000.f,
-100.f,
-4.0f,
-3.5f,
-3.0f,
HEX_FLT(-, 1, 800002, +, 1),
-2.5f,
HEX_FLT(-, 1, 7ffffe, +, 1),
-2.0f,
HEX_FLT(-, 1, 800002, +, 0),
-1.5f,
HEX_FLT(-, 1, 7ffffe, +, 0),
HEX_FLT(-, 1, 000002, +, 0),
-1.0f,
HEX_FLT(-, 1, fffffe, -, 1),
HEX_FLT(-, 1, 000002, -, 1),
-0.5f,
HEX_FLT(-, 1, fffffe, -, 2),
HEX_FLT(-, 1, 000002, -, 2),
-0.25f,
HEX_FLT(-, 1, fffffe, -, 3),
HEX_FLT(-, 1, 000002, -, 126),
-std::numeric_limits<float>::min(),
HEX_FLT(-, 0, fffffe, -, 126),
HEX_FLT(-, 0, 000ffe, -, 126),
HEX_FLT(-, 0, 0000fe, -, 126),
HEX_FLT(-, 0, 00000e, -, 126),
HEX_FLT(-, 0, 00000c, -, 126),
HEX_FLT(-, 0, 00000a, -, 126),
HEX_FLT(-, 0, 000008, -, 126),
HEX_FLT(-, 0, 000006, -, 126),
HEX_FLT(-, 0, 000004, -, 126),
HEX_FLT(-, 0, 000002, -, 126),
-0.0f,
std::numeric_limits<float>::quiet_NaN(),
std::numeric_limits<float>::infinity(),
std::numeric_limits<float>::max(),
HEX_FLT(+, 1, 000002, +, 64),
HEX_FLT(+, 1, 0, +, 64),
HEX_FLT(+, 1, fffffe, +, 63),
HEX_FLT(+, 1, 000002, +, 63),
HEX_FLT(+, 1, 0, +, 63),
HEX_FLT(+, 1, fffffe, +, 62),
HEX_FLT(+, 1, 000002, +, 32),
HEX_FLT(+, 1, 0, +, 32),
HEX_FLT(+, 1, fffffe, +, 31),
HEX_FLT(+, 1, 000002, +, 31),
HEX_FLT(+, 1, 0, +, 31),
HEX_FLT(+, 1, fffffe, +, 30),
+1000.f,
+100.f,
+4.0f,
+3.5f,
+3.0f,
HEX_FLT(+, 1, 800002, +, 1),
2.5f,
HEX_FLT(+, 1, 7ffffe, +, 1),
+2.0f,
HEX_FLT(+, 1, 800002, +, 0),
1.5f,
HEX_FLT(+, 1, 7ffffe, +, 0),
HEX_FLT(+, 1, 000002, +, 0),
+1.0f,
HEX_FLT(+, 1, fffffe, -, 1),
HEX_FLT(+, 1, 000002, -, 1),
+0.5f,
HEX_FLT(+, 1, fffffe, -, 2),
HEX_FLT(+, 1, 000002, -, 2),
+0.25f,
HEX_FLT(+, 1, fffffe, -, 3),
HEX_FLT(+, 1, 000002, -, 126),
+std::numeric_limits<float>::min(),
HEX_FLT(+, 0, fffffe, -, 126),
HEX_FLT(+, 0, 000ffe, -, 126),
HEX_FLT(+, 0, 0000fe, -, 126),
HEX_FLT(+, 0, 00000e, -, 126),
HEX_FLT(+, 0, 00000c, -, 126),
HEX_FLT(+, 0, 00000a, -, 126),
HEX_FLT(+, 0, 000008, -, 126),
HEX_FLT(+, 0, 000006, -, 126),
HEX_FLT(+, 0, 000004, -, 126),
HEX_FLT(+, 0, 000002, -, 126),
+0.0f,
};
inline constexpr std::array kSpecialValuesInt{
0, 1, 2, 3, 126, 127, 128, 1022, 1023, 1024, 0x02000001, 0x04000001, 1465264071, 1488522147,
std::numeric_limits<int>::max(), -1, -2, -3, -126, -127, -128, -1022, -1023, -11024, -0x02000001,
-0x04000001, -1465264071, -1488522147, std::numeric_limits<int>::min(), -std::numeric_limits<int>::max()
};
template <typename T> struct SpecialVals {
const T* const data;
const size_t size;
};
inline constexpr auto kSpecialValRegistry =
std::make_tuple(SpecialVals<float>{kSpecialValuesFloat.data(), kSpecialValuesFloat.size()},
SpecialVals<double>{kSpecialValuesDouble.data(), kSpecialValuesDouble.size()},
SpecialVals<int>{kSpecialValuesInt.data(), kSpecialValuesInt.size()});
+96
Wyświetl plik
@@ -0,0 +1,96 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "misc_negative_kernels_rtc.hh"
#include "unary_common.hh"
#include "binary_common.hh"
#include "ternary_common.hh"
MATH_UNARY_WITHIN_ULP_TEST_DEF(fabs, std::fabs, 0, 0)
TEST_CASE("Unit_Device_fabs_fabsf_Negative_RTC") { NegativeTestRTCWrapper<4>(kFabs); }
MATH_BINARY_WITHIN_ULP_TEST_DEF(copysign, std::copysign, 0, 0)
TEST_CASE("Unit_Device_copysign_copysignf_Negative_RTC") { NegativeTestRTCWrapper<8>(kCopySign); }
MATH_BINARY_WITHIN_ULP_TEST_DEF(fmax, std::fmax, 0, 0)
TEST_CASE("Unit_Device_fmax_fmaxf_Negative_RTC") { NegativeTestRTCWrapper<8>(kFmax); }
MATH_BINARY_WITHIN_ULP_TEST_DEF(fmin, std::fmin, 0, 0)
TEST_CASE("Unit_Device_fmin_fminf_Negative_RTC") { NegativeTestRTCWrapper<8>(kFmin); }
MATH_BINARY_WITHIN_ULP_TEST_DEF(nextafter, std::nextafter, 0, 0)
TEST_CASE("Unit_Device_nextafter_nextafterf_Negative_RTC") {
NegativeTestRTCWrapper<8>(kNextAfter);
}
MATH_TERNARY_WITHIN_ULP_TEST_DEF(fma, std::fma, 0, 0)
TEST_CASE("Unit_Device_fma_fmaf_Negative_RTC") { NegativeTestRTCWrapper<12>(kFma); }
__global__ void fdividef_kernel(float* const ys, const size_t num_xs, float* const x1s,
float* const x2s) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
ys[i] = fdividef(x1s[i], x2s[i]);
}
}
TEST_CASE("Unit_Device_fdividef_Accuracy_Positive") {
double (*ref)(double, double) = [](double x1, double x2) { return x1 / x2; };
BinaryFloatingPointTest(fdividef_kernel, ref, ULPValidatorBuilderFactory<float>(0));
}
TEST_CASE("Unit_Device_fdividef_Negative_RTC") { NegativeTestRTCWrapper<4>(kFdividef); }
#define MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(kern_name, ref_func) \
template <typename T> \
__global__ void kern_name##_kernel(bool* const ys, const size_t num_xs, T* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = kern_name(xs[i]); \
} \
} \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - float") { \
bool (*ref)(double) = ref_func; \
UnarySinglePrecisionTest(kern_name##_kernel<float>, ref, EqValidatorBuilderFactory<bool>()); \
} \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - double") { \
bool (*ref)(long double) = ref_func; \
UnaryDoublePrecisionTest(kern_name##_kernel<double>, ref, EqValidatorBuilderFactory<bool>()); \
}
MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(isfinite, std::isfinite)
TEST_CASE("Unit_Device_isfinite_Negative_RTC") { NegativeTestRTCWrapper<4>(kIsFinite); }
MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(isinf, std::isinf)
TEST_CASE("Unit_Device_isinf_Negative_RTC") { NegativeTestRTCWrapper<4>(kIsInf); }
MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(isnan, std::isnan)
TEST_CASE("Unit_Device_isnan_Negative_RTC") { NegativeTestRTCWrapper<4>(kIsNan); }
MATH_BOOL_RETURNING_FUNCTION_TEST_DEF(signbit, std::signbit)
TEST_CASE("Unit_Device_signbit_Negative_RTC") { NegativeTestRTCWrapper<4>(kSignBit); }
@@ -0,0 +1,87 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define MISC_UNARY_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); } \
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); }
#define MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(float* x) { bool result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { bool result = func_name(x); } \
__global__ void func_name##_kernel_v3(double* x) { bool result = func_name(x); } \
__global__ void func_name##_kernel_v4(Dummy x) { bool result = func_name(x); }
#define MISC_BINARY_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##f_kernel_v1(float* x, float y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v2(Dummy x, float y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v3(float x, float* y) { float result = func_name##f(x, y); } \
__global__ void func_name##f_kernel_v4(float x, Dummy y) { float result = func_name##f(x, y); } \
__global__ void func_name##_kernel_v1(double* x, double y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(Dummy x, double y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(double x, double* y) { double result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(double x, Dummy y) { double result = func_name(x, y); }
/*Expecting 4 errors*/
MISC_UNARY_NEGATIVE_KERNELS(fabs)
/*Expecting 8 errors per macro invocation - 40 total*/
MISC_BINARY_NEGATIVE_KERNELS(copysign)
MISC_BINARY_NEGATIVE_KERNELS(fmax)
MISC_BINARY_NEGATIVE_KERNELS(fmin)
MISC_BINARY_NEGATIVE_KERNELS(nextafter)
MISC_BINARY_NEGATIVE_KERNELS(fma)
/*Expecting 4 errors*/
__global__ void fdividef_kernel_v1(float* x, float y) { float result = fdividef(x, y); }
__global__ void fdividef_kernel_v2(Dummy x, float y) { float result = fdivide(x); }
__global__ void fdividef_kernel_v3(float x, float* y) { float result = fdivide(x); }
__global__ void fdividef_kernel_v4(float x, Dummy y) { float result = fdivide(x); }
/*Expecting 4 errors per macro invocation - 16 total*/
MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(isfinite)
MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(isinf)
MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(isnan)
MISC_UNARY_BOOL_RET_NEGATIVE_KERNELS(signbit)
/*Expecting 12 errors*/
__global__ void fmaf_kernel_v1(float* x, float y, float z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v2(Dummy x, float y, float z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v3(float x, float* y, float z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v4(float x, Dummy y, float z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v5(float x, float y, float* z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v6(float x, float y, Dummy z) { float result = fmaf(x, y, z); }
__global__ void fma_kernel_v1(double* x, double y, double z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v2(Dummy x, double y, double z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v3(double x, double* y, double z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v4(double x, Dummy y, double z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v5(double x, double y, double* z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v6(double x, double y, Dummy z) { double result = fmaf(x, y, z); }
@@ -0,0 +1,177 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
static constexpr auto kFabs{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void fabsf_kernel_v1(float* x) { float result = fabsf(x); }
__global__ void fabsf_kernel_v2(Dummy x) { float result = fabsf(x); }
__global__ void fabs_kernel_v1(double* x) { double result = fabs(x); }
__global__ void fabs_kernel_v2(Dummy x) { double result = fabs(x); }
)"};
static constexpr auto kCopySign{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void copysignf_kernel_v1(float* x, float y) { float result = copysignf(x, y); }
__global__ void copysignf_kernel_v2(Dummy x, float y) { float result = copysignf(x, y); }
__global__ void copysignf_kernel_v3(float x, float* y) { float result = copysignf(x, y); }
__global__ void copysignf_kernel_v4(float x, Dummy y) { float result = copysignf(x, y); }
__global__ void copysign_kernel_v1(double* x, double y) { double result = copysign(x, y); }
__global__ void copysign_kernel_v2(Dummy x, double y) { double result = copysign(x, y); }
__global__ void copysign_kernel_v3(double x, double* y) { double result = copysign(x, y); }
__global__ void copysign_kernel_v4(double x, Dummy y) { double result = copysign(x, y); }
)"};
static constexpr auto kFmax{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void fmaxf_kernel_v1(float* x, float y) { float result = fmaxf(x, y); }
__global__ void fmaxf_kernel_v2(Dummy x, float y) { float result = fmaxf(x, y); }
__global__ void fmaxf_kernel_v3(float x, float* y) { float result = fmaxf(x, y); }
__global__ void fmaxf_kernel_v4(float x, Dummy y) { float result = fmaxf(x, y); }
__global__ void fmax_kernel_v1(double* x, double y) { double result = fmax(x, y); }
__global__ void fmax_kernel_v2(Dummy x, double y) { double result = fmax(x, y); }
__global__ void fmax_kernel_v3(double x, double* y) { double result = fmax(x, y); }
__global__ void fmax_kernel_v4(double x, Dummy y) { double result = fmax(x, y); }
)"};
static constexpr auto kFmin{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void fminf_kernel_v1(float* x, float y) { float result = fminf(x, y); }
__global__ void fminf_kernel_v2(Dummy x, float y) { float result = fminf(x, y); }
__global__ void fminf_kernel_v3(float x, float* y) { float result = fminf(x, y); }
__global__ void fminf_kernel_v4(float x, Dummy y) { float result = fminf(x, y); }
__global__ void fmin_kernel_v1(double* x, double y) { double result = fmin(x, y); }
__global__ void fmin_kernel_v2(Dummy x, double y) { double result = fmin(x, y); }
__global__ void fmin_kernel_v3(double x, double* y) { double result = fmin(x, y); }
__global__ void fmin_kernel_v4(double x, Dummy y) { double result = fmin(x, y); }
)"};
static constexpr auto kNextAfter{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void nextafterf_kernel_v1(float* x, float y) { float result = nextafterf(x, y); }
__global__ void nextafterf_kernel_v2(Dummy x, float y) { float result = nextafterf(x, y); }
__global__ void nextafterf_kernel_v3(float x, float* y) { float result = nextafterf(x, y); }
__global__ void nextafterf_kernel_v4(float x, Dummy y) { float result = nextafterf(x, y); }
__global__ void nextafter_kernel_v1(double* x, double y) { double result = nextafter(x, y); }
__global__ void nextafter_kernel_v2(Dummy x, double y) { double result = nextafter(x, y); }
__global__ void nextafter_kernel_v3(double x, double* y) { double result = nextafter(x, y); }
__global__ void nextafter_kernel_v4(double x, Dummy y) { double result = nextafter(x, y); }
)"};
static constexpr auto kFma{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void fmaf_kernel_v1(float* x, float y, float z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v2(Dummy x, float y, float z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v3(float x, float* y, float z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v4(float x, Dummy y, float z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v5(float x, float y, float* z) { float result = fmaf(x, y, z); }
__global__ void fmaf_kernel_v6(float x, float y, Dummy z) { float result = fmaf(x, y, z); }
__global__ void fma_kernel_v1(double* x, double y, double z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v2(Dummy x, double y, double z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v3(double x, double* y, double z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v4(double x, Dummy y, double z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v5(double x, double y, double* z) { double result = fmaf(x, y, z); }
__global__ void fma_kernel_v6(double x, double y, Dummy z) { double result = fmaf(x, y, z); }
)"};
static constexpr auto kFdividef{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void fdividef_kernel_v1(float* x, float y) { float result = fdividef(x, y); }
__global__ void fdividef_kernel_v2(Dummy x, float y) { float result = fdivide(x); }
__global__ void fdividef_kernel_v3(float x, float* y) { float result = fdivide(x); }
__global__ void fdividef_kernel_v4(float x, Dummy y) { float result = fdivide(x); }
)"};
static constexpr auto kIsFinite{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void isfinite_kernel_v1(float* x) { bool result = isfinite(x); }
__global__ void isfinite_kernel_v2(Dummy x) { bool result = isfinite(x); }
__global__ void isfinite_kernel_v3(double* x) { bool result = isfinite(x); }
__global__ void isfinite_kernel_v4(Dummy x) { bool result = isfinite(x); }
)"};
static constexpr auto kIsInf{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void isinf_kernel_v1(float* x) { bool result = isinf(x); }
__global__ void isinf_kernel_v2(Dummy x) { bool result = isinf(x); }
__global__ void isinf_kernel_v3(double* x) { bool result = isinf(x); }
__global__ void isinf_kernel_v4(Dummy x) { bool result = isinf(x); }
)"};
static constexpr auto kIsNan{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void isnan_kernel_v1(float* x) { bool result = isnan(x); }
__global__ void isnan_kernel_v2(Dummy x) { bool result = isnan(x); }
__global__ void isnan_kernel_v3(double* x) { bool result = isnan(x); }
__global__ void isnan_kernel_v4(Dummy x) { bool result = isnan(x); }
)"};
static constexpr auto kSignBit{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void signbit_kernel_v1(float* x) { bool result = signbit(x); }
__global__ void signbit_kernel_v2(Dummy x) { bool result = signbit(x); }
__global__ void signbit_kernel_v3(double* x) { bool result = signbit(x); }
__global__ void signbit_kernel_v4(Dummy x) { bool result = signbit(x); }
)"};
+134
Wyświetl plik
@@ -0,0 +1,134 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "math_common.hh"
#include "math_special_values.hh"
#include <hip/hip_cooperative_groups.h>
namespace cg = cooperative_groups;
#define MATH_POW_INT_KERNEL_DEF(func_name) \
template <typename T1, typename T2> \
__global__ void func_name##_kernel(T1* const ys, const size_t num_xs, T1* const x1s, \
T2* const x2s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
if constexpr (std::is_same_v<float, T1>) { \
ys[i] = func_name##f(x1s[i], x2s[i]); \
} else if constexpr (std::is_same_v<double, T1>) { \
ys[i] = func_name(x1s[i], x2s[i]); \
} \
} \
}
template <typename T1, typename T2>
using kernel_pow_int_sig = void (*)(T1*, const size_t, T1*, T2*);
template <typename T1, typename T2> using ref_pow_int_sig = T1 (*)(T1, T2);
template <typename T1, typename T2, typename RT1, typename RT2, typename ValidatorBuilder>
void PowIntFloatingPointBruteForceTest(kernel_pow_int_sig<T1, T2> kernel,
ref_pow_int_sig<RT1, RT2> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const uint64_t num_iterations = GetTestIterationCount();
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(T1) * 2 + sizeof(T2)), num_iterations);
LinearAllocGuard<T1> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(T1)};
LinearAllocGuard<T2> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(T2)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &x1s, &x2s] {
const auto generator1 = [=] {
static thread_local std::mt19937 rng(std::random_device{}());
std::uniform_real_distribution<RefType_t<T1>> unif_dist(std::numeric_limits<T1>::lowest(),
std::numeric_limits<T1>::max());
return static_cast<T1>(unif_dist(rng));
};
const auto generator2 = [] {
static thread_local std::mt19937 rng(std::random_device{}());
std::uniform_int_distribution<T2> unif_dist(std::numeric_limits<T2>::lowest(),
std::numeric_limits<T2>::max());
return unif_dist(rng);
};
std::generate(x1s.ptr() + base_idx, x1s.ptr() + base_idx + sub_batch_size, generator1);
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator2);
});
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
x2s.ptr());
}
}
template <typename T1, typename T2, typename RT1, typename RT2, typename ValidatorBuilder>
void PowIntFloatingPointSpecialValuesTest(kernel_pow_int_sig<T1, T2> kernel,
ref_pow_int_sig<RT1, RT2> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const auto values1 = std::get<SpecialVals<T1>>(kSpecialValRegistry);
const auto values2 = std::get<SpecialVals<int>>(kSpecialValRegistry);
const auto size = values1.size * values2.size;
LinearAllocGuard<T1> x1s{LinearAllocs::hipHostMalloc, size * sizeof(T1)};
LinearAllocGuard<T2> x2s{LinearAllocs::hipHostMalloc, size * sizeof(T2)};
for (auto i = 0u; i < values1.size; ++i) {
for (auto j = 0u; j < values2.size; ++j) {
x1s.ptr()[i * values2.size + j] = values1.data[i];
x2s.ptr()[i * values2.size + j] = static_cast<T2>(values2.data[j]);
}
}
MathTest math_test(kernel, size);
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, size, x1s.ptr(),
x2s.ptr());
}
template <typename T1, typename T2, typename RT1, typename RT2, typename ValidatorBuilder>
void PowIntFloatingPointTest(kernel_pow_int_sig<T1, T2> kernel, ref_pow_int_sig<RT1, RT2> ref_func,
const ValidatorBuilder& validator_builder) {
SECTION("Special values") {
PowIntFloatingPointSpecialValuesTest(kernel, ref_func, validator_builder);
}
SECTION("Brute force") { PowIntFloatingPointBruteForceTest(kernel, ref_func, validator_builder); }
}
+455
Wyświetl plik
@@ -0,0 +1,455 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "unary_common.hh"
#include "binary_common.hh"
#include "pow_common.hh"
#include "math_pow_negative_kernels_rtc.hh"
/**
* @addtogroup PowMathFuncs PowMathFuncs
* @{
* @ingroup MathTest
*/
/********** Unary Functions **********/
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `expf(x)` for all possible inputs and `exp(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::exp(T)`. The maximum ulp error for single
* precision is 2 and for double precision is 1.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(exp, 2, 1)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for expf and exp.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_exp_expf_Negative_RTC") { NegativeTestRTCWrapper<4>(kExp); }
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `exp2f(x)` for all possible inputs and `exp2(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::exp2(T)`. The maximum ulp error for single
* precision is 2 and for double precision is 1.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(exp2, 2, 1)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for exp2f and exp2.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_exp2_exp2f_Negative_RTC") { NegativeTestRTCWrapper<4>(kExp2); }
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `expm1f(x)` for all possible inputs and `expm1(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::exp(T)`. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(expm1, 1, 1)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for expm1f and expm1.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_expm1_expm1f_Negative_RTC") { NegativeTestRTCWrapper<4>(kExpm1); }
MATH_UNARY_KERNEL_DEF(exp10)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `exp10f(x)` for all possible inputs. The maximum ulp error
* is 2.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_exp10f_Accuracy_Positive") {
auto exp10_ref = [](double arg) -> double { return std::pow(10, arg); };
double (*ref)(double) = exp10_ref;
UnarySinglePrecisionTest(exp10_kernel<float>, ref, ULPValidatorBuilderFactory<float>(2));
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `exp10(x)` against a table of difficult values,
* followed by a large number of randomly generated values. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_exp10_Accuracy_Positive") {
auto exp10_ref = [](long double arg) -> long double { return std::pow(10, arg); };
long double (*ref)(long double) = exp10_ref;
UnaryDoublePrecisionTest(exp10_kernel<double>, ref, ULPValidatorBuilderFactory<double>(1));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for exp10f and exp10.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_exp10_exp10f_Negative_RTC") { NegativeTestRTCWrapper<4>(kExp10); }
template <typename T>
__global__ void frexp_kernel(std::pair<T, int>* const ys, const size_t num_xs, T* const xs) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
if constexpr (std::is_same_v<float, T>) {
ys[i].first = frexpf(xs[i], &ys[i].second);
} else if constexpr (std::is_same_v<double, T>) {
ys[i].first = frexp(xs[i], &ys[i].second);
}
}
}
template <typename T> std::pair<T, int> frexp_ref(T arg) {
int exp_v;
T res = std::frexp(arg, &exp_v);
return {res, exp_v};
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `frexpf(x, exp)` for all possible inputs. The results are
* compared against reference function `double std::frexp(double, int*)`. The maximum ulp error is
* 0.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_frexpf_Accuracy_Positive") {
UnarySinglePrecisionTest(
frexp_kernel<float>, frexp_ref<double>,
PairValidatorBuilderFactory<float, int>(ULPValidatorBuilderFactory<float>(0),
EqValidatorBuilderFactory<int>()));
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `frexp(x, exp)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are
* compared against reference function `long double std::frexp(long double, int*)`. The maximum ulp
* error is 0.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_frexp_Accuracy_Positive") {
UnaryDoublePrecisionTest(
frexp_kernel<double>, frexp_ref<long double>,
PairValidatorBuilderFactory<double, int>(ULPValidatorBuilderFactory<double>(0),
EqValidatorBuilderFactory<int>()));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for frexpf and frexp.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_frexp_frexpf_Negative_RTC") { NegativeTestRTCWrapper<20>(kFrexp); }
/********** Binary Functions **********/
MATH_BINARY_KERNEL_DEF(pow)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `powf(x, y)` and `pow(x, y)`against a table of
* difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::pow(T, T)`. The maximum ulp error
* for single precision is 4 and for double precision is 2.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_pow_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto pow_ref = [](RT arg1, RT arg2) -> RT {
if (std::isinf(arg1) && arg2 < 0) return 0;
return std::pow(arg1, arg2);
};
RT (*ref)(RT, RT) = pow_ref;
const auto ulp = std::is_same_v<float, TestType> ? 4 : 2;
BinaryFloatingPointTest(pow_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(ulp));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for powf and pow.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_pow_powf_Negative_RTC") { NegativeTestRTCWrapper<8>(kPow); }
MATH_POW_INT_KERNEL_DEF(ldexp)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `ldexpf(x, exp)` and `ldexp(x, exp)`against a table of
* difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::ldexp(T, int)`. The maximum ulp error is 0.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_ldexp_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
RT (*ref)(RT, int) = std::ldexp;
PowIntFloatingPointTest(ldexp_kernel<TestType, int>, ref,
ULPValidatorBuilderFactory<TestType>(0));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for ldexpf and ldexp.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_ldexp_ldexpf_Negative_RTC") { NegativeTestRTCWrapper<8>(kLdexp); }
MATH_POW_INT_KERNEL_DEF(powi)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `powi(x, exp)` and `powi(x, exp)`against a table of
* difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::pow(T, T)`. The maximum ulp error
* for single precision is 4 and for double precision is 2.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_powi_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto pow_ref = [](RT arg1, int arg2) -> RT {
if (std::isinf(arg1) && arg2 < 0) return 0;
return std::pow(arg1, static_cast<RT>(arg2));
};
RT (*ref)(RT, int) = pow_ref;
const auto ulp = std::is_same_v<float, TestType> ? 4 : 2;
PowIntFloatingPointTest(powi_kernel<TestType, int>, ref,
ULPValidatorBuilderFactory<TestType>(ulp));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for powif and powi.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_powi_powif_Negative_RTC") { NegativeTestRTCWrapper<8>(kPowi); }
MATH_POW_INT_KERNEL_DEF(scalbn)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `scalbnf(x, n)` and `scalbn(x, n)`against a table of
* difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::scalbn(T, int)`. The maximum ulp error is 0.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_scalbn_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
RT (*ref)(RT, int) = std::scalbn;
PowIntFloatingPointTest(scalbn_kernel<TestType, int>, ref,
ULPValidatorBuilderFactory<TestType>(0));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for scalbnf and scalbn.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_scalbn_scalbnf_Negative_RTC") { NegativeTestRTCWrapper<8>(kScalbn); }
MATH_POW_INT_KERNEL_DEF(scalbln)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `scalblnf(x, l)` and `scalbln(x, l)`against a table of
* difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::scalbn(T, long int)`. The maximum ulp error is 0.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_scalbln_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
RT (*ref)(RT, long int) = std::scalbln;
PowIntFloatingPointTest(scalbln_kernel<TestType, long int>, ref,
ULPValidatorBuilderFactory<TestType>(0));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for scalblnf and scalbln.
*
* Test source
* ------------------------
* - unit/math/pow_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_scalbln_scalblnf_Negative_RTC") { NegativeTestRTCWrapper<8>(kScalbln); }
+246
Wyświetl plik
@@ -0,0 +1,246 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "math_common.hh"
#include "math_special_values.hh"
#include <hip/hip_cooperative_groups.h>
namespace cg = cooperative_groups;
#define MATH_QUATERNARY_KERNEL_DEF(func_name) \
template <typename T> \
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, T* const x1s, T* const x2s, \
T* const x3s, T* const x4s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
if constexpr (std::is_same_v<float, T>) { \
ys[i] = func_name##f(x1s[i], x2s[i], x3s[i], x4s[i]); \
} else if constexpr (std::is_same_v<double, T>) { \
ys[i] = func_name(x1s[i], x2s[i], x3s[i], x4s[i]); \
} \
} \
}
inline constexpr std::array kSpecialValuesReducedDouble{
-std::numeric_limits<double>::quiet_NaN(),
-std::numeric_limits<double>::infinity(),
-std::numeric_limits<double>::max(),
HEX_DBL(-, 1, 0000000000001, +, 64),
HEX_DBL(-, 1, fffffffffffff, +, 63),
HEX_DBL(-, 1, fffffffffffff, +, 62),
HEX_DBL(-, 1, 0, +, 32),
HEX_DBL(-, 1, 0000000000001, +, 31),
HEX_DBL(-, 1, fffffffffffff, +, 30),
-1000.0,
-3.5,
HEX_DBL(-, 1, 8000000000001, +, 1),
-2.5,
HEX_DBL(-, 1, 8000000000001, +, 0),
-1.5,
-0.5,
-0.25,
HEX_DBL(-, 1, fffffffffffff, -, 3),
-std::numeric_limits<double>::min(),
HEX_DBL(-, 0, fffffffffffff, -, 1022),
HEX_DBL(-, 0, 0000000000001, -, 1022),
-0.0,
std::numeric_limits<double>::quiet_NaN(),
std::numeric_limits<double>::infinity(),
std::numeric_limits<double>::max(),
HEX_DBL(+, 1, 0, +, 64),
HEX_DBL(+, 1, 0000000000001, +, 63),
HEX_DBL(+, 1, 000002, +, 32),
HEX_DBL(+, 1, fffffffffffff, +, 31),
HEX_DBL(+, 1, 0, +, 31),
HEX_DBL(+, 1, fffffffffffff, +, 30),
+100.0,
+3.0,
HEX_DBL(+, 1, 7ffffffffffff, +, 1),
+2.0,
HEX_DBL(+, 1, 7ffffffffffff, +, 0),
+1.0,
HEX_DBL(+, 1, fffffffffffff, -, 2),
+std::numeric_limits<double>::min(),
HEX_DBL(+, 0, 0000000000fff, -, 1022),
HEX_DBL(+, 0, 0000000000007, -, 1022),
+0.0,
};
inline constexpr std::array kSpecialValuesReducedFloat{
-std::numeric_limits<float>::quiet_NaN(),
-std::numeric_limits<float>::infinity(),
-std::numeric_limits<float>::max(),
HEX_FLT(-, 1, 000002, +, 64),
HEX_FLT(-, 1, fffffe, +, 63),
HEX_FLT(-, 1, fffffe, +, 62),
HEX_FLT(-, 1, 0, +, 32),
HEX_FLT(-, 1, fffffe, +, 31),
HEX_FLT(-, 1, fffffe, +, 30),
-1000.f,
-3.5f,
HEX_FLT(-, 1, 800002, +, 1),
-2.5f,
HEX_FLT(-, 1, 800002, +, 0),
-1.5f,
-0.5f,
-0.25f,
HEX_FLT(-, 1, fffffe, -, 3),
-std::numeric_limits<float>::min(),
HEX_FLT(-, 0, fffffe, -, 126),
HEX_FLT(-, 0, 000002, -, 126),
-0.0f,
std::numeric_limits<float>::quiet_NaN(),
std::numeric_limits<float>::infinity(),
std::numeric_limits<float>::max(),
HEX_FLT(+, 1, 0, +, 64),
HEX_FLT(+, 1, 000002, +, 63),
HEX_FLT(+, 1, 000002, +, 32),
HEX_FLT(+, 1, 000002, +, 31),
HEX_FLT(+, 1, fffffe, +, 30),
+100.f,
+4.0f,
HEX_FLT(+, 1, 7ffffe, +, 1),
+2.0f,
HEX_FLT(+, 1, 7ffffe, +, 0),
+1.0f,
HEX_FLT(+, 1, fffffe, -, 2),
+std::numeric_limits<float>::min(),
HEX_FLT(+, 0, 000ffe, -, 126),
HEX_FLT(+, 0, 000006, -, 126),
+0.0f,
};
inline constexpr auto kSpecialValReducedRegistry = std::make_tuple(
SpecialVals<float>{kSpecialValuesReducedFloat.data(), kSpecialValuesReducedFloat.size()},
SpecialVals<double>{kSpecialValuesReducedDouble.data(), kSpecialValuesReducedDouble.size()});
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void QuaternaryFloatingPointBruteForceTest(kernel_sig<T, TArg, TArg, TArg, TArg> kernel,
ref_sig<RT, RTArg, RTArg, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder,
const TArg a = std::numeric_limits<TArg>::lowest(),
const TArg b = std::numeric_limits<TArg>::max()) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const uint64_t num_iterations = GetTestIterationCount();
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(TArg) * 4 + sizeof(T)), num_iterations);
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
LinearAllocGuard<TArg> x3s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
LinearAllocGuard<TArg> x4s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &x1s, &x2s, &x3s, &x4s] {
const auto generator = [=] {
static thread_local std::mt19937 rng(std::random_device{}());
std::uniform_real_distribution<RefType_t<TArg>> unif_dist(a, b);
return static_cast<TArg>(unif_dist(rng));
};
std::generate(x1s.ptr() + base_idx, x1s.ptr() + base_idx + sub_batch_size, generator);
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator);
std::generate(x3s.ptr() + base_idx, x3s.ptr() + base_idx + sub_batch_size, generator);
std::generate(x4s.ptr() + base_idx, x4s.ptr() + base_idx + sub_batch_size, generator);
});
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
x2s.ptr(), x3s.ptr(), x4s.ptr());
}
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void QuaternaryFloatingPointSpecialValuesTest(kernel_sig<T, TArg, TArg, TArg, TArg> kernel,
ref_sig<RT, RTArg, RTArg, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const auto values = std::get<SpecialVals<TArg>>(kSpecialValReducedRegistry);
const auto size = values.size * values.size * values.size * values.size;
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
LinearAllocGuard<TArg> x3s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
LinearAllocGuard<TArg> x4s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
for (auto i = 0u; i < values.size; ++i) {
for (auto j = 0u; j < values.size; ++j) {
for (auto k = 0u; k < values.size; ++k) {
for (auto l = 0u; l < values.size; ++l) {
x1s.ptr()[((i * values.size + j) * values.size + k) * values.size + l] = values.data[i];
x2s.ptr()[((i * values.size + j) * values.size + k) * values.size + l] = values.data[j];
x3s.ptr()[((i * values.size + j) * values.size + k) * values.size + l] = values.data[k];
x4s.ptr()[((i * values.size + j) * values.size + k) * values.size + l] = values.data[l];
}
}
}
}
MathTest math_test(kernel, size);
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, size, x1s.ptr(),
x2s.ptr(), x3s.ptr(), x4s.ptr());
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void QuaternaryFloatingPointTest(kernel_sig<T, TArg, TArg, TArg, TArg> kernel,
ref_sig<RT, RTArg, RTArg, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
SECTION("Special values") {
QuaternaryFloatingPointSpecialValuesTest(kernel, ref_func, validator_builder);
}
SECTION("Brute force") {
QuaternaryFloatingPointBruteForceTest(kernel, ref_func, validator_builder);
}
}
#define MATH_QUATERNARY_WITHIN_ULP_TEST_DEF(kern_name, ref_func, sp_ulp, dp_ulp) \
MATH_QUATERNARY_KERNEL_DEF(kern_name) \
\
TEMPLATE_TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive", "", float, double) { \
using RT = RefType_t<TestType>; \
RT (*ref)(RT, RT, RT, RT) = ref_func; \
const auto ulp = std::is_same_v<float, TestType> ? sp_ulp : dp_ulp; \
\
QuaternaryFloatingPointTest(kern_name##_kernel<TestType>, ref, \
ULPValidatorBuilderFactory<TestType>(ulp)); \
}
@@ -0,0 +1,153 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "unary_common.hh"
#include "binary_common.hh"
#include "math_remainder_rounding_negative_kernels_rtc.hh"
MATH_BINARY_WITHIN_ULP_TEST_DEF(fmod, std::fmod, 0, 0)
TEST_CASE("Unit_Device_fmod_fmodf_Negative_RTC") { NegativeTestRTCWrapper<8>(kFmod); }
MATH_BINARY_WITHIN_ULP_TEST_DEF(remainder, std::remainder, 0, 0)
TEST_CASE("Unit_Device_remainder_remainder_Negative_RTC") { NegativeTestRTCWrapper<8>(kRemainder); }
MATH_BINARY_WITHIN_ULP_TEST_DEF(fdim, std::fdim, 0, 0)
TEST_CASE("Unit_Device_fdim_fdimf_Negative_RTC") { NegativeTestRTCWrapper<8>(kFdim); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(trunc, std::trunc, 0, 0)
TEST_CASE("Unit_Device_trunc_truncf_Negative_RTC") { NegativeTestRTCWrapper<4>(kTrunc); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(round, std::round, 0, 0)
TEST_CASE("Unit_Device_round_roundf_Negative_RTC") { NegativeTestRTCWrapper<4>(kRound); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(rint, std::rint, 0, 0)
TEST_CASE("Unit_Device_rint_rintf_Negative_RTC") { NegativeTestRTCWrapper<4>(kRint); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(nearbyint, std::nearbyint, 0, 0)
TEST_CASE("Unit_Device_nearbyint_nearbyintf_Negative_RTC") {
NegativeTestRTCWrapper<4>(kNearbyint);
}
MATH_UNARY_WITHIN_ULP_TEST_DEF(ceil, std::ceil, 0, 0)
TEST_CASE("Unit_Device_ceil_ceilf_Negative_RTC") { NegativeTestRTCWrapper<4>(kCeil); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(floor, std::floor, 0, 0)
TEST_CASE("Unit_Device_floor_floorf_Negative_RTC") { NegativeTestRTCWrapper<4>(kFloor); }
#define LONG_CONVERSION_FUNCTION_TEST_DEF(kern_name, ref_func, lt) \
MATH_UNARY_KERNEL_DEF(kern_name) \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - float") { \
lt (*ref)(double) = ref_func; \
UnarySinglePrecisionRangeTest(kern_name##_kernel<float, lt>, ref, \
EqValidatorBuilderFactory<lt>(), \
static_cast<float>(std::numeric_limits<lt>::lowest()), \
static_cast<float>(std::numeric_limits<lt>::max())); \
} \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - double") { \
lt (*ref)(long double) = ref_func; \
UnaryDoublePrecisionBruteForceTest(kern_name##_kernel<double, lt>, ref, \
EqValidatorBuilderFactory<lt>(), \
static_cast<double>(std::numeric_limits<lt>::lowest()), \
static_cast<double>(std::numeric_limits<lt>::max())); \
}
LONG_CONVERSION_FUNCTION_TEST_DEF(lrint, std::lrint, long)
TEST_CASE("Unit_Device_lrint_lrintf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLrint); }
LONG_CONVERSION_FUNCTION_TEST_DEF(lround, std::lround, long)
TEST_CASE("Unit_Device_lround_lroundf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLround); }
LONG_CONVERSION_FUNCTION_TEST_DEF(llrint, std::llrint, long long)
TEST_CASE("Unit_Device_llrint_llrintf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLlrint); }
LONG_CONVERSION_FUNCTION_TEST_DEF(llround, std::llround, long long)
TEST_CASE("Unit_Device_llround_llroundf_Negative_RTC") { NegativeTestRTCWrapper<4>(kLlround); }
template <typename T>
__global__ void remquo_kernel(std::pair<T, int>* const ys, const size_t num_xs, T* const x1s,
T* const x2s) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
if constexpr (std::is_same_v<float, T>) {
ys[i].first = remquof(x1s[i], x2s[i], &ys[i].second);
} else if constexpr (std::is_same_v<double, T>) {
ys[i].first = remquo(x1s[i], x2s[i], &ys[i].second);
}
}
}
template <typename T> std::pair<T, int> remquo_wrapper(T x1, T x2) {
std::pair<T, int> ret;
ret.first = std::remquo(x1, x2, &ret.second);
return ret;
}
TEMPLATE_TEST_CASE("Unit_Device_remquo_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
std::pair<RT, int> (*ref)(RT, RT) = remquo_wrapper;
const auto ulp_builder = ULPValidatorBuilderFactory<TestType>(0);
const auto eq_builder = EqValidatorBuilderFactory<int>();
BinaryFloatingPointTest(remquo_kernel<TestType>, ref,
PairValidatorBuilderFactory<TestType, int>(ulp_builder, eq_builder));
}
TEST_CASE("Unit_Device_remquo_remquof_Negative_RTC") { NegativeTestRTCWrapper<24>(kRemquo); }
template <typename T>
__global__ void modf_kernel(std::pair<T, T>* const ys, const size_t num_xs, T* const xs) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
if constexpr (std::is_same_v<float, T>) {
ys[i].first = modff(xs[i], &ys[i].second);
} else if constexpr (std::is_same_v<double, T>) {
ys[i].first = modf(xs[i], &ys[i].second);
}
}
}
template <typename T> std::pair<T, T> modf_wrapper(T x) {
std::pair<T, T> ret;
ret.first = std::modf(x, &ret.second);
return ret;
}
TEST_CASE("Unit_Device_modf_Accuracy_Positive - float") {
UnarySinglePrecisionTest(
modf_kernel<float>, modf_wrapper<double>,
PairValidatorBuilderFactory<float>(ULPValidatorBuilderFactory<float>(0)));
}
TEST_CASE("Unit_Device_modf_Accuracy_Positive - double") {
UnaryDoublePrecisionTest(
modf_kernel<double>, modf_wrapper<long double>,
PairValidatorBuilderFactory<double>(ULPValidatorBuilderFactory<double>(0)));
}
TEST_CASE("Unit_Device_modf_modff_Negative_RTC") { NegativeTestRTCWrapper<20>(kModf); }
+604
Wyświetl plik
@@ -0,0 +1,604 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "unary_common.hh"
#include "binary_common.hh"
#include "ternary_common.hh"
#include "quaternary_common.hh"
#include "math_root_negative_kernels_rtc.hh"
/**
* @addtogroup RootMathFuncs RootMathFuncs
* @{
* @ingroup MathTest
*/
/********** Unary Functions **********/
MATH_UNARY_KERNEL_DEF(sqrt)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `sqrtf(x)` for all possible inputs. The results are
* compared against reference function `float std::exp(float)`. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_sqrtf_Accuracy_Positive") {
float (*ref)(float) = std::sqrt;
UnarySinglePrecisionTest(sqrt_kernel<float>, ref, ULPValidatorBuilderFactory<float>(1));
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `sqrt(x)` against a table of difficult values,
* followed by a large number of randomly generated values. The results are
* compared against reference function `double std::sqrt(double)`. The error bounds are
* IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_sqrt_Accuracy_Positive") {
double (*ref)(double) = std::sqrt;
UnaryDoublePrecisionTest<double>(sqrt_kernel<double>, ref, ULPValidatorBuilderFactory<double>(0));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for sqrtf and sqrt.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_sqrt_sqrtf_Negative_RTC") { NegativeTestRTCWrapper<4>(kSqrt); }
MATH_UNARY_KERNEL_DEF(rsqrt)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `rsqrtf(x)` for all possible inputs. The maximum ulp error
* is 2.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rsqrtf_Accuracy_Positive") {
auto rsqrt_ref = [](double arg) -> double { return 1. / std::sqrt(arg); };
double (*ref)(double) = rsqrt_ref;
UnarySinglePrecisionTest(rsqrt_kernel<float>, ref, ULPValidatorBuilderFactory<float>(2));
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `rsqrt(x)` against a table of difficult values,
* followed by a large number of randomly generated values. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rsqrt_Accuracy_Positive") {
auto rsqrt_ref = [](long double arg) -> long double { return 1.L / std::sqrt(arg); };
long double (*ref)(long double) = rsqrt_ref;
UnaryDoublePrecisionTest(rsqrt_kernel<double>, ref, ULPValidatorBuilderFactory<double>(1));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for rsqrtf and rsqrt.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rsqrt_rsqrtf_Negative_RTC") { NegativeTestRTCWrapper<4>(kRsqrt); }
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `cbrtf(x)` for all possible inputs and `cbrt(x)` against a
* table of difficult values, followed by a large number of randomly generated values. The results
* are compared against reference function `T std::cbrt(T)`. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_WITHIN_ULP_TEST_DEF(cbrt, std::cbrt, 1, 1)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for cbrtf and cbrt.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_cbrt_cbrtf_Negative_RTC") { NegativeTestRTCWrapper<4>(kCbrt); }
MATH_UNARY_KERNEL_DEF(rcbrt)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `rcbrtf(x)` for all possible inputs. The maximum ulp error
* is 1.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rcbrtf_Accuracy_Positive") {
auto rcbrt_ref = [](double arg) -> double { return 1. / std::cbrt(arg); };
double (*ref)(double) = rcbrt_ref;
UnarySinglePrecisionTest(rcbrt_kernel<float>, ref, ULPValidatorBuilderFactory<float>(1));
}
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `rcbrt(x)` against a table of difficult values,
* followed by a large number of randomly generated values. The maximum ulp error is 1.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rcbrt_Accuracy_Positive") {
auto rcbrt_ref = [](long double arg) -> long double { return 1. / std::cbrt(arg); };
long double (*ref)(long double) = rcbrt_ref;
UnaryDoublePrecisionTest(rcbrt_kernel<double>, ref, ULPValidatorBuilderFactory<double>(1));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass argument of invalid type for rcbrtf and rcbrt.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rcbrt_rcbrtf_Negative_RTC") { NegativeTestRTCWrapper<4>(kRcbrt); }
/********** Binary Functions **********/
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `hypotf(x, y)` and `hypot(x, y)` against a table of
* difficult values, followed by a large number of randomly generated values. The results are
* compared against reference function `T std::hypot(T, T)`. The maximum ulp error for single
* precision is 3 and for double precision is 2.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_WITHIN_ULP_TEST_DEF(hypot, std::hypot, 3, 2)
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for hypotf and hypot.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_hypot_hypotf_Negative_RTC") { NegativeTestRTCWrapper<8>(kHypot); }
MATH_BINARY_KERNEL_DEF(rhypot)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `rhypotf(x, y)` and `rhypot(x, y)`against a table of
* difficult values, followed by a large number of randomly generated values. The maximum ulp error
* for single precision is 2 and for double precision is 1.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_rhypot_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto rhypot_ref = [](RT arg1, RT arg2) -> RT { return 1. / std::hypot(arg1, arg2); };
RT (*ref)(RT, RT) = rhypot_ref;
const auto ulp = std::is_same_v<float, TestType> ? 2 : 1;
BinaryFloatingPointTest(rhypot_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(ulp));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for rhypotf and rhypot.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rhypot_rhypotf_Negative_RTC") { NegativeTestRTCWrapper<8>(kRhypot); }
/********** Ternary Functions **********/
MATH_TERNARY_KERNEL_DEF(norm3d)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `norm3df(x, y, z)` and `norm3d(x, y, z)` against a table of
* difficult values, followed by a large number of randomly generated values. The maximum ulp error
* for single precision is 3 and for double precision is 2.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_norm3d_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto norm3d_ref = [](RT arg1, RT arg2, RT arg3) -> RT {
if (std::isinf(arg1) || std::isinf(arg2) || std::isinf(arg3)) {
return std::numeric_limits<RT>::infinity();
}
return std::sqrt(arg1 * arg1 + arg2 * arg2 + arg3 * arg3);
};
RT (*ref)(RT, RT, RT) = norm3d_ref;
const auto ulp = std::is_same_v<float, TestType> ? 3 : 2;
TernaryFloatingPointTest(norm3d_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(ulp));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for norm3df and norm3d.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_norm3d_norm3df_Negative_RTC") { NegativeTestRTCWrapper<12>(kNorm3D); }
MATH_TERNARY_KERNEL_DEF(rnorm3d)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `rnorm3df(x, y, z)` and `rnorm3d(x, y, z)`against a table of
* difficult values, followed by a large number of randomly generated values. The maximum ulp error
* for single precision is 2 and for double precision is 1.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_rnorm3d_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto rnorm3d_ref = [](RT arg1, RT arg2, RT arg3) -> RT {
if (std::isinf(arg1) || std::isinf(arg2) || std::isinf(arg3)) {
return 0;
}
return 1. / std::sqrt(arg1 * arg1 + arg2 * arg2 + arg3 * arg3);
};
RT (*ref)(RT, RT, RT) = rnorm3d_ref;
const auto ulp = std::is_same_v<float, TestType> ? 2 : 1;
TernaryFloatingPointTest(rnorm3d_kernel<TestType>, ref,
ULPValidatorBuilderFactory<TestType>(ulp));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for rnorm3df and rnorm3d.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rnorm3d_rnorm3df_Negative_RTC") { NegativeTestRTCWrapper<12>(kRnorm3D); }
/********** Quaternary Functions **********/
MATH_QUATERNARY_KERNEL_DEF(norm4d)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `norm4df(x, y, z, t)` and `norm4d(x, y, z, t)` against a
* table of difficult values, followed by a large number of randomly generated values. The maximum
* ulp error for single precision is 3 and for double precision is 2.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_norm4d_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto norm4d_ref = [](RT arg1, RT arg2, RT arg3, RT arg4) -> RT {
if (std::isinf(arg1) || std::isinf(arg2) || std::isinf(arg3) || std::isinf(arg4)) {
return std::numeric_limits<RT>::infinity();
}
return std::sqrt(arg1 * arg1 + arg2 * arg2 + arg3 * arg3 + arg4 * arg4);
};
RT (*ref)(RT, RT, RT, RT) = norm4d_ref;
const auto ulp = std::is_same_v<float, TestType> ? 3 : 2;
QuaternaryFloatingPointTest(norm4d_kernel<TestType>, ref,
ULPValidatorBuilderFactory<TestType>(ulp));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for norm4df and norm4d.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_norm4d_norm4df_Negative_RTC") { NegativeTestRTCWrapper<16>(kNorm4D); }
MATH_QUATERNARY_KERNEL_DEF(rnorm4d)
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `rnorm4df(x, y, z, t)` and `rnorm4d(x, y, z, t)`against a
* table of difficult values, followed by a large number of randomly generated values. The maximum
* ulp error for single precision is 2 and for double precision is 1.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_rnorm4d_Accuracy_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto rnorm4d_ref = [](RT arg1, RT arg2, RT arg3, RT arg4) -> RT {
if (std::isinf(arg1) || std::isinf(arg2) || std::isinf(arg3) || std::isinf(arg4)) {
return 0;
}
return 1. / std::sqrt(arg1 * arg1 + arg2 * arg2 + arg3 * arg3 + arg4 * arg4);
};
RT (*ref)(RT, RT, RT, RT) = rnorm4d_ref;
const auto ulp = std::is_same_v<float, TestType> ? 2 : 1;
QuaternaryFloatingPointTest(rnorm4d_kernel<TestType>, ref,
ULPValidatorBuilderFactory<TestType>(ulp));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for rnorm4df and rnorm4d.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rnorm4d_rnorm4df_Negative_RTC") { NegativeTestRTCWrapper<16>(kRnorm4D); }
/********** norm Function **********/
#define MATH_NORM_KERNEL_DEF(func_name) \
template <typename T> __global__ void func_name##_kernel(T* const ys, int dim, T* const x1s) { \
if constexpr (std::is_same_v<float, T>) { \
*ys = func_name##f(dim, x1s); \
} else if constexpr (std::is_same_v<double, T>) { \
*ys = func_name(dim, x1s); \
} \
}
template <typename T, typename F, typename RF, typename ValidatorBuilder>
void NormSimpleTest(F kernel, RF ref_func, const ValidatorBuilder& validator_builder) {
const auto max_dim = 10000;
LinearAllocGuard<T> x{LinearAllocs::hipHostMalloc, max_dim * sizeof(T)};
LinearAllocGuard<T> x_dev{LinearAllocs::hipMalloc, max_dim * sizeof(T)};
LinearAllocGuard<T> y{LinearAllocs::hipHostMalloc, sizeof(T)};
LinearAllocGuard<T> y_dev{LinearAllocs::hipMalloc, sizeof(T)};
std::fill_n(x.ptr(), max_dim, 1);
HIP_CHECK(hipMemcpy(x_dev.ptr(), x.ptr(), max_dim * sizeof(T), hipMemcpyHostToDevice));
for (uint64_t i = 1u; i < max_dim; i++) {
kernel<<<1, 1>>>(y_dev.ptr(), i, x_dev.ptr());
HIP_CHECK(hipGetLastError());
HIP_CHECK(hipMemcpy(y.ptr(), y_dev.ptr(), sizeof(T), hipMemcpyDeviceToHost));
const auto actual_val = *y.ptr();
const auto ref_val = static_cast<T>(ref_func(i, x.ptr()));
const auto validator = validator_builder(ref_val);
if (!validator->match(actual_val)) {
std::stringstream ss;
ss << std::scientific << std::setprecision(std::numeric_limits<T>::max_digits10 - 1);
ss << "Validation fails for dim: " << i << " " << actual_val << " " << ref_val;
INFO(ss.str());
REQUIRE(false);
}
}
}
MATH_NORM_KERNEL_DEF(norm)
/**
* Test Description
* ------------------------
* - Sanity test for `normf(dim, arr)` and `norm(dim, arr)`.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_norm_Sanity_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto norm_ref = [](int dim, TestType* args) -> RT {
RT sum = 0;
for (int i = 0; i < dim; i++) {
if (std::isinf(args[i])) return std::numeric_limits<RT>::infinity();
sum += static_cast<RT>(args[i]) * static_cast<RT>(args[i]);
}
return std::sqrt(sum);
};
RT (*ref)(int, TestType*) = norm_ref;
NormSimpleTest<TestType>(norm_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(10));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for normf and norm.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_norm_normf_Negative_RTC") { NegativeTestRTCWrapper<18>(kNorm); }
MATH_NORM_KERNEL_DEF(rnorm)
/**
* Test Description
* ------------------------
* - Sanity test for `rnormf(dim, arr)` and `rnorm(dim, arr)`.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEMPLATE_TEST_CASE("Unit_Device_rnorm_Sanity_Positive", "", float, double) {
using RT = RefType_t<TestType>;
auto rnorm_ref = [](int dim, TestType* args) -> RT {
RT sum = 0;
for (int i = 0; i < dim; i++) {
if (std::isinf(args[i])) return std::numeric_limits<RT>::infinity();
sum += static_cast<RT>(args[i]) * static_cast<RT>(args[i]);
}
return 1. / std::sqrt(sum);
};
RT (*ref)(int, TestType*) = rnorm_ref;
NormSimpleTest<TestType>(rnorm_kernel<TestType>, ref, ULPValidatorBuilderFactory<TestType>(10));
}
/**
* Test Description
* ------------------------
* - RTCs kernels that pass combinations of arguments of invalid types for rnormf and rnorm.
*
* Test source
* ------------------------
* - unit/math/root_funcs.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_Device_rnorm_rnormf_Negative_RTC") { NegativeTestRTCWrapper<18>(kRnorm); }
@@ -0,0 +1,530 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
#include "unary_common.hh"
#include "binary_common.hh"
#include "ternary_common.hh"
/********** Unary Functions **********/
#define MATH_UNARY_SP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(float* const ys, const size_t num_xs, float* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(xs[i]); \
} \
}
#define MATH_UNARY_SP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
UnarySinglePrecisionTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_UNARY_SP_TEST_DEF(func_name, ref_func) \
MATH_UNARY_SP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x)
static float __frcp_rn_ref(float x) { return 1.0f / x; }
MATH_UNARY_SP_KERNEL_DEF(__frcp_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__frcp_rn(x)` for all possible inputs. The error bounds are
* IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF_IMPL(__frcp_rn, __frcp_rn_ref, EqValidatorBuilderFactory<float>());
MATH_UNARY_SP_KERNEL_DEF(__fsqrt_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__fsqrt_rn(x)` for all possible inputs. The results are
* compared against reference function `float std::sqrt(float)`. The error bounds are
* IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF_IMPL(__fsqrt_rn, static_cast<float (*)(float)>(std::sqrt),
EqValidatorBuilderFactory<float>());
static float __frsqrt_rn_ref(float x) { return 1.0f / std::sqrt(x); }
MATH_UNARY_SP_KERNEL_DEF(__frsqrt_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__frsqrt_rn(x)` for all possible inputs. The results are
* compared against reference function `float std::sqrt(float)`. The error bounds are
* IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF_IMPL(__frsqrt_rn, __frsqrt_rn_ref, EqValidatorBuilderFactory<float>());
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__expf) {
const int64_t ulp_err = 2 + static_cast<int64_t>(std::floor(std::abs(1.16f * x)));
return ULPValidatorBuilderFactory<float>(ulp_err)(target);
}
MATH_UNARY_SP_KERNEL_DEF(__expf);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__expf(x)` for all possible inputs. The results are
* compared against reference function `double std::exp(double)`. The maximum ulp error is `2 +
* floor(abs(1.16 * x))`.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF(__expf, static_cast<double (*)(double)>(std::exp));
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__exp10f) {
const int64_t ulp_err = 2 + static_cast<int64_t>(std::floor(std::abs(2.95f * x)));
return ULPValidatorBuilderFactory<float>(ulp_err)(target);
}
MATH_UNARY_SP_KERNEL_DEF(__exp10f);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__exp10f(x)` for all possible inputs. The results are
* compared against reference function `double exp10(double)`. The maximum ulp error is `2 +
* floor(abs(2.95 * x))`.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF(__exp10f, static_cast<double (*)(double)>(exp10));
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__logf) {
if (0.5f <= x && x <= 2.0f) {
const auto abs_err = std::pow(2.0, -21.41);
return AbsValidatorBuilderFactory<float>(abs_err)(target);
} else {
return ULPValidatorBuilderFactory<float>(3)(target);
}
}
MATH_UNARY_SP_KERNEL_DEF(__logf);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__logf(x)` for all possible inputs. The results are
* compared against reference function `double std::log(double)`. For `x` in [0.5, 2], the maximum
* absolute error is 2^-21.41, otherwise, the maximum ulp error is 3.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF(__logf, static_cast<double (*)(double)>(std::log));
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__log2f) {
if (0.5f <= x && x <= 2.0f) {
const auto abs_err = std::pow(2.0, -22.0);
return AbsValidatorBuilderFactory<float>(abs_err)(target);
} else {
return ULPValidatorBuilderFactory<float>(2)(target);
}
}
MATH_UNARY_SP_KERNEL_DEF(__log2f);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__log2f(x)` for all possible inputs. The results are
* compared against reference function `double std::log2(double)`. For `x` in [0.5, 2], the maximum
* absolute error is 2^-22, otherwise, the maximum ulp error is 2.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF(__log2f, static_cast<double (*)(double)>(std::log2));
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__log10f) {
if (0.5f <= x && x <= 2.0f) {
const auto abs_err = std::pow(2.0, -24.0);
return AbsValidatorBuilderFactory<float>(abs_err)(target);
} else {
return ULPValidatorBuilderFactory<float>(3)(target);
}
}
MATH_UNARY_SP_KERNEL_DEF(__log10f);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__log10f(x)` for all possible inputs. The results are
* compared against reference function `double std::log10(double)`. For `x` in [0.5, 2], the maximum
* absolute error is 2^-24, otherwise, the maximum ulp error is 3.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF(__log10f, static_cast<double (*)(double)>(std::log10));
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__sinf) {
if (-M_PI <= x && x <= M_PI) {
const auto abs_err = std::pow(2.0, -21.41);
return AbsValidatorBuilderFactory<float>(abs_err)(target);
} else {
return NopValidatorBuilderFactory<float>()();
}
}
MATH_UNARY_SP_KERNEL_DEF(__sinf);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__sinf(x)` for all possible inputs. The results are
* compared against reference function `double std::sin(double)`. For `x` in [-PI, PI], the maximum
* absolute error is 2^-21.41, and larger otherwise.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF(__sinf, static_cast<double (*)(double)>(std::sin));
__device__ float __sincosf_sin(float x) {
float sin, cos;
__sincosf(x, &sin, &cos);
return sin;
}
MATH_UNARY_SP_KERNEL_DEF(__sincosf_sin);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__sincosf(x, sptr, cptr)` for all possible inputs. The
* results in `sptr` are compared against reference function `double std::sin(double)`. For `x` in
* [-PI, PI], the maximum absolute error is 2^-21.41, and larger otherwise.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF_IMPL(__sincosf_sin, static_cast<double (*)(double)>(std::sin),
__sinf_validator_builder);
MATH_UNARY_SP_VALIDATOR_BUILDER_DEF(__cosf) {
if (-M_PI <= x && x <= M_PI) {
const auto abs_err = std::pow(2.0, -21.19);
return AbsValidatorBuilderFactory<float>(abs_err)(target);
} else {
return NopValidatorBuilderFactory<float>()();
}
}
MATH_UNARY_SP_KERNEL_DEF(__cosf);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__cosf(x)` for all possible inputs. The results are
* compared against reference function `double std::cos(double)`. For `x` in [-PI, PI], the maximum
* absolute error is 2^-21.19, and larger otherwise.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF(__cosf, static_cast<double (*)(double)>(std::cos));
__device__ float __sincosf_cos(float x) {
float sin, cos;
__sincosf(x, &sin, &cos);
return cos;
}
MATH_UNARY_SP_KERNEL_DEF(__sincosf_cos);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__sincosf(x, sptr, cptr)` for all possible inputs. The
* results in `cptr` are compared against reference function `double std::cos(double)`. For `x` in
* [-PI, PI], the maximum absolute error is 2^-21.19, and larger otherwise.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_UNARY_SP_TEST_DEF_IMPL(__sincosf_cos, static_cast<double (*)(double)>(std::cos),
__cosf_validator_builder);
/********** Binary Functions **********/
#define MATH_BINARY_SP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(float* const ys, const size_t num_xs, float* const x1s, \
float* const x2s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(x1s[i], x2s[i]); \
} \
}
#define MATH_BINARY_SP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
BinaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_BINARY_SP_TEST_DEF(func_name, ref_func) \
MATH_BINARY_SP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_BINARY_SP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x1, \
float x2)
static float __fadd_rn_ref(float x1, float x2) { return x1 + x2; }
MATH_BINARY_SP_KERNEL_DEF(__fadd_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__fadd_rn(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_SP_TEST_DEF_IMPL(__fadd_rn, __fadd_rn_ref, EqValidatorBuilderFactory<float>());
static float __fsub_rn_ref(float x1, float x2) { return x1 - x2; }
MATH_BINARY_SP_KERNEL_DEF(__fsub_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__fsub_rn(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_SP_TEST_DEF_IMPL(__fsub_rn, __fsub_rn_ref, EqValidatorBuilderFactory<float>());
static float __fmul_rn_ref(float x1, float x2) { return x1 * x2; }
MATH_BINARY_SP_KERNEL_DEF(__fmul_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__fmul_rn(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_SP_TEST_DEF_IMPL(__fmul_rn, __fmul_rn_ref, EqValidatorBuilderFactory<float>());
static float __fdiv_rn_ref(float x1, float x2) { return x1 / x2; }
MATH_BINARY_SP_KERNEL_DEF(__fdiv_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__fdiv_rn(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_SP_TEST_DEF_IMPL(__fdiv_rn, __fdiv_rn_ref, EqValidatorBuilderFactory<float>());
MATH_BINARY_SP_VALIDATOR_BUILDER_DEF(__fdividef) {
x1 = 2.0f;
const auto abs_x2 = std::abs(x2);
if (std::pow(x1, -126.0f) <= abs_x2 && abs_x2 <= std::pow(x1, 126.0f)) {
return ULPValidatorBuilderFactory<float>(2)(target);
} else {
return NopValidatorBuilderFactory<float>()();
}
}
MATH_BINARY_SP_KERNEL_DEF(__fdividef);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__fdividef(x,y)` against a table of difficult values,
* followed by a large number of randomly generated values. For `|y|` in [2^-126, 2^126], the
* maximum ulp error is 2.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_BINARY_SP_TEST_DEF(__fdividef, __fdiv_rn_ref);
/********** Ternary Functions **********/
#define MATH_TERNARY_SP_KERNEL_DEF(func_name) \
__global__ void func_name##_kernel(float* const ys, const size_t num_xs, float* const x1s, \
float* const x2s, float* const x3s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
ys[i] = func_name(x1s[i], x2s[i], x3s[i]); \
} \
}
#define MATH_TERNARY_SP_TEST_DEF_IMPL(func_name, ref_func, validator_builder) \
TEST_CASE("Unit_Device_" #func_name "_Accuracy_Positive") { \
TernaryFloatingPointTest(func_name##_kernel, ref_func, validator_builder); \
}
#define MATH_TERNARY_SP_TEST_DEF(func_name, ref_func, validator_builder) \
MATH_TERNARY_SP_TEST_DEF_IMPL(func_name, ref_func, func_name##_validator_builder)
#define MATH_TERNARY_SP_VALIDATOR_BUILDER_DEF(func_name) \
static std::unique_ptr<MatcherBase<float>> func_name##_validator_builder(float target, float x1, \
float x2, float x3)
MATH_TERNARY_SP_KERNEL_DEF(__fmaf_rn);
/**
* Test Description
* ------------------------
* - Tests the numerical accuracy of `__fmaf(x,y,z)` against a table of difficult values,
* followed by a large number of randomly generated values. The error bounds are IEEE-compliant.
*
* Test source
* ------------------------
* - unit/math/single_precision_intrinsics.cc
* Test requirements
* ------------------------
* - HIP_VERSION >= 5.2
*/
MATH_TERNARY_SP_TEST_DEF_IMPL(__fmaf_rn, static_cast<float (*)(float, float, float)>(std::fma),
EqValidatorBuilderFactory<float>());
@@ -0,0 +1,56 @@
/*
Copyright (c) 2021 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(float* x) { float result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { float result = func_name(x); }
#define INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(float* x, float y) { float result = func_name(x, y); } \
__global__ void func_name##_kernel_v2(float x, float* y) { float result = func_name(x, y); } \
__global__ void func_name##_kernel_v3(Dummy x, float y) { float result = func_name(x, y); } \
__global__ void func_name##_kernel_v4(float x, Dummy y) { float result = func_name(x, y); }
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__fsqrt_rn)
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__expf)
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__exp10f)
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__logf)
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__log2f)
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__log10f)
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__sinf)
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__cosf)
INTRINSIC_UNARY_FLOAT_NEGATIVE_KERNELS(__tanf)
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fadd_rn)
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fsub_rn)
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fmul_rn)
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fdiv_rn)
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__fdividef)
INTRINSIC_BINARY_FLOAT_NEGATIVE_KERNELS(__powf)
+145
Wyświetl plik
@@ -0,0 +1,145 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "math_common.hh"
#include "math_special_values.hh"
#include <hip/hip_cooperative_groups.h>
namespace cg = cooperative_groups;
#define MATH_BESSEL_N_KERNEL_DEF(func_name) \
template <typename T> \
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, int* n, T* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
if constexpr (std::is_same_v<float, T>) { \
ys[i] = func_name##f(n[i], xs[i]); \
} else if constexpr (std::is_same_v<double, T>) { \
ys[i] = func_name(n[i], xs[i]); \
} \
} \
}
template <typename T> using kernel_bessel_n_sig = void (*)(T*, const size_t, int*, T*);
template <typename T> using ref_bessel_n_sig = T (*)(int, T);
template <typename ValidatorBuilder>
void BesselDoublePrecisionBruteForceTest(kernel_bessel_n_sig<double> kernel,
ref_bessel_n_sig<long double> ref_func,
const ValidatorBuilder& validator_builder, int n_input = 0,
const double a = std::numeric_limits<double>::lowest(),
const double b = std::numeric_limits<double>::max()) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const uint64_t num_iterations = GetTestIterationCount();
const auto max_batch_size = std::min(
GetMaxAllowedDeviceMemoryUsage() / (sizeof(double) * 2 + sizeof(int)), num_iterations);
LinearAllocGuard<int> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(int)};
LinearAllocGuard<double> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(double)};
MathTest math_test(kernel, max_batch_size);
std::fill_n(x1s.ptr(), max_batch_size, n_input);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &x2s] {
const auto generator = [=] {
static thread_local std::mt19937 rng(std::random_device{}());
std::uniform_real_distribution<RefType_t<double>> unif_dist(a, b);
return static_cast<double>(unif_dist(rng));
};
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator);
});
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
x2s.ptr());
}
}
template <typename ValidatorBuilder>
void BesselSinglePrecisionRangeTest(kernel_bessel_n_sig<float> kernel,
ref_bessel_n_sig<double> ref_func,
const ValidatorBuilder& validator_builder, int n_input,
const float a, const float b) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const auto max_batch_size = GetMaxAllowedDeviceMemoryUsage() / (sizeof(float) * 2 + sizeof(int));
LinearAllocGuard<int> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(int)};
LinearAllocGuard<float> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(float)};
MathTest math_test(kernel, max_batch_size);
std::fill_n(x1s.ptr(), max_batch_size, n_input);
size_t inserted = 0u;
for (float v = a; v != b; v = std::nextafter(v, b)) {
x2s.ptr()[inserted++] = v;
if (inserted < max_batch_size) continue;
math_test.Run(validator_builder, grid_size, block_size, ref_func, inserted, x1s.ptr(),
x2s.ptr());
inserted = 0u;
}
}
template <typename T, typename F, typename ValidatorBuilder>
void SpecialSimpleTest(F kernel, const ValidatorBuilder& validator_builder, const T* x,
const T* ref, size_t num_args) {
LinearAllocGuard<T> x_dev{LinearAllocs::hipMalloc, num_args * sizeof(T)};
LinearAllocGuard<T> y{LinearAllocs::hipHostMalloc, num_args * sizeof(T)};
LinearAllocGuard<T> y_dev{LinearAllocs::hipMalloc, num_args * sizeof(T)};
HIP_CHECK(hipMemcpy(x_dev.ptr(), x, num_args * sizeof(T), hipMemcpyHostToDevice));
kernel<<<1, num_args>>>(y_dev.ptr(), num_args, x_dev.ptr());
HIP_CHECK(hipGetLastError());
HIP_CHECK(hipMemcpy(y.ptr(), y_dev.ptr(), num_args * sizeof(T), hipMemcpyDeviceToHost));
for (auto i = 0u; i < num_args; ++i) {
const auto actual_val = y.ptr()[i];
const auto ref_val = ref[i];
const auto validator = validator_builder(ref_val);
if (!validator->match(actual_val)) {
std::stringstream ss;
ss << "Input value(s): " << std::scientific
<< std::setprecision(std::numeric_limits<T>::max_digits10 - 1);
ss << x[i] << " " << actual_val << " " << ref_val << "\n";
INFO(ss.str());
REQUIRE(false);
}
}
}
Plik diff jest za duży Load Diff
+151
Wyświetl plik
@@ -0,0 +1,151 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "math_common.hh"
#include "math_special_values.hh"
#include <hip/hip_cooperative_groups.h>
namespace cg = cooperative_groups;
#define MATH_TERNARY_KERNEL_DEF(func_name) \
template <typename T> \
__global__ void func_name##_kernel(T* const ys, const size_t num_xs, T* const x1s, T* const x2s, \
T* const x3s) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
if constexpr (std::is_same_v<float, T>) { \
ys[i] = func_name##f(x1s[i], x2s[i], x3s[i]); \
} else if constexpr (std::is_same_v<double, T>) { \
ys[i] = func_name(x1s[i], x2s[i], x3s[i]); \
} \
} \
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void TernaryFloatingPointBruteForceTest(kernel_sig<T, TArg, TArg, TArg> kernel,
ref_sig<RT, RTArg, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder,
const TArg a = std::numeric_limits<TArg>::lowest(),
const TArg b = std::numeric_limits<TArg>::max()) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const uint64_t num_iterations = GetTestIterationCount();
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(TArg) * 3 + sizeof(T)), num_iterations);
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
LinearAllocGuard<TArg> x3s{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(TArg)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &x1s, &x2s, &x3s] {
const auto generator = [=] {
static thread_local std::mt19937 rng(std::random_device{}());
if constexpr (std::is_same_v<TArg, Float16>) {
std::uniform_real_distribution<RefType_t<Float16>> unif_dist(-FLOAT16_MAX, FLOAT16_MAX);
return static_cast<Float16>(unif_dist(rng));
} else {
std::uniform_real_distribution<RefType_t<TArg>> unif_dist(a, b);
return static_cast<TArg>(unif_dist(rng));
}
};
std::generate(x1s.ptr() + base_idx, x1s.ptr() + base_idx + sub_batch_size, generator);
std::generate(x2s.ptr() + base_idx, x2s.ptr() + base_idx + sub_batch_size, generator);
std::generate(x3s.ptr() + base_idx, x3s.ptr() + base_idx + sub_batch_size, generator);
});
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, x1s.ptr(),
x2s.ptr(), x3s.ptr());
}
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void TernaryFloatingPointSpecialValuesTest(kernel_sig<T, TArg, TArg, TArg> kernel,
ref_sig<RT, RTArg, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
using SpecialValsType = std::conditional_t<std::is_same_v<TArg, Float16>, float, TArg>;
const auto values = std::get<SpecialVals<SpecialValsType>>(kSpecialValRegistry);
const auto size = values.size * values.size * values.size;
LinearAllocGuard<TArg> x1s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
LinearAllocGuard<TArg> x2s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
LinearAllocGuard<TArg> x3s{LinearAllocs::hipHostMalloc, size * sizeof(TArg)};
for (auto i = 0u; i < values.size; ++i) {
for (auto j = 0u; j < values.size; ++j) {
for (auto k = 0u; k < values.size; ++k) {
x1s.ptr()[(i * values.size + j) * values.size + k] = values.data[i];
x2s.ptr()[(i * values.size + j) * values.size + k] = values.data[j];
x3s.ptr()[(i * values.size + j) * values.size + k] = values.data[k];
}
}
}
MathTest math_test(kernel, size);
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, size, x1s.ptr(),
x2s.ptr(), x3s.ptr());
}
template <typename T, typename TArg, typename RT, typename RTArg, typename ValidatorBuilder>
void TernaryFloatingPointTest(kernel_sig<T, TArg, TArg, TArg> kernel,
ref_sig<RT, RTArg, RTArg, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
SECTION("Special values") {
TernaryFloatingPointSpecialValuesTest(kernel, ref_func, validator_builder);
}
SECTION("Brute force") {
TernaryFloatingPointBruteForceTest(kernel, ref_func, validator_builder);
}
}
#define MATH_TERNARY_WITHIN_ULP_TEST_DEF(kern_name, ref_func, sp_ulp, dp_ulp) \
MATH_TERNARY_KERNEL_DEF(kern_name) \
\
TEMPLATE_TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive", "", float, double) { \
using RT = RefType_t<TestType>; \
RT (*ref)(RT, RT, RT) = ref_func; \
const auto ulp = std::is_same_v<float, TestType> ? sp_ulp : dp_ulp; \
\
TernaryFloatingPointTest(kern_name##_kernel<TestType>, ref, \
ULPValidatorBuilderFactory<TestType>(ulp)); \
}
+64
Wyświetl plik
@@ -0,0 +1,64 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include <atomic>
#include <thread>
#include <boost/asio/post.hpp>
#include <boost/asio/thread_pool.hpp>
// This is a simple wrapper around boost::asio::thread_pool that keeps track of the number of
// currently active tasks using an atomic counter.
class ThreadPool {
public:
ThreadPool(size_t thread_count = std::thread::hardware_concurrency())
: thread_count_(thread_count) {}
~ThreadPool() { thread_pool_.join(); }
// Submits a task to the thread pool and increments the number of active tasks. The task is
// wrapped in a lambda that decrements the number of active tasks upon completion.
template <typename T> void Post(T&& task) {
++active_tasks_;
auto&& task_wrapper = [task, this] {
task();
--active_tasks_;
};
boost::asio::post(thread_pool_, task_wrapper);
}
// Busy waits for the number of active tasks to reach zero.
void Wait() const {
while (active_tasks_.load(std::memory_order_relaxed))
;
}
size_t thread_count() const { return thread_count_; }
private:
const size_t thread_count_;
boost::asio::thread_pool thread_pool_{thread_count_};
std::atomic<size_t> active_tasks_;
};
inline ThreadPool thread_pool{};
@@ -0,0 +1,108 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define TRIG_DP_UNARY_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); } \
__global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); }
/*Expecting 2 errors per macro invocation - 26 total*/
TRIG_DP_UNARY_NEGATIVE_KERNELS(sin)
TRIG_DP_UNARY_NEGATIVE_KERNELS(cos)
TRIG_DP_UNARY_NEGATIVE_KERNELS(tan)
TRIG_DP_UNARY_NEGATIVE_KERNELS(asin)
TRIG_DP_UNARY_NEGATIVE_KERNELS(acos)
TRIG_DP_UNARY_NEGATIVE_KERNELS(atan)
TRIG_DP_UNARY_NEGATIVE_KERNELS(sinh)
TRIG_DP_UNARY_NEGATIVE_KERNELS(cosh)
TRIG_DP_UNARY_NEGATIVE_KERNELS(tanh)
TRIG_DP_UNARY_NEGATIVE_KERNELS(asinh)
TRIG_DP_UNARY_NEGATIVE_KERNELS(atanh)
TRIG_DP_UNARY_NEGATIVE_KERNELS(sinpi)
TRIG_DP_UNARY_NEGATIVE_KERNELS(cospi)
/*Expecting 4 errors*/
__global__ void atan2_kernel_v1(double* x, double y) { double result = atan2(x, y); }
__global__ void atan2_kernel_v2(double x, double* y) { double result = atan2(x, y); }
__global__ void atan2_kernel_v3(Dummy x, double y) { double result = atan2(x, y); }
__global__ void atan2_kernel_v4(double x, Dummy y) { double result = atan2(x, y); }
/*Expecting 18 errors*/
__global__ void sincos_kernel_v1(double* x, double* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v2(Dummy x, double* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v3(double x, char* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v4(double x, short* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v5(double x, int* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v6(double x, long* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v7(double x, long long* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v8(double x, float* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v9(double x, Dummy* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v10(double x, const double* sptr, double* cptr) {
sincos(x, sptr, cptr);
}
__global__ void sincos_kernel_v11(double x, double* sptr, char* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v12(double x, double* sptr, short* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v13(double x, double* sptr, int* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v14(double x, double* sptr, long* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v15(double x, double* sptr, long long* cptr) {
sincos(x, sptr, cptr);
}
__global__ void sincos_kernel_v16(double x, double* sptr, float* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v17(double x, double* sptr, Dummy* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v18(double x, double* sptr, const double* cptr) {
sincos(x, sptr, cptr);
}
/*Expecting 18 errors*/
__global__ void sincospi_kernel_v1(float* x, float* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v2(Dummy x, float* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v3(float x, char* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v4(float x, short* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v5(float x, int* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v6(float x, long* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v7(float x, long long* sptr, float* cptr) {
sincospi(x, sptr, cptr);
}
__global__ void sincospi_kernel_v8(float x, double* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v9(float x, Dummy* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v10(float x, const float* sptr, float* cptr) {
sincospi(x, sptr, cptr);
}
__global__ void sincospi_kernel_v11(float x, float* sptr, char* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v12(float x, float* sptr, short* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v13(float x, float* sptr, int* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v14(float x, float* sptr, long* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v15(float x, float* sptr, long long* cptr) {
sincospi(x, sptr, cptr);
}
__global__ void sincospi_kernel_v16(float x, float* sptr, double* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v17(float x, float* sptr, Dummy* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v18(float x, float* sptr, const float* cptr) {
sincospi(x, sptr, cptr);
}
+137
Wyświetl plik
@@ -0,0 +1,137 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include "trig_negative_kernels_rtc.hh"
#include "unary_common.hh"
#include "binary_common.hh"
#include <boost/math/special_functions.hpp>
MATH_UNARY_WITHIN_ULP_TEST_DEF(sin, std::sin, 2, 2);
TEST_CASE("Unit_Device_sin_sinf_Negative_RTC") { NegativeTestRTCWrapper<4>(kSin); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(cos, std::cos, 2, 2)
TEST_CASE("Unit_Device_cos_cosf_Negative_RTC") { NegativeTestRTCWrapper<4>(kCos); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(tan, std::tan, 4, 2)
TEST_CASE("Unit_Device_tan_tanf_Negative_RTC") { NegativeTestRTCWrapper<4>(kTan); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(asin, std::asin, 2, 2)
TEST_CASE("Unit_Device_asin_asinf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAsin); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(acos, std::acos, 2, 2)
TEST_CASE("Unit_Device_acos_acosf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAcos); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(atan, std::atan, 2, 2)
TEST_CASE("Unit_Device_atan_atanf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAtan); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(sinh, std::sinh, 3, 2)
TEST_CASE("Unit_Device_sinh_sinhf_Negative_RTC") { NegativeTestRTCWrapper<4>(kSinh); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(cosh, std::cosh, 2, 1)
TEST_CASE("Unit_Device_cosh_coshf_Negative_RTC") { NegativeTestRTCWrapper<4>(kCosh); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(tanh, std::tanh, 2, 1)
TEST_CASE("Unit_Device_tanh_tanhf_Negative_RTC") { NegativeTestRTCWrapper<4>(kTanh); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(asinh, std::asinh, 3, 2)
TEST_CASE("Unit_Device_asinh_asinhf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAsinh); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(acosh, std::acosh, 4, 2)
TEST_CASE("Unit_Device_acosh_acoshf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAcosh); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(atanh, std::atanh, 3, 2)
TEST_CASE("Unit_Device_atanh_atanhf_Negative_RTC") { NegativeTestRTCWrapper<4>(kAtanh); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(sinpi, boost::math::sin_pi, 2, 2);
TEST_CASE("Unit_Device_sinpi_sinpif_Negative_RTC") { NegativeTestRTCWrapper<4>(kSinpi); }
MATH_UNARY_WITHIN_ULP_TEST_DEF(cospi, boost::math::cos_pi, 2, 2);
TEST_CASE("Unit_Device_cospi_cospif_Negative_RTC") { NegativeTestRTCWrapper<4>(kCospi); }
MATH_BINARY_WITHIN_ULP_TEST_DEF(atan2, std::atan2, 3, 2);
TEST_CASE("Unit_Device_atan2_atan2f_Negative_RTC") { NegativeTestRTCWrapper<8>(kAtan2); }
template <typename T>
__global__ void sincos_kernel(std::pair<T, T>* const ys, const size_t num_xs, T* const xs) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
if constexpr (std::is_same_v<float, T>) {
sincosf(xs[i], &ys[i].first, &ys[i].second);
} else if constexpr (std::is_same_v<double, T>) {
sincos(xs[i], &ys[i].first, &ys[i].second);
}
}
}
template <typename T> std::pair<T, T> sincos(T x) { return {std::sin(x), std::cos(x)}; }
TEST_CASE("Unit_Device_sincos_Accuracy_Positive - float") {
UnarySinglePrecisionTest(
sincos_kernel<float>, sincos<double>,
PairValidatorBuilderFactory<float>(ULPValidatorBuilderFactory<float>(2)));
}
TEST_CASE("Unit_Device_sincos_Accuracy_Positive - double") {
const auto validator_builder =
PairValidatorBuilderFactory<double>(ULPValidatorBuilderFactory<double>(2));
UnaryDoublePrecisionTest(sincos_kernel<double>, sincos<long double>, validator_builder);
}
TEST_CASE("Unit_Device_sincos_sincosf_Negative_RTC") { NegativeTestRTCWrapper<36>(kSincos); }
template <typename T>
__global__ void sincospi_kernel(std::pair<T, T>* const ys, const size_t num_xs, T* const xs) {
const auto tid = cg::this_grid().thread_rank();
const auto stride = cg::this_grid().size();
for (auto i = tid; i < num_xs; i += stride) {
if constexpr (std::is_same_v<float, T>) {
sincospif(xs[i], &ys[i].first, &ys[i].second);
} else if constexpr (std::is_same_v<double, T>) {
sincospi(xs[i], &ys[i].first, &ys[i].second);
}
}
}
template <typename T> std::pair<T, T> sincospi(T x) {
return {boost::math::sin_pi(x), boost::math::cos_pi(x)};
}
TEST_CASE("Unit_Device_sincospi_Accuracy_Positive - float") {
UnarySinglePrecisionTest(
sincospi_kernel<float>, sincospi<double>,
PairValidatorBuilderFactory<float>(ULPValidatorBuilderFactory<float>(2)));
}
TEST_CASE("Unit_Device_sincospi_Accuracy_Positive - double") {
const auto validator_builder =
PairValidatorBuilderFactory<double>(ULPValidatorBuilderFactory<double>(2));
UnaryDoublePrecisionTest(sincospi_kernel<double>, sincospi<long double>, validator_builder);
}
TEST_CASE("Unit_Device_sincospi_sincospif_Negative_RTC") { NegativeTestRTCWrapper<36>(kSincospi); }
@@ -0,0 +1,320 @@
// #define TRIG_UNARY_NEGATIVE_KERNELS(func_name)
// class Dummy {
// public:
// __device__ Dummy() {}
// __device__ ~Dummy() {}
// };
// __global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); }
// __global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
// __global__ void func_name##_kernel_v1(double* x) { double result = func_name(x); }
// __global__ void func_name##_kernel_v2(Dummy x) { double result = func_name(x); }
static constexpr auto kSin{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void sinf_kernel_v1(float* x) { float result = sinf(x); }
__global__ void sinf_kernel_v2(Dummy x) { float result = sinf(x); }
__global__ void sin_kernel_v1(double* x) { double result = sin(x); }
__global__ void sin_kernel_v2(Dummy x) { double result = sin(x); }
)"};
static constexpr auto kCos{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void cosf_kernel_v1(float* x) { float result = cosf(x); }
__global__ void cosf_kernel_v2(Dummy x) { float result = cosf(x); }
__global__ void cos_kernel_v1(double* x) { double result = cos(x); }
__global__ void cos_kernel_v2(Dummy x) { double result = cos(x); }
)"};
static constexpr auto kTan{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void tanf_kernel_v1(float* x) { float result = tanf(x); }
__global__ void tanf_kernel_v2(Dummy x) { float result = tanf(x); }
__global__ void tan_kernel_v1(double* x) { double result = tan(x); }
__global__ void tan_kernel_v2(Dummy x) { double result = tan(x); }
)"};
static constexpr auto kAsin{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void asinf_kernel_v1(float* x) { float result = asinf(x); }
__global__ void asinf_kernel_v2(Dummy x) { float result = asinf(x); }
__global__ void asin_kernel_v1(double* x) { double result = asin(x); }
__global__ void asin_kernel_v2(Dummy x) { double result = asin(x); }
)"};
static constexpr auto kAcos{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void acosf_kernel_v1(float* x) { float result = acosf(x); }
__global__ void acosf_kernel_v2(Dummy x) { float result = acosf(x); }
__global__ void acos_kernel_v1(double* x) { double result = acos(x); }
__global__ void acos_kernel_v2(Dummy x) { double result = acos(x); }
)"};
static constexpr auto kAtan{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void atanf_kernel_v1(float* x) { float result = atanf(x); }
__global__ void atanf_kernel_v2(Dummy x) { float result = atanf(x); }
__global__ void atan_kernel_v1(double* x) { double result = atan(x); }
__global__ void atan_kernel_v2(Dummy x) { double result = atan(x); }
)"};
static constexpr auto kSinh{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void sinhf_kernel_v1(float* x) { float result = sinhf(x); }
__global__ void sinhf_kernel_v2(Dummy x) { float result = sinhf(x); }
__global__ void sinh_kernel_v1(double* x) { double result = sinh(x); }
__global__ void sinh_kernel_v2(Dummy x) { double result = sinh(x); }
)"};
static constexpr auto kCosh{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void coshf_kernel_v1(float* x) { float result = coshf(x); }
__global__ void coshf_kernel_v2(Dummy x) { float result = coshf(x); }
__global__ void cosh_kernel_v1(double* x) { double result = cosh(x); }
__global__ void cosh_kernel_v2(Dummy x) { double result = cosh(x); }
)"};
static constexpr auto kTanh{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void tanhf_kernel_v1(float* x) { float result = tanhf(x); }
__global__ void tanhf_kernel_v2(Dummy x) { float result = tanhf(x); }
__global__ void tanh_kernel_v1(double* x) { double result = tanh(x); }
__global__ void tanh_kernel_v2(Dummy x) { double result = tanh(x); }
)"};
static constexpr auto kAsinh{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void asinhf_kernel_v1(float* x) { float result = asinhf(x); }
__global__ void asinhf_kernel_v2(Dummy x) { float result = asinhf(x); }
__global__ void asinh_kernel_v1(double* x) { double result = asinh(x); }
__global__ void asinh_kernel_v2(Dummy x) { double result = asinh(x); }
)"};
static constexpr auto kAcosh{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void acoshf_kernel_v1(float* x) { float result = acoshf(x); }
__global__ void acoshf_kernel_v2(Dummy x) { float result = acoshf(x); }
__global__ void acosh_kernel_v1(double* x) { double result = acosh(x); }
__global__ void acosh_kernel_v2(Dummy x) { double result = acosh(x); }
)"};
static constexpr auto kAtanh{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void atanhf_kernel_v1(float* x) { float result = atanhf(x); }
__global__ void atanhf_kernel_v2(Dummy x) { float result = atanhf(x); }
__global__ void atanh_kernel_v1(double* x) { double result = atanh(x); }
__global__ void atanh_kernel_v2(Dummy x) { double result = atanh(x); }
)"};
static constexpr auto kSinpi{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void sinpif_kernel_v1(float* x) { float result = sinpif(x); }
__global__ void sinpif_kernel_v2(Dummy x) { float result = sinpif(x); }
__global__ void sinpi_kernel_v1(double* x) { double result = sinpi(x); }
__global__ void sinpi_kernel_v2(Dummy x) { double result = sinpi(x); }
)"};
static constexpr auto kCospi{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void cospif_kernel_v1(float* x) { float result = cospif(x); }
__global__ void cospif_kernel_v2(Dummy x) { float result = cospif(x); }
__global__ void cospi_kernel_v1(double* x) { double result = cospi(x); }
__global__ void cospi_kernel_v2(Dummy x) { double result = cospi(x); }
)"};
static constexpr auto kAtan2{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void atan2f_kernel_v1(float* x, float y) { float result = atan2f(x, y); }
__global__ void atan2f_kernel_v2(float x, float* y) { float result = atan2f(x, y); }
__global__ void atan2f_kernel_v3(Dummy x, float y) { float result = atan2f(x, y); }
__global__ void atan2f_kernel_v4(float x, Dummy y) { float result = atan2f(x, y); }
__global__ void atan2_kernel_v1(double* x, double y) { double result = atan2(x, y); }
__global__ void atan2_kernel_v2(double x, double* y) { double result = atan2(x, y); }
__global__ void atan2_kernel_v3(Dummy x, double y) { double result = atan2(x, y); }
__global__ void atan2_kernel_v4(double x, Dummy y) { double result = atan2(x, y); }
)"};
static constexpr auto kSincos{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void sincosf_kernel_v1(float* x, float* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v2(Dummy x, float* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v3(float x, char* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v4(float x, short* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v5(float x, int* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v6(float x, long* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v7(float x, long long* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v8(float x, double* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v9(float x, Dummy* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v10(float x, const float* sptr, float* cptr) {
sincosf(x, sptr, cptr);
}
__global__ void sincosf_kernel_v11(float x, float* sptr, char* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v12(float x, float* sptr, short* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v13(float x, float* sptr, int* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v14(float x, float* sptr, long* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v15(float x, float* sptr, long long* cptr) {
sincosf(x, sptr, cptr);
}
__global__ void sincosf_kernel_v16(float x, float* sptr, double* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v17(float x, float* sptr, Dummy* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v18(float x, float* sptr, const float* cptr) {
sincosf(x, sptr, cptr);
}
__global__ void sincos_kernel_v1(double* x, double* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v2(Dummy x, double* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v3(double x, char* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v4(double x, short* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v5(double x, int* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v6(double x, long* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v7(double x, long long* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v8(double x, float* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v9(double x, Dummy* sptr, double* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v10(double x, const double* sptr, double* cptr) {
sincos(x, sptr, cptr);
}
__global__ void sincos_kernel_v11(double x, double* sptr, char* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v12(double x, double* sptr, short* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v13(double x, double* sptr, int* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v14(double x, double* sptr, long* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v15(double x, double* sptr, long long* cptr) {
sincos(x, sptr, cptr);
}
__global__ void sincos_kernel_v16(double x, double* sptr, float* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v17(double x, double* sptr, Dummy* cptr) { sincos(x, sptr, cptr); }
__global__ void sincos_kernel_v18(double x, double* sptr, const double* cptr) {
sincos(x, sptr, cptr);
}
)"};
static constexpr auto kSincospi{R"(
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
__global__ void sincospif_kernel_v1(float* x, float* sptr, float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v2(Dummy x, float* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v3(float x, char* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v4(float x, short* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v5(float x, int* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v6(float x, long* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v7(float x, long long* sptr, float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v8(float x, double* sptr, float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v9(float x, Dummy* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v10(float x, const float* sptr, float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v11(float x, float* sptr, char* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v12(float x, float* sptr, short* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v13(float x, float* sptr, int* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v14(float x, float* sptr, long* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v15(float x, float* sptr, long long* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v16(float x, float* sptr, double* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v17(float x, float* sptr, Dummy* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v18(float x, float* sptr, const float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospi_kernel_v1(float* x, float* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v2(Dummy x, float* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v3(float x, char* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v4(float x, short* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v5(float x, int* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v6(float x, long* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v7(float x, long long* sptr, float* cptr) {
sincospi(x, sptr, cptr);
}
__global__ void sincospi_kernel_v8(float x, double* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v9(float x, Dummy* sptr, float* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v10(float x, const float* sptr, float* cptr) {
sincospi(x, sptr, cptr);
}
__global__ void sincospi_kernel_v11(float x, float* sptr, char* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v12(float x, float* sptr, short* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v13(float x, float* sptr, int* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v14(float x, float* sptr, long* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v15(float x, float* sptr, long long* cptr) {
sincospi(x, sptr, cptr);
}
__global__ void sincospi_kernel_v16(float x, float* sptr, double* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v17(float x, float* sptr, Dummy* cptr) { sincospi(x, sptr, cptr); }
__global__ void sincospi_kernel_v18(float x, float* sptr, const float* cptr) {
sincospi(x, sptr, cptr);
}
)"};
@@ -0,0 +1,118 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include <hip_test_common.hh>
class Dummy {
public:
__device__ Dummy() {}
__device__ ~Dummy() {}
};
#define TRIG_SP_UNARY_NEGATIVE_KERNELS(func_name) \
__global__ void func_name##f_kernel_v1(float* x) { float result = func_name##f(x); } \
__global__ void func_name##f_kernel_v2(Dummy x) { float result = func_name##f(x); }
/*Expecting 2 errors per macro invocation - 26 total*/
TRIG_SP_UNARY_NEGATIVE_KERNELS(sin)
TRIG_SP_UNARY_NEGATIVE_KERNELS(cos)
TRIG_SP_UNARY_NEGATIVE_KERNELS(tan)
TRIG_SP_UNARY_NEGATIVE_KERNELS(asin)
TRIG_SP_UNARY_NEGATIVE_KERNELS(acos)
TRIG_SP_UNARY_NEGATIVE_KERNELS(atan)
TRIG_SP_UNARY_NEGATIVE_KERNELS(sinh)
TRIG_SP_UNARY_NEGATIVE_KERNELS(cosh)
TRIG_SP_UNARY_NEGATIVE_KERNELS(tanh)
TRIG_SP_UNARY_NEGATIVE_KERNELS(asinh)
TRIG_SP_UNARY_NEGATIVE_KERNELS(atanh)
TRIG_SP_UNARY_NEGATIVE_KERNELS(sinpi)
TRIG_SP_UNARY_NEGATIVE_KERNELS(cospi)
/*Expecting 4 errors*/
__global__ void atan2f_kernel_v1(float* x, float y) { float result = atan2f(x, y); }
__global__ void atan2f_kernel_v2(float x, float* y) { float result = atan2f(x, y); }
__global__ void atan2f_kernel_v3(Dummy x, float y) { float result = atan2f(x, y); }
__global__ void atan2f_kernel_v4(float x, Dummy y) { float result = atan2f(x, y); }
/*Expecting 18 errors*/
__global__ void sincosf_kernel_v1(float* x, float* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v2(Dummy x, float* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v3(float x, char* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v4(float x, short* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v5(float x, int* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v6(float x, long* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v7(float x, long long* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v8(float x, double* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v9(float x, Dummy* sptr, float* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v10(float x, const float* sptr, float* cptr) {
sincosf(x, sptr, cptr);
}
__global__ void sincosf_kernel_v11(float x, float* sptr, char* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v12(float x, float* sptr, short* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v13(float x, float* sptr, int* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v14(float x, float* sptr, long* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v15(float x, float* sptr, long long* cptr) {
sincosf(x, sptr, cptr);
}
__global__ void sincosf_kernel_v16(float x, float* sptr, double* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v17(float x, float* sptr, Dummy* cptr) { sincosf(x, sptr, cptr); }
__global__ void sincosf_kernel_v18(float x, float* sptr, const float* cptr) {
sincosf(x, sptr, cptr);
}
/*Expecting 18 errors*/
__global__ void sincospif_kernel_v1(float* x, float* sptr, float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v2(Dummy x, float* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v3(float x, char* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v4(float x, short* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v5(float x, int* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v6(float x, long* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v7(float x, long long* sptr, float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v8(float x, double* sptr, float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v9(float x, Dummy* sptr, float* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v10(float x, const float* sptr, float* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v11(float x, float* sptr, char* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v12(float x, float* sptr, short* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v13(float x, float* sptr, int* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v14(float x, float* sptr, long* cptr) { sincospif(x, sptr, cptr); }
__global__ void sincospif_kernel_v15(float x, float* sptr, long long* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v16(float x, float* sptr, double* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v17(float x, float* sptr, Dummy* cptr) {
sincospif(x, sptr, cptr);
}
__global__ void sincospif_kernel_v18(float x, float* sptr, const float* cptr) {
sincospif(x, sptr, cptr);
}
+243
Wyświetl plik
@@ -0,0 +1,243 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include "math_common.hh"
#include "math_special_values.hh"
#include <hip/hip_cooperative_groups.h>
namespace cg = cooperative_groups;
#define MATH_UNARY_KERNEL_DEF(func_name) \
template <typename T, typename RT = T> \
__global__ void func_name##_kernel(RT* const ys, const size_t num_xs, T* const xs) { \
const auto tid = cg::this_grid().thread_rank(); \
const auto stride = cg::this_grid().size(); \
\
for (auto i = tid; i < num_xs; i += stride) { \
if constexpr (std::is_same_v<float, T>) { \
ys[i] = func_name##f(xs[i]); \
} else if constexpr (std::is_same_v<double, T>) { \
ys[i] = func_name(xs[i]); \
} \
} \
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void UnaryHalfPrecisionBruteForceTest(kernel_sig<T, Float16> kernel, ref_sig<RT, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
uint64_t stop = std::numeric_limits<uint16_t>::max() + 1ul;
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(Float16) + sizeof(T)), stop);
LinearAllocGuard<Float16> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(Float16)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t v = 0u; v < stop;) {
batch_size = std::min<uint64_t>(max_batch_size, stop - v);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &values] {
auto t = v;
uint16_t val;
for (auto j = 0u; j < sub_batch_size; ++j) {
val = static_cast<uint16_t>(t++);
values.ptr()[base_idx + j] = *reinterpret_cast<Float16*>(&val);
}
});
v += sub_batch_size;
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
}
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void UnarySinglePrecisionBruteForceTest(kernel_sig<T, float> kernel, ref_sig<RT, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
uint64_t stop = std::numeric_limits<uint32_t>::max() + 1ul;
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(float) + sizeof(T)), stop);
LinearAllocGuard<float> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(float)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t v = 0u; v < stop;) {
batch_size = std::min<uint64_t>(max_batch_size, stop - v);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &values] {
auto t = v;
uint32_t val;
for (auto j = 0u; j < sub_batch_size; ++j) {
val = static_cast<uint32_t>(t++);
values.ptr()[base_idx + j] = *reinterpret_cast<float*>(&val);
}
});
v += sub_batch_size;
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
}
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void UnarySinglePrecisionRangeTest(kernel_sig<T, float> kernel, ref_sig<RT, RTArg> ref_func,
const ValidatorBuilder& validator_builder, const float a,
const float b) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const auto max_batch_size = GetMaxAllowedDeviceMemoryUsage() / (sizeof(float) + sizeof(T));
LinearAllocGuard<float> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(float)};
MathTest math_test(kernel, max_batch_size);
size_t inserted = 0u;
for (float v = a; v != b; v = std::nextafter(v, b)) {
values.ptr()[inserted++] = v;
if (inserted < max_batch_size) continue;
math_test.Run(validator_builder, grid_size, block_size, ref_func, inserted, values.ptr());
inserted = 0u;
}
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void UnaryDoublePrecisionBruteForceTest(kernel_sig<T, double> kernel, ref_sig<RT, RTArg> ref_func,
const ValidatorBuilder& validator_builder,
const double a = std::numeric_limits<double>::lowest(),
const double b = std::numeric_limits<double>::max()) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const uint64_t num_iterations = GetTestIterationCount();
const auto max_batch_size =
std::min(GetMaxAllowedDeviceMemoryUsage() / (sizeof(double) + sizeof(T)), num_iterations);
LinearAllocGuard<double> values{LinearAllocs::hipHostMalloc, max_batch_size * sizeof(double)};
MathTest math_test(kernel, max_batch_size);
auto batch_size = max_batch_size;
const auto num_threads = thread_pool.thread_count();
for (uint64_t i = 0ul; i < num_iterations; i += batch_size) {
batch_size = std::min<uint64_t>(max_batch_size, num_iterations - i);
const auto min_sub_batch_size = batch_size / num_threads;
const auto tail = batch_size % num_threads;
auto base_idx = 0u;
for (auto i = 0u; i < num_threads; ++i) {
const auto sub_batch_size = min_sub_batch_size + (i < tail);
thread_pool.Post([=, &values] {
const auto generator = [=] {
static thread_local std::mt19937 rng(std::random_device{}());
std::uniform_real_distribution<long double> unif_dist(a, b);
return static_cast<double>(unif_dist(rng));
};
std::generate(values.ptr() + base_idx, values.ptr() + base_idx + sub_batch_size, generator);
});
base_idx += sub_batch_size;
}
thread_pool.Wait();
math_test.Run(validator_builder, grid_size, block_size, ref_func, batch_size, values.ptr());
}
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void UnaryDoublePrecisionSpecialValuesTest(kernel_sig<T, double> kernel,
ref_sig<RT, RTArg> ref_func,
const ValidatorBuilder& validator_builder) {
const auto [grid_size, block_size] = GetOccupancyMaxPotentialBlockSize(kernel);
const auto values = std::get<SpecialVals<double>>(kSpecialValRegistry);
MathTest math_test(kernel, values.size);
math_test.template Run<false>(validator_builder, grid_size, block_size, ref_func, values.size,
values.data);
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void UnaryHalfPrecisionTest(kernel_sig<T, Float16> kernel, ref_sig<RT, RTArg> ref,
const ValidatorBuilder& validator_builder) {
SECTION("Brute force") { UnaryHalfPrecisionBruteForceTest(kernel, ref, validator_builder); }
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void UnarySinglePrecisionTest(kernel_sig<T, float> kernel, ref_sig<RT, RTArg> ref,
const ValidatorBuilder& validator_builder) {
SECTION("Brute force") { UnarySinglePrecisionBruteForceTest(kernel, ref, validator_builder); }
}
template <typename T, typename RT, typename RTArg, typename ValidatorBuilder>
void UnaryDoublePrecisionTest(kernel_sig<T, double> kernel, ref_sig<RT, RTArg> ref,
const ValidatorBuilder& validator_builder) {
SECTION("Special values") {
UnaryDoublePrecisionSpecialValuesTest(kernel, ref, validator_builder);
}
SECTION("Brute force") { UnaryDoublePrecisionBruteForceTest(kernel, ref, validator_builder); }
}
#define MATH_UNARY_WITHIN_ULP_TEST_DEF(kern_name, ref_func, sp_ulp, dp_ulp) \
MATH_UNARY_KERNEL_DEF(kern_name) \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - float") { \
double (*ref)(double) = ref_func; \
UnarySinglePrecisionTest(kern_name##_kernel<float>, ref, \
ULPValidatorBuilderFactory<float>(sp_ulp)); \
} \
\
TEST_CASE("Unit_Device_" #kern_name "_Accuracy_Positive - double") { \
long double (*ref)(long double) = ref_func; \
UnaryDoublePrecisionTest(kern_name##_kernel<double>, ref, \
ULPValidatorBuilderFactory<double>(dp_ulp)); \
}
#define MATH_UNARY_WITHIN_ULP_STL_REF_TEST_DEF(func_name, sp_ulp, dp_ulp) \
MATH_UNARY_WITHIN_ULP_TEST_DEF(func_name, std::func_name, sp_ulp, dp_ulp)
+152
Wyświetl plik
@@ -0,0 +1,152 @@
/*
Copyright (c) 2023 Advanced Micro Devices, Inc. All rights reserved.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#pragma once
#include <catch.hpp>
// Define a new MatcherBase class with a public 'describe' member function because
// Catch::MatcherBase::describe is protected and thus can't be used via a pointer to
// Catch::MatcherBase.
template <typename T> class MatcherBase : public Catch::MatcherBase<T> {
public:
virtual std::string describe() const = 0;
virtual ~MatcherBase() = default;
};
template <typename T, typename Matcher> class ValidatorBase : public MatcherBase<T> {
public:
template <typename... Ts>
ValidatorBase(T target, Ts&&... args) : matcher_{std::forward<Ts>(args)...}, target_{target} {}
bool match(const T& val) const override {
if (std::isnan(target_)) {
return std::isnan(val);
}
return matcher_.match(val);
}
std::string describe() const override {
if (std::isnan(target_)) {
return "is not NaN";
}
return matcher_.describe();
}
private:
Matcher matcher_;
T target_;
bool nan = false;
};
template <typename T> auto ULPValidatorBuilderFactory(int64_t ulps) {
return [=](T target, auto&&...) {
return std::make_unique<ValidatorBase<T, Catch::Matchers::Floating::WithinUlpsMatcher>>(
target, Catch::WithinULP(target, ulps));
};
};
template <typename T> auto AbsValidatorBuilderFactory(double margin) {
return [=](T target, auto&&...) {
return std::make_unique<ValidatorBase<T, Catch::Matchers::Floating::WithinAbsMatcher>>(
target, Catch::WithinAbs(target, margin));
};
}
template <typename T> auto RelValidatorBuilderFactory(T margin) {
return [=](T target, auto&&...) {
return std::make_unique<ValidatorBase<T, Catch::Matchers::Floating::WithinRelMatcher>>(
target, Catch::WithinRel(target, margin));
};
}
template <typename T> class EqValidator : public MatcherBase<T> {
public:
EqValidator(T target) : target_{target} {}
bool match(const T& val) const override {
if (std::isnan(target_)) {
return std::isnan(val);
}
return target_ == val;
}
std::string describe() const override {
std::stringstream ss;
ss << " is not equal to " << target_;
return ss.str();
}
private:
T target_;
};
template <typename T> auto EqValidatorBuilderFactory() {
return [](T val, auto&&...) { return std::make_unique<EqValidator<T>>(val); };
}
template <typename T, typename U, typename VBF, typename VBS>
class PairValidator : public MatcherBase<std::pair<T, U>> {
public:
PairValidator(const std::pair<T, U>& target, const VBF& vbf, const VBS& vbs)
: first_matcher_{vbf(target.first)}, second_matcher_{vbs(target.second)} {}
bool match(const std::pair<T, U>& val) const override {
return first_matcher_->match(val.first) && second_matcher_->match(val.second);
}
std::string describe() const override {
return "<" + first_matcher_->describe() + ", " + second_matcher_->describe() + ">";
}
private:
decltype(std::declval<VBF>()(std::declval<T>())) first_matcher_;
decltype(std::declval<VBS>()(std::declval<U>())) second_matcher_;
};
template <typename T, typename ValidatorBuilder>
auto PairValidatorBuilderFactory(const ValidatorBuilder& vb) {
return [=](const std::pair<T, T>& t, auto&&...) {
return std::make_unique<PairValidator<T, T, ValidatorBuilder, ValidatorBuilder>>(t, vb, vb);
};
}
template <typename T, typename U, typename VBF, typename VBS>
auto PairValidatorBuilderFactory(const VBF& vbf, const VBS& vbs) {
return [=](const std::pair<T, U>& t, auto&&...) {
return std::make_unique<PairValidator<T, U, VBF, VBS>>(t, vbf, vbs);
};
}
template <typename T> class NopValidator : public MatcherBase<T> {
public:
bool match(const T&) const override { return true; }
std::string describe() const override { return ""; }
};
template <typename T> auto NopValidatorBuilderFactory() {
return [](auto&&...) { return std::make_unique<NopValidator<T>>(); };
}