Files
project_6/cccl_upstream/libcudacxx/include/cuda/__numeric/isclose.h
EngineX CI 56fd68e7dd [INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
2026-07-30 09:35:51 +00:00

337 lines
14 KiB
C++

//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___NUMERIC_ISCLOSE_H
#define _CUDA___NUMERIC_ISCLOSE_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__cmath/ceil_div.h>
#include <cuda/__cmath/mul_hi.h>
#include <cuda/__cmath/uabs.h>
#include <cuda/__complex/get_real_imag.h>
#include <cuda/__complex/traits.h>
#include <cuda/__type_traits/is_floating_point.h>
#include <cuda/__utility/in_range.h>
#include <cuda/std/__algorithm/max.h>
#include <cuda/std/__cmath/abs.h>
#include <cuda/std/__cmath/exponential_functions.h>
#include <cuda/std/__cmath/hypot.h>
#include <cuda/std/__cmath/isfinite.h>
#include <cuda/std/__cmath/min_max.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__limits/numeric_limits.h>
#include <cuda/std/__type_traits/conditional.h>
#include <cuda/std/__type_traits/is_extended_floating_point.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/std/__type_traits/is_unsigned.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__utility/cmp.h>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <typename _Tp>
using __isclose_compare_t _CCCL_NODEBUG_ALIAS = ::cuda::std::
conditional_t<(::cuda::std::__is_extended_floating_point_v<_Tp> && sizeof(_Tp) <= sizeof(float)), float, _Tp>;
// compute 10^-(digits10 / 2)
template <typename _Tp>
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL float __isclose_default_relative_tolerance() noexcept
{
constexpr auto __digits = ::cuda::ceil_div(::cuda::std::numeric_limits<_Tp>::max_digits10, 2);
auto __exp = 1.0f;
for (int __i = 0; __i < __digits; ++__i)
{
__exp *= 10.0f;
}
return 1.0f / __exp;
}
template <typename _Tp>
[[nodiscard]] _CCCL_API constexpr bool
__isclose_fp_impl(const _Tp __lhs, const _Tp __rhs, const float __rel_tol, const _Tp __abs_tol) noexcept
{
_CCCL_ASSERT(::cuda::in_range(__rel_tol, 0.0f, 1.0f),
"cuda::isclose: relative tolerance must be in the range [0.0, 1.0]");
_CCCL_ASSERT(::cuda::std::isfinite(__abs_tol) && __abs_tol >= _Tp{0},
"cuda::isclose: absolute tolerance must be finite and non-negative");
if (__lhs == __rhs)
{
return true;
}
if (!::cuda::std::isfinite(__lhs) || !::cuda::std::isfinite(__rhs))
{
return false;
}
const auto __diff = ::cuda::std::fabs(__lhs - __rhs);
const auto __lhs_abs = ::cuda::std::fabs(__lhs);
const auto __rhs_abs = ::cuda::std::fabs(__rhs);
const auto __rel_value = static_cast<_Tp>(__rel_tol * ::cuda::std::fmax(__lhs_abs, __rhs_abs));
return __diff <= ::cuda::std::fmax(__abs_tol, __rel_value);
}
template <typename _ComplexType, typename _AbsTol>
[[nodiscard]] _CCCL_HOST_DEVICE_API bool __isclose_complex_impl(
const _ComplexType& __lhs, const _ComplexType& __rhs, const float __rel_tol, const _AbsTol __abs_tol) noexcept
{
using __scalar_t _CCCL_NODEBUG_ALIAS = typename _ComplexType::value_type;
using __compare_t _CCCL_NODEBUG_ALIAS = __isclose_compare_t<__scalar_t>;
static_assert(::cuda::is_floating_point_v<__scalar_t>, "cuda::isclose: __scalar_t must be a floating point type");
#if _CCCL_HAS_FLOAT128()
// __float128 is not supported because cuda::std::hypot is not implemented for this type
static_assert(!::cuda::std::is_same_v<__scalar_t, __float128>, "cuda::isclose: __float128 is not supported");
#endif // _CCCL_HAS_FLOAT128()
_CCCL_ASSERT(::cuda::in_range(__rel_tol, 0.0f, 1.0f),
"cuda::isclose: relative tolerance must be in the range [0.0, 1.0]");
_CCCL_ASSERT(::cuda::std::isfinite(__abs_tol) && __abs_tol >= __scalar_t{0},
"cuda::isclose: absolute tolerance must be finite and non-negative");
const auto __lhs_real = static_cast<__compare_t>(::cuda::__get_real(__lhs));
const auto __lhs_imag = static_cast<__compare_t>(::cuda::__get_imag(__lhs));
const auto __rhs_real = static_cast<__compare_t>(::cuda::__get_real(__rhs));
const auto __rhs_imag = static_cast<__compare_t>(::cuda::__get_imag(__rhs));
const auto __abs = static_cast<__compare_t>(__abs_tol);
if (__lhs_real == __rhs_real && __lhs_imag == __rhs_imag)
{
return true;
}
if (!::cuda::std::isfinite(__lhs_real) || !::cuda::std::isfinite(__lhs_imag) || !::cuda::std::isfinite(__rhs_real)
|| !::cuda::std::isfinite(__rhs_imag))
{
return false;
}
const auto __diff = ::cuda::std::hypot(__lhs_real - __rhs_real, __lhs_imag - __rhs_imag);
const auto __lhs_abs = ::cuda::std::hypot(__lhs_real, __lhs_imag);
const auto __rhs_abs = ::cuda::std::hypot(__rhs_real, __rhs_imag);
const auto __rel_value = __rel_tol * ::cuda::std::fmax(__lhs_abs, __rhs_abs);
return __diff <= ::cuda::std::fmax(__abs, __rel_value);
}
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::make_unsigned_t<_Tp>
__safe_abs_diff(const _Tp __lhs, const _Tp __rhs) noexcept
{
using __unsigned_t _CCCL_NODEBUG_ALIAS = ::cuda::std::make_unsigned_t<_Tp>;
const auto __lhs_abs = ::cuda::uabs(__lhs);
const auto __rhs_abs = ::cuda::uabs(__rhs);
const auto __is_lhs_negative = ::cuda::std::cmp_less(__lhs, _Tp{0});
const auto __is_rhs_negative = ::cuda::std::cmp_less(__rhs, _Tp{0});
if (__is_lhs_negative != __is_rhs_negative)
{
return static_cast<__unsigned_t>(__lhs_abs + __rhs_abs);
}
return (__lhs_abs < __rhs_abs)
? static_cast<__unsigned_t>(__rhs_abs - __lhs_abs)
: static_cast<__unsigned_t>(__lhs_abs - __rhs_abs);
}
// Represents a non-negative float exactly as __mantissa_ / 2^__shift_.
struct __float_ratio
{
::cuda::std::uint32_t __mantissa_{};
int __shift_{};
_CCCL_HOST_DEVICE_API explicit __float_ratio(const float __value) noexcept
{
_CCCL_ASSERT(__value >= 0.0f, "cuda::__float_ratio: value must be non-negative");
constexpr int __digits = ::cuda::std::numeric_limits<float>::digits;
int __exponent = 0;
const auto __fraction = ::cuda::std::frexp(__value, &__exponent);
__mantissa_ = static_cast<::cuda::std::uint32_t>(::cuda::std::ldexp(__fraction, __digits));
__shift_ = __digits - __exponent;
}
template <typename _Unsigned>
[[nodiscard]] _CCCL_HOST_DEVICE_API _Unsigned operator*(const _Unsigned __value) const noexcept
{
static_assert(::cuda::std::is_unsigned_v<_Unsigned>, "cuda::__float_ratio::operator* requires an unsigned type");
// The result is floor(__value * __mantissa_ / 2^__shift_).
constexpr int __digits = ::cuda::std::numeric_limits<_Unsigned>::digits;
constexpr int __float_digits = ::cuda::std::numeric_limits<float>::digits;
constexpr auto __power_of_two_mant = ::cuda::std::uint32_t{1} << (__float_digits - 1);
static_assert(__digits >= __float_digits, "__float_ratio requires an unsigned integer at least as wide as float");
// A zero mantissa represents zero. If the shift is at least the width of the double-width product, all bits are
// shifted out and the result rounds down to zero.
if (__mantissa_ == 0 || __shift_ >= 2 * __digits)
{
return _Unsigned{0};
}
// if the floating-point value is a power-of-two frexp normalizes an exact power of two, we can simplify the code
if (__mantissa_ == __power_of_two_mant)
{
const auto __pow2_shift = __shift_ - (__float_digits - 1);
return (__pow2_shift >= __digits) ? _Unsigned{0} : __value >> __pow2_shift;
}
const auto __mantissa = static_cast<_Unsigned>(__mantissa_);
const auto __low = static_cast<_Unsigned>(__value * __mantissa);
const auto __high = ::cuda::mul_hi(__value, __mantissa);
// product = (__high << __digits) | __low
// then product >> shift
if (__shift_ < __digits)
{
return (__high << (__digits - __shift_)) | (__low >> __shift_);
}
return __high >> (__shift_ - __digits);
}
};
template <typename _Tp>
[[nodiscard]] _CCCL_HOST_DEVICE_API bool
__isclose_integer_impl(const _Tp __lhs, const _Tp __rhs, const float __rel_tol, const _Tp __abs_tol) noexcept
{
_CCCL_ASSERT(::cuda::in_range(__rel_tol, 0.0f, 1.0f),
"cuda::isclose: relative tolerance must be in the range [0.0, 1.0]");
_CCCL_ASSERT(::cuda::std::cmp_greater_equal(__abs_tol, _Tp{0}),
"cuda::isclose: absolute tolerance must be non-negative");
using __unsigned_t _CCCL_NODEBUG_ALIAS = ::cuda::std::make_unsigned_t<_Tp>;
const auto __lhs_abs = ::cuda::uabs(__lhs);
const auto __rhs_abs = ::cuda::uabs(__rhs);
const auto __diff = ::cuda::__safe_abs_diff(__lhs, __rhs);
const auto __abs = static_cast<__unsigned_t>(__abs_tol);
const auto __max_abs = ::cuda::std::max(__lhs_abs, __rhs_abs);
const auto __rel_value = ::cuda::__float_ratio{__rel_tol} * __max_abs;
return __diff <= ::cuda::std::max(__abs, __rel_value);
}
//----------------------------------------------------------------------------------------------------------------------
// Public API
// Scalar overloads
//! @brief Checks whether two arithmetic values are close to each other using a relative and absolute tolerance.
//!
//! @param __lhs The first value to compare.
//! @param __rhs The second value to compare.
//! @param __rel_tol The relative tolerance.
//! @param __abs_tol The absolute tolerance.
//! @return True if __lhs and __rhs are close to each other, false otherwise.
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> || ::cuda::is_floating_point_v<_Tp>)
[[nodiscard]] _CCCL_HOST_DEVICE_API bool
isclose(const _Tp __lhs, const _Tp __rhs, const float __rel_tol, const _Tp __abs_tol) noexcept
{
if constexpr (::cuda::std::__cccl_is_integer_v<_Tp>)
{
return ::cuda::__isclose_integer_impl(+__lhs, +__rhs, __rel_tol, +__abs_tol);
}
else
{
using __value_t _CCCL_NODEBUG_ALIAS = __isclose_compare_t<_Tp>;
return ::cuda::__isclose_fp_impl(
static_cast<__value_t>(__lhs), static_cast<__value_t>(__rhs), __rel_tol, static_cast<__value_t>(__abs_tol));
}
}
//! @brief Checks whether two arithmetic values are close to each other using a relative tolerance.
//!
//! @param __lhs The first value to compare.
//! @param __rhs The second value to compare.
//! @param __rel_tol The relative tolerance.
//! @return True if __lhs and __rhs are close to each other, false otherwise.
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> || ::cuda::is_floating_point_v<_Tp>)
[[nodiscard]] _CCCL_HOST_DEVICE_API bool isclose(const _Tp __lhs, const _Tp __rhs, const float __rel_tol) noexcept
{
return ::cuda::isclose(__lhs, __rhs, __rel_tol, _Tp{0});
}
//! @brief Checks whether two arithmetic values are close to each other using the default relative tolerance.
//!
//! @param __lhs The first value to compare.
//! @param __rhs The second value to compare.
//! @return True if __lhs and __rhs are close to each other, false otherwise.
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> || ::cuda::is_floating_point_v<_Tp>)
[[nodiscard]] _CCCL_HOST_DEVICE_API bool isclose(const _Tp __lhs, const _Tp __rhs) noexcept
{
if constexpr (::cuda::std::__cccl_is_integer_v<_Tp>)
{
return __lhs == __rhs;
}
else
{
constexpr auto __rel_tol = ::cuda::__isclose_default_relative_tolerance<_Tp>();
return ::cuda::isclose(__lhs, __rhs, __rel_tol, _Tp{0});
}
}
// Complex overloads
template <typename _Tp, typename _AbsTol, bool = __is_any_complex_v<_Tp>>
inline constexpr bool __isclose_complex_comparison_v = false;
template <typename _Tp, typename _AbsTol>
inline constexpr bool __isclose_complex_comparison_v<_Tp, _AbsTol, true> =
::cuda::std::is_same_v<typename _Tp::value_type, _AbsTol>;
//! @brief Checks whether two complex values are close to each other using a relative and absolute tolerance.
//!
//! @param __lhs The first value to compare.
//! @param __rhs The second value to compare.
//! @param __rel_tol The relative tolerance.
//! @param __abs_tol The absolute tolerance.
//! @return True if __lhs and __rhs are close to each other, false otherwise.
_CCCL_TEMPLATE(typename _ComplexType, typename _AbsTol)
_CCCL_REQUIRES(__isclose_complex_comparison_v<_ComplexType, _AbsTol>)
[[nodiscard]] _CCCL_HOST_DEVICE_API bool
isclose(const _ComplexType& __lhs, const _ComplexType& __rhs, const float __rel_tol, const _AbsTol __abs_tol) noexcept
{
return ::cuda::__isclose_complex_impl(__lhs, __rhs, __rel_tol, __abs_tol);
}
//! @brief Checks whether two complex values are close to each other using a relative tolerance.
//!
//! @param __lhs The first value to compare.
//! @param __rhs The second value to compare.
//! @param __rel_tol The relative tolerance.
//! @return True if __lhs and __rhs are close to each other, false otherwise.
_CCCL_TEMPLATE(typename _ComplexType)
_CCCL_REQUIRES(__is_any_complex_v<_ComplexType>)
[[nodiscard]] _CCCL_HOST_DEVICE_API bool
isclose(const _ComplexType& __lhs, const _ComplexType& __rhs, const float __rel_tol) noexcept
{
using __scalar_t _CCCL_NODEBUG_ALIAS = typename _ComplexType::value_type;
return ::cuda::isclose(__lhs, __rhs, __rel_tol, __scalar_t{0});
}
//! @brief Checks whether two complex values are close to each other using the default relative tolerance.
//!
//! @param __lhs The first value to compare.
//! @param __rhs The second value to compare.
//! @return True if __lhs and __rhs are close to each other, false otherwise.
_CCCL_TEMPLATE(typename _ComplexType)
_CCCL_REQUIRES(__is_any_complex_v<_ComplexType>)
[[nodiscard]] _CCCL_HOST_DEVICE_API bool isclose(const _ComplexType& __lhs, const _ComplexType& __rhs) noexcept
{
using __scalar_t _CCCL_NODEBUG_ALIAS = typename _ComplexType::value_type;
return ::cuda::isclose(__lhs, __rhs, ::cuda::__isclose_default_relative_tolerance<__scalar_t>(), __scalar_t{0});
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___NUMERIC_ISCLOSE_H