[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
123
cccl_upstream/libcudacxx/include/cuda/__cmath/ceil_div.h
Normal file
123
cccl_upstream/libcudacxx/include/cuda/__cmath/ceil_div.h
Normal file
@@ -0,0 +1,123 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_CEIL_DIV_H
|
||||
#define _CUDA___CMATH_CEIL_DIV_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__algorithm/min.h>
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__type_traits/common_type.h>
|
||||
#include <cuda/std/__type_traits/is_enum.h>
|
||||
#include <cuda/std/__type_traits/is_integral.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
#include <cuda/std/__type_traits/underlying_type.h>
|
||||
#include <cuda/std/__utility/to_underlying.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief Divides two numbers \p __a and \p __b, rounding up if there is a remainder
|
||||
//! @param __a The dividend
|
||||
//! @param __b The divisor
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, _Up> ceil_div(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__b > _Up{0}, "cuda::ceil_div: 'b' must be positive");
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
_CCCL_ASSERT(__a >= _Tp{0}, "cuda::ceil_div: 'a' must be non negative");
|
||||
}
|
||||
using _Common = ::cuda::std::common_type_t<_Tp, _Up>;
|
||||
using _Prom = decltype(_Tp{} / _Up{});
|
||||
using _UProm = ::cuda::std::make_unsigned_t<_Prom>;
|
||||
auto __a1 = static_cast<_UProm>(__a);
|
||||
auto __b1 = static_cast<_UProm>(__b);
|
||||
if constexpr (::cuda::std::is_signed_v<_Prom>)
|
||||
{
|
||||
return static_cast<_Common>((__a1 + __b1 - 1) / __b1);
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_IF_CONSTEVAL_DEFAULT
|
||||
{
|
||||
const auto __res = __a1 / __b1;
|
||||
return static_cast<_Common>(__res + (__res * __b1 != __a1));
|
||||
}
|
||||
else
|
||||
{
|
||||
// the ::min method is faster even if __b is a compile-time constant
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE,
|
||||
(return static_cast<_Common>(::cuda::std::min(__a1, 1 + ((__a1 - 1) / __b1)));),
|
||||
(const auto __res = __a1 / __b1; //
|
||||
return static_cast<_Common>(__res + (__res * __b1 != __a1));))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
//! @brief Divides two numbers \p __a and \p __b, rounding up if there is a remainder, \p __b is an enum
|
||||
//! @param __a The dividend
|
||||
//! @param __b The divisor
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, ::cuda::std::underlying_type_t<_Up>>
|
||||
ceil_div(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::ceil_div(__a, ::cuda::std::to_underlying(__b));
|
||||
}
|
||||
|
||||
//! @brief Divides two numbers \p __a and \p __b, rounding up if there is a remainder, \p __b is an enum
|
||||
//! @param __a The dividend
|
||||
//! @param __b The divisor
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, _Up>
|
||||
ceil_div(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::ceil_div(::cuda::std::to_underlying(__a), __b);
|
||||
}
|
||||
|
||||
//! @brief Divides two numbers \p __a and \p __b, rounding up if there is a remainder, \p __b is an enum
|
||||
//! @param __a The dividend
|
||||
//! @param __b The divisor
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
|
||||
[[nodiscard]]
|
||||
_CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, ::cuda::std::underlying_type_t<_Up>>
|
||||
ceil_div(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::ceil_div(::cuda::std::to_underlying(__a), ::cuda::std::to_underlying(__b));
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_CEIL_DIV_H
|
||||
@@ -0,0 +1,237 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_FAST_MODULO_DIVISION_H
|
||||
#define _CUDA___CMATH_FAST_MODULO_DIVISION_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__cmath/ilog.h>
|
||||
#include <cuda/__cmath/mul_hi.h>
|
||||
#include <cuda/__cmath/pow2.h>
|
||||
#include <cuda/std/__limits/numeric_limits.h>
|
||||
#include <cuda/std/__type_traits/common_type.h>
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
#include <cuda/std/__type_traits/num_bits.h>
|
||||
#include <cuda/std/__utility/cmp.h>
|
||||
#include <cuda/std/__utility/pair.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
/***********************************************************************************************************************
|
||||
* Fast Modulo/Division based on Precomputation
|
||||
**********************************************************************************************************************/
|
||||
|
||||
// The implementation is based on the following references depending on the data type:
|
||||
// - Hacker's Delight, Second Edition, Chapter 10
|
||||
// - Labor of Division (Episode III): Faster Unsigned Division by Constants (libdivide)
|
||||
// https://ridiculousfish.com/blog/posts/labor-of-division-episode-iii.html
|
||||
// - Classic Round-Up Variant of Fast Unsigned Division by Constants
|
||||
// https://arxiv.org/pdf/2412.03680
|
||||
|
||||
//! @brief Fast modulo and division by precomputation
|
||||
//! @tparam _Tp The integer type of the divisor
|
||||
//! @tparam _DivisorIsNeverOne If \c true, the divisor is guaranteed to never be one, enabling optimizations
|
||||
template <typename _Tp, bool _DivisorIsNeverOne = false>
|
||||
class fast_mod_div
|
||||
{
|
||||
static_assert(::cuda::std::__cccl_is_integer_v<_Tp>, "cuda::fast_mod_div: T is required to be an integer type");
|
||||
|
||||
using __unsigned_t = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
|
||||
public:
|
||||
fast_mod_div() = delete;
|
||||
|
||||
//! @brief Constructs a fast_mod_div object from a divisor value
|
||||
//! @param[in] __divisor1 The divisor value, must be positive
|
||||
//! @pre \p __divisor1 must be positive
|
||||
_CCCL_API explicit fast_mod_div(_Tp __divisor1) noexcept
|
||||
: __divisor{__divisor1}
|
||||
{
|
||||
constexpr int __num_bits = ::cuda::std::__num_bits_v<_Tp>;
|
||||
_CCCL_ASSERT(__divisor > 0, "divisor must be positive");
|
||||
_CCCL_ASSERT(!_DivisorIsNeverOne || __divisor1 != 1, "cuda::fast_mod_div: divisor must not be one");
|
||||
const auto __u_divisor = static_cast<__unsigned_t>(__divisor);
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
__shift = ::cuda::ceil_ilog2(__divisor) - 1; // is_pow2(x) ? log2(x) : ceil(log2(x))
|
||||
const auto __k = __num_bits + __shift; // k: [N, 2*N-2]
|
||||
// __multiplier: ceil(2^k / divisor)
|
||||
// computed as 2^k / divisor + (remainder != 0)
|
||||
const auto __pow2_div = __divmod_pow2(__k, __u_divisor);
|
||||
__multiplier = __pow2_div.first + (__pow2_div.second != 0);
|
||||
}
|
||||
else
|
||||
{
|
||||
__shift = ::cuda::ilog2(__divisor); // floor(log2(divisor))
|
||||
if (::cuda::is_power_of_two(__divisor))
|
||||
{
|
||||
__multiplier = 0;
|
||||
return;
|
||||
}
|
||||
const auto __k = __num_bits + __shift;
|
||||
const auto __pow2_div = __divmod_pow2(__k, __u_divisor);
|
||||
// __multiplier: (2^k + 2^shift) / divisor
|
||||
// computed as 2^k / divisor + (2^k % divisor + 2^shift) / divisor
|
||||
// we know 0 < (divisor - 2^shift) < 2^shift
|
||||
// so __multiplier is 2^k / divisor + (2^k % divisor) >= (divisor - 2^shift)
|
||||
// where (divisor - 2^shift) is the threshold
|
||||
const auto __threshold = __u_divisor - (__unsigned_t{1} << __shift);
|
||||
__multiplier = __pow2_div.first + (__pow2_div.second >= __threshold);
|
||||
__add = (__pow2_div.second < __threshold);
|
||||
}
|
||||
}
|
||||
|
||||
//! @brief Divides the dividend by the precomputed divisor
|
||||
//! @param[in] __dividend The dividend value, must be non-negative
|
||||
//! @param[in] __divisor1 The precomputed divisor
|
||||
//! @return The quotient of the division
|
||||
template <typename _Lhs>
|
||||
[[nodiscard]] _CCCL_API friend ::cuda::std::common_type_t<_Tp, _Lhs>
|
||||
operator/(_Lhs __dividend, fast_mod_div __divisor1) noexcept
|
||||
{
|
||||
using ::cuda::std::is_same_v;
|
||||
using ::cuda::std::is_signed_v;
|
||||
using ::cuda::std::is_unsigned_v;
|
||||
static_assert(::cuda::std::__cccl_is_integer_v<_Lhs>, "cuda::fast_mod_div: T is required to be an integer type");
|
||||
static_assert(
|
||||
::cuda::std::cmp_less_equal(::cuda::std::numeric_limits<_Lhs>::max(), ::cuda::std::numeric_limits<_Tp>::max()),
|
||||
"cuda::fast_mod_div: dividend type must be less than or equal to divisor type");
|
||||
if constexpr (is_signed_v<_Lhs>)
|
||||
{
|
||||
_CCCL_ASSERT(__dividend >= 0, "dividend must be non-negative");
|
||||
}
|
||||
using __common_t = ::cuda::std::common_type_t<_Tp, _Lhs>;
|
||||
using __ucommon_t = ::cuda::std::make_unsigned_t<__common_t>;
|
||||
using __unsigned_lhs_t = ::cuda::std::make_unsigned_t<_Lhs>;
|
||||
const auto __div = __divisor1.__divisor; // cannot use structure binding because of clang-14
|
||||
const auto __mul = __divisor1.__multiplier;
|
||||
const auto __shift_ = __divisor1.__shift;
|
||||
auto __udividend = static_cast<__unsigned_lhs_t>(__dividend);
|
||||
if constexpr (is_unsigned_v<_Tp>)
|
||||
{
|
||||
if (__mul == 0) // divisor is a power of two
|
||||
{
|
||||
return static_cast<__common_t>(static_cast<__ucommon_t>(__udividend) >> __shift_);
|
||||
}
|
||||
// if dividend is a signed type, overflow is not possible
|
||||
if (is_signed_v<_Lhs> || __udividend != ::cuda::std::numeric_limits<__unsigned_lhs_t>::max()) // avoid overflow
|
||||
{
|
||||
__udividend += static_cast<__unsigned_lhs_t>(__divisor1.__add);
|
||||
}
|
||||
}
|
||||
else if (!_DivisorIsNeverOne && __div == 1)
|
||||
{
|
||||
return static_cast<__common_t>(__dividend);
|
||||
}
|
||||
const auto __higher_bits = ::cuda::mul_hi(static_cast<__ucommon_t>(__udividend), static_cast<__ucommon_t>(__mul));
|
||||
const auto __quotient = static_cast<__common_t>(__higher_bits >> __shift_);
|
||||
_CCCL_ASSERT(__quotient == static_cast<__common_t>(__dividend / __div), "wrong __quotient");
|
||||
return __quotient;
|
||||
}
|
||||
|
||||
//! @brief Computes the remainder of dividing the dividend by the precomputed divisor
|
||||
//! @param[in] __dividend The dividend value
|
||||
//! @param[in] __divisor1 The precomputed divisor
|
||||
//! @return The remainder of the division
|
||||
template <typename _Lhs>
|
||||
[[nodiscard]] _CCCL_API friend ::cuda::std::common_type_t<_Tp, _Lhs>
|
||||
operator%(_Lhs __dividend, fast_mod_div __divisor1) noexcept
|
||||
{
|
||||
return __dividend - (__dividend / __divisor1) * __divisor1.__divisor;
|
||||
}
|
||||
|
||||
//! @brief Converts to the underlying divisor value
|
||||
//! @return The divisor value
|
||||
[[nodiscard]] _CCCL_API operator _Tp() const noexcept
|
||||
{
|
||||
return static_cast<_Tp>(__divisor);
|
||||
}
|
||||
|
||||
private:
|
||||
//! @brief Computes {2^power / divisor, 2^power % divisor}
|
||||
//!
|
||||
//! @param[in] __power The exponent, in the range [0, 2*N) where N is the bit-width of \c __unsigned_t
|
||||
//! @param[in] __divisor The divisor, must be positive
|
||||
//! @return A pair of (quotient, remainder)
|
||||
[[nodiscard]] _CCCL_API static ::cuda::std::pair<__unsigned_t, __unsigned_t>
|
||||
__divmod_pow2(int __power, __unsigned_t __divisor) noexcept
|
||||
{
|
||||
constexpr int __num_bits = ::cuda::std::__num_bits_v<__unsigned_t>;
|
||||
_CCCL_ASSERT(__power >= 0, "power must be non-negative");
|
||||
_CCCL_ASSERT(__power < 2 * __num_bits, "power must be less than 2 * N");
|
||||
_CCCL_ASSERT(__divisor > 0, "divisor must be positive");
|
||||
__unsigned_t __quotient = 0;
|
||||
__unsigned_t __remainder = 0;
|
||||
// Algorithm: restoring binary division
|
||||
// Reference: https://marz.utk.edu/my-courses/cosc130/lectures/binary-arithmetic/
|
||||
// The dividend 2^power may exceed the range of __unsigned_t. The algorithm processes the dividend bit-by-bit from
|
||||
// MSB to LSB without materializing it.
|
||||
// Since 2^power has exactly one set bit, the loop contributes to 1 only when the current bit index equals __power,
|
||||
// and 0 otherwise.
|
||||
// At each iteration, the remainder is shifted left, the next dividend bit is appended.
|
||||
// If the result is >= __divisor, the divisor is subtracted and 1-bit is recorded in the quotient.
|
||||
for (int __bit = __power; __bit >= 0; --__bit)
|
||||
{
|
||||
// __carry_flag only matters for unsigned types with large divisors (MSB set: >= 2^(N-1))
|
||||
const bool __carry_flag = (__remainder >> (__num_bits - 1)) != 0; // __remainder / 2^N != 0
|
||||
__remainder <<= 1;
|
||||
__remainder |= (__bit == __power); // append the remainder bit
|
||||
const bool __quotient_bit = __carry_flag || (__remainder >= __divisor);
|
||||
__quotient <<= 1; // shift
|
||||
__quotient |= unsigned{__quotient_bit}; // append the quotient bit
|
||||
if (__quotient_bit)
|
||||
{
|
||||
__remainder -= __divisor;
|
||||
}
|
||||
}
|
||||
return {__quotient, __remainder};
|
||||
}
|
||||
|
||||
_Tp __divisor = 1;
|
||||
__unsigned_t __multiplier = 0;
|
||||
unsigned __add = 0;
|
||||
int __shift = 0;
|
||||
};
|
||||
|
||||
/***********************************************************************************************************************
|
||||
* Non-member functions
|
||||
**********************************************************************************************************************/
|
||||
|
||||
//! @brief Computes both quotient and remainder of dividing the dividend by the precomputed divisor
|
||||
//! @param[in] __dividend The dividend value
|
||||
//! @param[in] __divisor The precomputed divisor
|
||||
//! @return A pair of (quotient, remainder)
|
||||
template <typename _Tp, typename _Lhs, bool _DivisorIsNeverOne>
|
||||
[[nodiscard]] _CCCL_API ::cuda::std::pair<_Tp, _Lhs>
|
||||
div(_Tp __dividend, fast_mod_div<_Lhs, _DivisorIsNeverOne> __divisor) noexcept
|
||||
{
|
||||
const auto __quotient = __dividend / __divisor;
|
||||
const auto __remainder = __dividend - __quotient * __divisor;
|
||||
return {__quotient, __remainder};
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_FAST_MODULO_DIVISION_H
|
||||
203
cccl_upstream/libcudacxx/include/cuda/__cmath/ilog.h
Normal file
203
cccl_upstream/libcudacxx/include/cuda/__cmath/ilog.h
Normal file
@@ -0,0 +1,203 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_ILOG_H
|
||||
#define _CUDA___CMATH_ILOG_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__bit/has_single_bit.h>
|
||||
#include <cuda/std/__bit/integral.h>
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__limits/numeric_limits.h>
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/is_same.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
_CCCL_TEMPLATE(typename _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr int ilog2(const _Tp __t) noexcept
|
||||
{
|
||||
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
_CCCL_ASSERT(__t > 0, "ilog2() argument must be strictly positive");
|
||||
auto __log2_approx = ::cuda::std::__bit_log2(static_cast<_Up>(__t));
|
||||
_CCCL_ASSUME(__log2_approx <= ::cuda::std::numeric_limits<_Tp>::digits);
|
||||
return __log2_approx;
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(typename _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr int ceil_ilog2(const _Tp __t) noexcept
|
||||
{
|
||||
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
return ::cuda::ilog2(__t) + !::cuda::std::has_single_bit(static_cast<_Up>(__t));
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL ::cuda::std::array<::cuda::std::uint32_t, 10> __power_of_10_32bit() noexcept
|
||||
{
|
||||
return {10,
|
||||
100,
|
||||
1'000,
|
||||
10'000,
|
||||
100'000,
|
||||
1'000'000,
|
||||
10'000'000,
|
||||
100'000'000,
|
||||
1'000'000'000,
|
||||
::cuda::std::numeric_limits<::cuda::std::uint32_t>::max()};
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL ::cuda::std::array<::cuda::std::uint64_t, 20> __power_of_10_64bit() noexcept
|
||||
{
|
||||
return {
|
||||
10,
|
||||
100,
|
||||
1'000,
|
||||
10'000,
|
||||
100'000,
|
||||
1'000'000,
|
||||
10'000'000,
|
||||
100'000'000,
|
||||
1'000'000'000,
|
||||
10'000'000'000,
|
||||
100'000'000'000,
|
||||
1'000'000'000'000,
|
||||
10'000'000'000'000,
|
||||
100'000'000'000'000,
|
||||
1'000'000'000'000'000,
|
||||
10'000'000'000'000'000,
|
||||
100'000'000'000'000'000,
|
||||
1'000'000'000'000'000'000,
|
||||
10'000'000'000'000'000'000ull,
|
||||
::cuda::std::numeric_limits<::cuda::std::uint64_t>::max()};
|
||||
}
|
||||
|
||||
#if _CCCL_HAS_INT128()
|
||||
|
||||
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL ::cuda::std::array<__uint128_t, 39> __power_of_10_128bit() noexcept
|
||||
{
|
||||
return {
|
||||
10,
|
||||
100,
|
||||
1'000,
|
||||
10'000,
|
||||
100'000,
|
||||
1'000'000,
|
||||
10'000'000,
|
||||
100'000'000,
|
||||
1'000'000'000,
|
||||
10'000'000'000,
|
||||
100'000'000'000,
|
||||
1'000'000'000'000,
|
||||
10'000'000'000'000,
|
||||
100'000'000'000'000,
|
||||
1'000'000'000'000'000,
|
||||
10'000'000'000'000'000,
|
||||
100'000'000'000'000'000,
|
||||
1'000'000'000'000'000'000,
|
||||
10'000'000'000'000'000'000ull,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 10,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 100,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 1'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 10'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 100'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 10'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 100'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 10'000'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 100'000'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 10'000'000'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 100'000'000'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000'000'000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000'000'0000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 10'000'000'000'000'0000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 100'000'000'000'000'0000,
|
||||
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000'000'000'0000ull,
|
||||
::cuda::std::numeric_limits<__uint128_t>::max()};
|
||||
}
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
|
||||
_CCCL_TEMPLATE(typename _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr int ilog10(const _Tp __t) noexcept
|
||||
{
|
||||
using ::cuda::std::uint32_t;
|
||||
using ::cuda::std::uint64_t;
|
||||
_CCCL_ASSERT(__t > 0, "cuda::ilog10() argument must be strictly positive");
|
||||
constexpr auto __reciprocal_log2_10 = 0.301029995663f; // 1 / log2(10)
|
||||
const auto __log2 = ::cuda::ilog2(__t) * __reciprocal_log2_10;
|
||||
auto __log10_approx = static_cast<int>(__log2);
|
||||
if constexpr (sizeof(_Tp) <= sizeof(uint32_t))
|
||||
{
|
||||
_CCCL_ASSERT(__log10_approx < static_cast<int>(::cuda::__power_of_10_32bit().size()), "out of bounds");
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, uint32_t>)
|
||||
{
|
||||
// don't replace +1 with >= because wraparound behavior is needed here
|
||||
__log10_approx += static_cast<uint32_t>(__t) + 1 > ::cuda::__power_of_10_32bit()[__log10_approx];
|
||||
}
|
||||
else
|
||||
{
|
||||
__log10_approx += static_cast<uint32_t>(__t) >= ::cuda::__power_of_10_32bit()[__log10_approx];
|
||||
}
|
||||
}
|
||||
else if constexpr (sizeof(_Tp) == sizeof(uint64_t))
|
||||
{
|
||||
_CCCL_ASSERT(__log10_approx < static_cast<int>(::cuda::__power_of_10_64bit().size()), "out of bounds");
|
||||
// +1 is not needed here
|
||||
__log10_approx += static_cast<uint64_t>(__t) >= ::cuda::__power_of_10_64bit()[__log10_approx];
|
||||
}
|
||||
#if _CCCL_HAS_INT128()
|
||||
else
|
||||
{
|
||||
_CCCL_ASSERT(__log10_approx < static_cast<int>(::cuda::__power_of_10_128bit().size()), "out of bounds");
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, __uint128_t>)
|
||||
{
|
||||
// don't replace +1 with >= because wraparound behavior is needed here
|
||||
__log10_approx += static_cast<__uint128_t>(__t) + 1 > ::cuda::__power_of_10_128bit()[__log10_approx];
|
||||
}
|
||||
else
|
||||
{
|
||||
__log10_approx += static_cast<__uint128_t>(__t) >= ::cuda::__power_of_10_128bit()[__log10_approx];
|
||||
}
|
||||
}
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
_CCCL_ASSUME(__log10_approx <= ::cuda::std::numeric_limits<_Tp>::digits / 3); // 2^X < 10^(x/3) -> 8^X < 10^x
|
||||
return __log10_approx;
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(typename _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr int ceil_ilog10(const _Tp __t) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__t > 0, "cuda::ceil_ilog10() argument must be strictly positive");
|
||||
return __t == 1 ? 0 : ::cuda::ilog10(static_cast<_Tp>(__t - 1)) + 1;
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_ILOG_H
|
||||
111
cccl_upstream/libcudacxx/include/cuda/__cmath/ipow.h
Normal file
111
cccl_upstream/libcudacxx/include/cuda/__cmath/ipow.h
Normal file
@@ -0,0 +1,111 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_IPOW_H
|
||||
#define _CUDA___CMATH_IPOW_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__cmath/ilog.h>
|
||||
#include <cuda/__cmath/neg.h>
|
||||
#include <cuda/__cmath/pow2.h>
|
||||
#include <cuda/__cmath/uabs.h>
|
||||
#include <cuda/std/__bit/countl.h>
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/is_unsigned.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
#include <cuda/std/__utility/cmp.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
template <class _Tp, class _Ep>
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp __cccl_ipow_impl_base_pow2(_Tp __b, _Ep __e) noexcept
|
||||
{
|
||||
const auto __shift = static_cast<int>(__e - 1) * ::cuda::ilog2(__b);
|
||||
const auto __lz = ::cuda::std::countl_zero(__b);
|
||||
return (__shift >= __lz) ? _Tp{0} : (_Tp{__b} << __shift);
|
||||
}
|
||||
|
||||
template <class _Tp, class _Ep>
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp __cccl_ipow_impl(_Tp __b, _Ep __e) noexcept
|
||||
{
|
||||
static_assert(::cuda::std::is_unsigned_v<_Tp>);
|
||||
|
||||
if (::cuda::is_power_of_two(__b))
|
||||
{
|
||||
return ::cuda::__cccl_ipow_impl_base_pow2(__b, __e);
|
||||
}
|
||||
|
||||
auto __x = __b;
|
||||
auto __y = _Tp{1};
|
||||
|
||||
while (__e > 1)
|
||||
{
|
||||
if (__e % 2 == 1)
|
||||
{
|
||||
__y *= __x;
|
||||
--__e;
|
||||
}
|
||||
__x *= __x;
|
||||
__e /= 2;
|
||||
}
|
||||
return __x * __y;
|
||||
}
|
||||
|
||||
//! @brief Computes the integer power of a base to an exponent.
|
||||
//! @param __b The base
|
||||
//! @param __e The exponent
|
||||
//! @pre \p __b must be an integer type
|
||||
//! @pre \p __e must be an integer type
|
||||
//! @return The result of raising \p __b to the power of \p __e
|
||||
//! @note The result is undefined if \p __b is 0 and \p __e is negative.
|
||||
_CCCL_TEMPLATE(class _Tp, class _Ep)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND ::cuda::std::__cccl_is_integer_v<_Ep>)
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp ipow(_Tp __b, _Ep __e) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__b != _Tp{0} || ::cuda::std::cmp_greater_equal(__e, _Ep{0}),
|
||||
"cuda::ipow() requires non-negative exponent for base 0");
|
||||
|
||||
if (__e == _Ep{0} || __b == _Tp{1})
|
||||
{
|
||||
return _Tp{1};
|
||||
}
|
||||
else if (::cuda::std::cmp_less(__e, _Ep{0}) || __b == _Tp{0})
|
||||
{
|
||||
return _Tp{0};
|
||||
}
|
||||
auto __res = ::cuda::__cccl_ipow_impl(::cuda::uabs(__b), ::cuda::std::__to_unsigned_like(__e));
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
if (__b < _Tp{0} && (__e % 2u == 1))
|
||||
{
|
||||
__res = cuda::neg(__res);
|
||||
}
|
||||
}
|
||||
return static_cast<_Tp>(__res);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_IPOW_H
|
||||
80
cccl_upstream/libcudacxx/include/cuda/__cmath/isqrt.h
Normal file
80
cccl_upstream/libcudacxx/include/cuda/__cmath/isqrt.h
Normal file
@@ -0,0 +1,80 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_ISQRT_H
|
||||
#define _CUDA___CMATH_ISQRT_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__bit/integral.h>
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief Returns the square root of the given non-negative integer rounded down
|
||||
//! @param __v The input number
|
||||
//! @pre \p __v must be an integer type
|
||||
//! @pre \p __v must be non-negative
|
||||
//! @return The square root of \p __v rounded down
|
||||
//! @warning If \p __v is negative, the behavior is undefined
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp isqrt(_Tp __v) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
_CCCL_ASSERT(__v >= _Tp{0}, "cuda::isqrt requires non-negative input");
|
||||
}
|
||||
|
||||
if (__v <= 1)
|
||||
{
|
||||
return __v;
|
||||
}
|
||||
|
||||
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
|
||||
_Up __uv = static_cast<_Up>(__v);
|
||||
_Up __ret{};
|
||||
_Up __bit = static_cast<_Up>(_Up{1} << ((::cuda::std::bit_width(__uv) - 1) & (~1)));
|
||||
|
||||
while (__bit != 0)
|
||||
{
|
||||
if (__uv >= __ret + __bit)
|
||||
{
|
||||
__uv -= __ret + __bit;
|
||||
__ret = (__ret >> 1) + __bit;
|
||||
}
|
||||
else
|
||||
{
|
||||
__ret >>= 1;
|
||||
}
|
||||
__bit >>= 2;
|
||||
}
|
||||
return static_cast<_Tp>(__ret);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_ISQRT_H
|
||||
147
cccl_upstream/libcudacxx/include/cuda/__cmath/mul_hi.h
Normal file
147
cccl_upstream/libcudacxx/include/cuda/__cmath/mul_hi.h
Normal file
@@ -0,0 +1,147 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_MUL_HI_H
|
||||
#define _CUDA___CMATH_MUL_HI_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/make_nbit_int.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
#include <cuda/std/__type_traits/num_bits.h>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#if _CCCL_COMPILER(MSVC)
|
||||
# include <intrin.h>
|
||||
#endif // _CCCL_COMPILER(MSVC)
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
/***********************************************************************************************************************
|
||||
* Extract higher bits after multiplication
|
||||
**********************************************************************************************************************/
|
||||
|
||||
template <typename _Tp>
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp __mul_hi_fallback(_Tp __lhs, _Tp __rhs) noexcept
|
||||
{
|
||||
static_assert(::cuda::std::is_unsigned_v<_Tp>, "__mul_hi_fallback: T is required to be a unsigned integer type");
|
||||
constexpr int __half_bits = ::cuda::std::__num_bits_v<_Tp> / 2;
|
||||
using __half_bits_t = ::cuda::std::__make_nbit_uint_t<__half_bits>;
|
||||
const auto __lhs_low = static_cast<__half_bits_t>(__lhs); // 32-bit
|
||||
const auto __lhs_high = static_cast<__half_bits_t>(__lhs >> __half_bits); // 32-bit
|
||||
const auto __rhs_low = static_cast<__half_bits_t>(__rhs); // 32-bit
|
||||
const auto __rhs_high = static_cast<__half_bits_t>(__rhs >> __half_bits); // 32-bit
|
||||
const auto __po_half = (static_cast<_Tp>(__lhs_low) * __rhs_low) >> __half_bits;
|
||||
const auto __p1 = static_cast<_Tp>(__lhs_low) * __rhs_high; // 64-bit
|
||||
const auto __p2 = static_cast<_Tp>(__lhs_high) * __rhs_low; // 64-bit
|
||||
const auto __p3 = static_cast<_Tp>(__lhs_high) * __rhs_high; // 64-bit
|
||||
const auto __p1_half = static_cast<__half_bits_t>(__p1); // 32-bit
|
||||
const auto __p2_half = static_cast<__half_bits_t>(__p2); // 32-bit
|
||||
const auto __carry = (__po_half + __p1_half + __p2_half) >> __half_bits; // 64-bit
|
||||
return __p3 + (__p1 >> __half_bits) + (__p2 >> __half_bits) + __carry;
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(typename _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
|
||||
[[nodiscard]]
|
||||
_CCCL_API constexpr _Tp mul_hi(_Tp __lhs, _Tp __rhs) noexcept
|
||||
{
|
||||
using ::cuda::std::int64_t;
|
||||
using ::cuda::std::is_signed_v;
|
||||
#if !_CCCL_TILE_COMPILATION() // nvbug6085239 error: calling a __device__ function from a __tile__ function
|
||||
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
|
||||
{
|
||||
if constexpr (sizeof(_Tp) == sizeof(int))
|
||||
{
|
||||
if constexpr (is_signed_v<_Tp>)
|
||||
{
|
||||
[[maybe_unused]] const auto __lhs1 = static_cast<int>(__lhs);
|
||||
[[maybe_unused]] const auto __rhs1 = static_cast<int>(__rhs);
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (return ::__mulhi(__lhs1, __rhs1);));
|
||||
}
|
||||
else // is_unsigned_v<_Tp>
|
||||
{
|
||||
[[maybe_unused]] const auto __lhs1 = static_cast<unsigned>(__lhs);
|
||||
[[maybe_unused]] const auto __rhs1 = static_cast<unsigned>(__rhs);
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (return ::__umulhi(__lhs1, __rhs1);));
|
||||
}
|
||||
}
|
||||
else if constexpr (sizeof(_Tp) == sizeof(int64_t))
|
||||
{
|
||||
if constexpr (is_signed_v<_Tp>)
|
||||
{
|
||||
[[maybe_unused]] const auto __lhs1 = static_cast<long long>(__lhs);
|
||||
[[maybe_unused]] const auto __rhs1 = static_cast<long long>(__rhs);
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (return ::__mul64hi(__lhs1, __rhs1);));
|
||||
# if _CCCL_COMPILER(MSVC)
|
||||
NV_IF_TARGET(NV_IS_HOST, (return ::__mulh(__lhs1, __rhs1);));
|
||||
# endif // _CCCL_COMPILER(MSVC)
|
||||
}
|
||||
else // is_unsigned_v<_Tp>
|
||||
{
|
||||
[[maybe_unused]] const auto __lhs1 = static_cast<unsigned long long>(__lhs);
|
||||
[[maybe_unused]] const auto __rhs1 = static_cast<unsigned long long>(__rhs);
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (return ::__umul64hi(__lhs1, __rhs1);));
|
||||
# if _CCCL_COMPILER(MSVC)
|
||||
NV_IF_TARGET(NV_IS_HOST, (return ::__umulh(__lhs1, __rhs1);));
|
||||
# endif // _CCCL_COMPILER(MSVC)
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif // !_CCCL_TILE_COMPILATION()
|
||||
if constexpr (sizeof(_Tp) < sizeof(int64_t) || (sizeof(_Tp) == sizeof(int64_t) && _CCCL_HAS_INT128()))
|
||||
{
|
||||
constexpr auto __bits = ::cuda::std::__num_bits_v<_Tp>;
|
||||
using __larger_t = ::cuda::std::__make_nbit_int_t<__bits * 2, is_signed_v<_Tp>>;
|
||||
const auto __ret = (static_cast<__larger_t>(__lhs) * __rhs) >> __bits;
|
||||
return static_cast<_Tp>(__ret);
|
||||
}
|
||||
else // sizeof(_Tp) >= sizeof(int64_t) && !_CCCL_HAS_INT128()
|
||||
{
|
||||
if constexpr (is_signed_v<_Tp>)
|
||||
{
|
||||
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
const auto __lhs1 = static_cast<_Up>(__lhs);
|
||||
const auto __rhs1 = static_cast<_Up>(__rhs);
|
||||
auto __hi = ::cuda::__mul_hi_fallback(__lhs1, __rhs1);
|
||||
if (__lhs < 0)
|
||||
{
|
||||
__hi -= __rhs1;
|
||||
}
|
||||
if (__rhs < 0)
|
||||
{
|
||||
__hi -= __lhs1;
|
||||
}
|
||||
return static_cast<_Tp>(__hi);
|
||||
}
|
||||
else
|
||||
{
|
||||
return ::cuda::__mul_hi_fallback(__lhs, __rhs);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_MULTIPLY_HIGH_HALF_H
|
||||
47
cccl_upstream/libcudacxx/include/cuda/__cmath/neg.h
Normal file
47
cccl_upstream/libcudacxx/include/cuda/__cmath/neg.h
Normal file
@@ -0,0 +1,47 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_NEG_H
|
||||
#define _CUDA___CMATH_NEG_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief Returns the negative value of the input number
|
||||
//! @param __v The input number
|
||||
//! @return The signed negative value of \p __v
|
||||
//! @note This function doesn't cause undefined behavior when negating the minimum value of a signed integer type.
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp neg(_Tp __v) noexcept
|
||||
{
|
||||
return static_cast<_Tp>(~::cuda::std::__to_unsigned_like(__v) + 1);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_NEG_H
|
||||
74
cccl_upstream/libcudacxx/include/cuda/__cmath/pow2.h
Normal file
74
cccl_upstream/libcudacxx/include/cuda/__cmath/pow2.h
Normal file
@@ -0,0 +1,74 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_POW2_H
|
||||
#define _CUDA___CMATH_POW2_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__bit/has_single_bit.h>
|
||||
#include <cuda/std/__bit/integral.h>
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
_CCCL_TEMPLATE(typename _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr bool is_power_of_two(_Tp __t) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
_CCCL_ASSERT(__t >= _Tp{0}, "cuda::is_power_of_two requires non-negative input");
|
||||
}
|
||||
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
return ::cuda::std::has_single_bit(static_cast<_Up>(__t));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(typename _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp next_power_of_two(_Tp __t) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
_CCCL_ASSERT(__t >= _Tp{0}, "cuda::is_power_of_two requires non-negative input");
|
||||
}
|
||||
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
return ::cuda::std::bit_ceil(static_cast<_Up>(__t));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(typename _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp prev_power_of_two(_Tp __t) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
_CCCL_ASSERT(__t >= _Tp{0}, "cuda::is_power_of_two requires non-negative input");
|
||||
}
|
||||
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
return ::cuda::std::bit_floor(static_cast<_Up>(__t));
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_POW2_H
|
||||
102
cccl_upstream/libcudacxx/include/cuda/__cmath/round_down.h
Normal file
102
cccl_upstream/libcudacxx/include/cuda/__cmath/round_down.h
Normal file
@@ -0,0 +1,102 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_ROUND_DOWN_H
|
||||
#define _CUDA___CMATH_ROUND_DOWN_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__type_traits/common_type.h>
|
||||
#include <cuda/std/__type_traits/is_enum.h>
|
||||
#include <cuda/std/__type_traits/is_integral.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
#include <cuda/std/__utility/to_underlying.h>
|
||||
#include <cuda/std/limits>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief Round the number \p __a to the previous multiple of \p __b
|
||||
//! @param __a The input number
|
||||
//! @param __b The multiplicand
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, _Up> round_down(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__b > _Up{0}, "cuda::round_down: 'b' must be positive");
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
_CCCL_ASSERT(__a >= _Tp{0}, "cuda::round_down: 'a' must be non negative");
|
||||
}
|
||||
using _Common = ::cuda::std::common_type_t<_Tp, _Up>;
|
||||
using _Prom = decltype(_Tp{} / _Up{});
|
||||
using _UProm = ::cuda::std::make_unsigned_t<_Prom>;
|
||||
auto __c1 = static_cast<_UProm>(__a) / static_cast<_UProm>(__b);
|
||||
return static_cast<_Common>(__c1 * static_cast<_UProm>(__b));
|
||||
}
|
||||
|
||||
//! @brief Round the number \p __a to the previous multiple of \p __b
|
||||
//! @param __a The input number
|
||||
//! @param __b The multiplicand
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, ::cuda::std::underlying_type_t<_Up>>
|
||||
round_down(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::round_down(__a, ::cuda::std::to_underlying(__b));
|
||||
}
|
||||
|
||||
//! @brief Round the number \p __a to the previous multiple of \p __b
|
||||
//! @param __a The input number
|
||||
//! @param __b The multiplicand
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, _Up>
|
||||
round_down(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::round_down(::cuda::std::to_underlying(__a), __b);
|
||||
}
|
||||
|
||||
//! @brief Round the number \p __a to the previous multiple of \p __b
|
||||
//! @param __a The input number
|
||||
//! @param __b The multiplicand
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
|
||||
[[nodiscard]]
|
||||
_CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, ::cuda::std::underlying_type_t<_Up>>
|
||||
round_down(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::round_down(::cuda::std::to_underlying(__a), ::cuda::std::to_underlying(__b));
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_ROUND_DOWN_H
|
||||
104
cccl_upstream/libcudacxx/include/cuda/__cmath/round_up.h
Normal file
104
cccl_upstream/libcudacxx/include/cuda/__cmath/round_up.h
Normal file
@@ -0,0 +1,104 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_ROUND_UP_H
|
||||
#define _CUDA___CMATH_ROUND_UP_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__cmath/ceil_div.h>
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__type_traits/common_type.h>
|
||||
#include <cuda/std/__type_traits/is_enum.h>
|
||||
#include <cuda/std/__type_traits/is_integral.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
#include <cuda/std/__utility/to_underlying.h>
|
||||
#include <cuda/std/limits>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief Round the number \p __a to the next multiple of \p __b
|
||||
//! @param __a The input number
|
||||
//! @param __b The multiplicand
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, _Up> round_up(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__b > _Up{0}, "cuda::round_up: 'b' must be positive");
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
_CCCL_ASSERT(__a >= _Tp{0}, "cuda::round_up: 'a' must be non negative");
|
||||
}
|
||||
using _Common = ::cuda::std::common_type_t<_Tp, _Up>;
|
||||
using _Prom = decltype(_Tp{} / _Up{});
|
||||
auto __c = ::cuda::ceil_div(static_cast<_Prom>(__a), static_cast<_Prom>(__b));
|
||||
_CCCL_ASSERT(static_cast<_Common>(__c) <= ::cuda::std::numeric_limits<_Common>::max() / static_cast<_Common>(__b),
|
||||
"cuda::round_up: result overflow");
|
||||
return static_cast<_Common>(static_cast<_Prom>(__c) * static_cast<_Prom>(__b));
|
||||
}
|
||||
|
||||
//! @brief Round the number \p __a to the next multiple of \p __b
|
||||
//! @param __a The input number
|
||||
//! @param __b The multiplicand
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, ::cuda::std::underlying_type_t<_Up>>
|
||||
round_up(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::round_up(__a, ::cuda::std::to_underlying(__b));
|
||||
}
|
||||
|
||||
//! @brief Round the number \p __a to the next multiple of \p __b
|
||||
//! @param __a The input number
|
||||
//! @param __b The multiplicand
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, _Up>
|
||||
round_up(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::round_up(::cuda::std::to_underlying(__a), __b);
|
||||
}
|
||||
|
||||
//! @brief Round the number \p __a to the next multiple of \p __b
|
||||
//! @param __a The input number
|
||||
//! @param __b The multiplicand
|
||||
//! @pre \p __a must be non-negative
|
||||
//! @pre \p __b must be positive
|
||||
_CCCL_TEMPLATE(class _Tp, class _Up)
|
||||
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
|
||||
[[nodiscard]]
|
||||
_CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, ::cuda::std::underlying_type_t<_Up>>
|
||||
round_up(const _Tp __a, const _Up __b) noexcept
|
||||
{
|
||||
return ::cuda::round_up(::cuda::std::to_underlying(__a), ::cuda::std::to_underlying(__b));
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_ROUND_UP_H
|
||||
134
cccl_upstream/libcudacxx/include/cuda/__cmath/sincos.h
Normal file
134
cccl_upstream/libcudacxx/include/cuda/__cmath/sincos.h
Normal file
@@ -0,0 +1,134 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_SINCOS_H
|
||||
#define _CUDA___CMATH_SINCOS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__cmath/trigonometric_functions.h>
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__type_traits/conditional.h>
|
||||
#include <cuda/std/__type_traits/is_extended_arithmetic.h>
|
||||
#include <cuda/std/__type_traits/is_integral.h>
|
||||
#include <cuda/std/__type_traits/is_same.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
#if _CCCL_HAS_BUILTIN(__builtin_sincosf) || _CCCL_COMPILER(GCC)
|
||||
# define _CCCL_BUILTIN_SINCOSF(...) __builtin_sincosf(__VA_ARGS__)
|
||||
#endif // _CCCL_HAS_BUILTIN(__builtin_sincosf) || _CCCL_COMPILER(GCC)
|
||||
|
||||
#if _CCCL_HAS_BUILTIN(__builtin_sincos) || _CCCL_COMPILER(GCC)
|
||||
# define _CCCL_BUILTIN_SINCOS(...) __builtin_sincos(__VA_ARGS__)
|
||||
#endif // _CCCL_HAS_BUILTIN(__builtin_sincos) || _CCCL_COMPILER(GCC)
|
||||
|
||||
#if _CCCL_HAS_BUILTIN(__builtin_sincosl) || _CCCL_COMPILER(GCC)
|
||||
# define _CCCL_BUILTIN_SINCOSL(...) __builtin_sincosl(__VA_ARGS__)
|
||||
#endif // _CCCL_HAS_BUILTIN(__builtin_sincosl) || _CCCL_COMPILER(GCC)
|
||||
|
||||
// clang-cuda crashes if these builtins are used.
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
# undef _CCCL_BUILTIN_SINCOSF
|
||||
# undef _CCCL_BUILTIN_SINCOS
|
||||
# undef _CCCL_BUILTIN_SINCOSL
|
||||
#endif // _CCCL_CUDA_COMPILER(CLANG)
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief Type returned by \c cuda::sincos.
|
||||
template <class _Tp>
|
||||
struct _CCCL_TYPE_VISIBILITY_DEFAULT sincos_result
|
||||
{
|
||||
_Tp sin; //!< The sin result.
|
||||
_Tp cos; //!< The cos result.
|
||||
};
|
||||
|
||||
//! @brief Computes sin and cos operation of a value.
|
||||
//!
|
||||
//! @param __v The value.
|
||||
//!
|
||||
//! @return The \c cuda::sincos_result with the results of sin and cos operations.
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__is_extended_arithmetic_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API auto sincos(_Tp __v) noexcept
|
||||
-> sincos_result<::cuda::std::conditional_t<::cuda::std::is_integral_v<_Tp>, double, _Tp>>
|
||||
{
|
||||
if constexpr (::cuda::std::is_integral_v<_Tp>)
|
||||
{
|
||||
return ::cuda::sincos(static_cast<double>(__v));
|
||||
}
|
||||
else
|
||||
{
|
||||
[[maybe_unused]] sincos_result<_Tp> __ret{};
|
||||
#if defined(_CCCL_BUILTIN_SINCOSF)
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, float>)
|
||||
{
|
||||
_CCCL_BUILTIN_SINCOSF(__v, &__ret.sin, &__ret.cos);
|
||||
return __ret;
|
||||
}
|
||||
#endif // _CCCL_BUILTIN_SINCOSF
|
||||
#if defined(_CCCL_BUILTIN_SINCOS)
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, double>)
|
||||
{
|
||||
_CCCL_BUILTIN_SINCOS(__v, &__ret.sin, &__ret.cos);
|
||||
return __ret;
|
||||
}
|
||||
#endif // _CCCL_BUILTIN_SINCOS
|
||||
#if _CCCL_HAS_LONG_DOUBLE() && defined(_CCCL_BUILTIN_SINCOSL)
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, long double>)
|
||||
{
|
||||
_CCCL_BUILTIN_SINCOSL(__v, &__ret.sin, &__ret.cos);
|
||||
return __ret;
|
||||
}
|
||||
#endif // _CCCL_HAS_LONG_DOUBLE() && _CCCL_BUILTIN_SINCOSL
|
||||
|
||||
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
|
||||
{
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, float>)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (::sincosf(__v, &__ret.sin, &__ret.cos); return __ret;))
|
||||
}
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, double>)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (::sincos(__v, &__ret.sin, &__ret.cos); return __ret;))
|
||||
}
|
||||
#if _LIBCUDACXX_HAS_NVFP16()
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, ::__half>)
|
||||
{
|
||||
const auto __result_float = ::cuda::sincos(::__half2float(__v));
|
||||
return {::__float2half(__result_float.sin), ::__float2half(__result_float.cos)};
|
||||
}
|
||||
#endif // _LIBCUDACXX_HAS_NVFP16()
|
||||
#if _LIBCUDACXX_HAS_NVBF16()
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, ::__nv_bfloat16>)
|
||||
{
|
||||
const auto __result_float = ::cuda::sincos(::__bfloat162float(__v));
|
||||
return {::__float2bfloat16(__result_float.sin), ::__float2bfloat16(__result_float.cos)};
|
||||
}
|
||||
#endif // _LIBCUDACXX_HAS_NVBF16()
|
||||
}
|
||||
return {::cuda::std::sin(__v), ::cuda::std::cos(__v)};
|
||||
}
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_SINCOS_H
|
||||
57
cccl_upstream/libcudacxx/include/cuda/__cmath/uabs.h
Normal file
57
cccl_upstream/libcudacxx/include/cuda/__cmath/uabs.h
Normal file
@@ -0,0 +1,57 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___CMATH_UABS_H
|
||||
#define _CUDA___CMATH_UABS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__cmath/neg.h>
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/make_unsigned.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief Returns the *unsigned* absolute value of the given number.
|
||||
//! @param __v The input number
|
||||
//! @pre \p __v must be an integer type
|
||||
//! @return The unsigned absolute value of \p __v
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::make_unsigned_t<_Tp> uabs(_Tp __v) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_signed_v<_Tp>)
|
||||
{
|
||||
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
|
||||
return (__v < _Tp(0)) ? static_cast<_Up>(::cuda::neg(__v)) : static_cast<_Up>(__v);
|
||||
}
|
||||
else
|
||||
{
|
||||
return __v;
|
||||
}
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___CMATH_UABS_H
|
||||
Reference in New Issue
Block a user