[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,123 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_CEIL_DIV_H
#define _CUDA___CMATH_CEIL_DIV_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__algorithm/min.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__type_traits/common_type.h>
#include <cuda/std/__type_traits/is_enum.h>
#include <cuda/std/__type_traits/is_integral.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__type_traits/underlying_type.h>
#include <cuda/std/__utility/to_underlying.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Divides two numbers \p __a and \p __b, rounding up if there is a remainder
//! @param __a The dividend
//! @param __b The divisor
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, _Up> ceil_div(const _Tp __a, const _Up __b) noexcept
{
_CCCL_ASSERT(__b > _Up{0}, "cuda::ceil_div: 'b' must be positive");
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
_CCCL_ASSERT(__a >= _Tp{0}, "cuda::ceil_div: 'a' must be non negative");
}
using _Common = ::cuda::std::common_type_t<_Tp, _Up>;
using _Prom = decltype(_Tp{} / _Up{});
using _UProm = ::cuda::std::make_unsigned_t<_Prom>;
auto __a1 = static_cast<_UProm>(__a);
auto __b1 = static_cast<_UProm>(__b);
if constexpr (::cuda::std::is_signed_v<_Prom>)
{
return static_cast<_Common>((__a1 + __b1 - 1) / __b1);
}
else
{
_CCCL_IF_CONSTEVAL_DEFAULT
{
const auto __res = __a1 / __b1;
return static_cast<_Common>(__res + (__res * __b1 != __a1));
}
else
{
// the ::min method is faster even if __b is a compile-time constant
NV_IF_ELSE_TARGET(NV_IS_DEVICE,
(return static_cast<_Common>(::cuda::std::min(__a1, 1 + ((__a1 - 1) / __b1)));),
(const auto __res = __a1 / __b1; //
return static_cast<_Common>(__res + (__res * __b1 != __a1));))
}
}
}
//! @brief Divides two numbers \p __a and \p __b, rounding up if there is a remainder, \p __b is an enum
//! @param __a The dividend
//! @param __b The divisor
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, ::cuda::std::underlying_type_t<_Up>>
ceil_div(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::ceil_div(__a, ::cuda::std::to_underlying(__b));
}
//! @brief Divides two numbers \p __a and \p __b, rounding up if there is a remainder, \p __b is an enum
//! @param __a The dividend
//! @param __b The divisor
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, _Up>
ceil_div(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::ceil_div(::cuda::std::to_underlying(__a), __b);
}
//! @brief Divides two numbers \p __a and \p __b, rounding up if there is a remainder, \p __b is an enum
//! @param __a The dividend
//! @param __b The divisor
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
[[nodiscard]]
_CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, ::cuda::std::underlying_type_t<_Up>>
ceil_div(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::ceil_div(::cuda::std::to_underlying(__a), ::cuda::std::to_underlying(__b));
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_CEIL_DIV_H

View File

@@ -0,0 +1,237 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_FAST_MODULO_DIVISION_H
#define _CUDA___CMATH_FAST_MODULO_DIVISION_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__cmath/ilog.h>
#include <cuda/__cmath/mul_hi.h>
#include <cuda/__cmath/pow2.h>
#include <cuda/std/__limits/numeric_limits.h>
#include <cuda/std/__type_traits/common_type.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__type_traits/num_bits.h>
#include <cuda/std/__utility/cmp.h>
#include <cuda/std/__utility/pair.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
/***********************************************************************************************************************
* Fast Modulo/Division based on Precomputation
**********************************************************************************************************************/
// The implementation is based on the following references depending on the data type:
// - Hacker's Delight, Second Edition, Chapter 10
// - Labor of Division (Episode III): Faster Unsigned Division by Constants (libdivide)
// https://ridiculousfish.com/blog/posts/labor-of-division-episode-iii.html
// - Classic Round-Up Variant of Fast Unsigned Division by Constants
// https://arxiv.org/pdf/2412.03680
//! @brief Fast modulo and division by precomputation
//! @tparam _Tp The integer type of the divisor
//! @tparam _DivisorIsNeverOne If \c true, the divisor is guaranteed to never be one, enabling optimizations
template <typename _Tp, bool _DivisorIsNeverOne = false>
class fast_mod_div
{
static_assert(::cuda::std::__cccl_is_integer_v<_Tp>, "cuda::fast_mod_div: T is required to be an integer type");
using __unsigned_t = ::cuda::std::make_unsigned_t<_Tp>;
public:
fast_mod_div() = delete;
//! @brief Constructs a fast_mod_div object from a divisor value
//! @param[in] __divisor1 The divisor value, must be positive
//! @pre \p __divisor1 must be positive
_CCCL_API explicit fast_mod_div(_Tp __divisor1) noexcept
: __divisor{__divisor1}
{
constexpr int __num_bits = ::cuda::std::__num_bits_v<_Tp>;
_CCCL_ASSERT(__divisor > 0, "divisor must be positive");
_CCCL_ASSERT(!_DivisorIsNeverOne || __divisor1 != 1, "cuda::fast_mod_div: divisor must not be one");
const auto __u_divisor = static_cast<__unsigned_t>(__divisor);
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
__shift = ::cuda::ceil_ilog2(__divisor) - 1; // is_pow2(x) ? log2(x) : ceil(log2(x))
const auto __k = __num_bits + __shift; // k: [N, 2*N-2]
// __multiplier: ceil(2^k / divisor)
// computed as 2^k / divisor + (remainder != 0)
const auto __pow2_div = __divmod_pow2(__k, __u_divisor);
__multiplier = __pow2_div.first + (__pow2_div.second != 0);
}
else
{
__shift = ::cuda::ilog2(__divisor); // floor(log2(divisor))
if (::cuda::is_power_of_two(__divisor))
{
__multiplier = 0;
return;
}
const auto __k = __num_bits + __shift;
const auto __pow2_div = __divmod_pow2(__k, __u_divisor);
// __multiplier: (2^k + 2^shift) / divisor
// computed as 2^k / divisor + (2^k % divisor + 2^shift) / divisor
// we know 0 < (divisor - 2^shift) < 2^shift
// so __multiplier is 2^k / divisor + (2^k % divisor) >= (divisor - 2^shift)
// where (divisor - 2^shift) is the threshold
const auto __threshold = __u_divisor - (__unsigned_t{1} << __shift);
__multiplier = __pow2_div.first + (__pow2_div.second >= __threshold);
__add = (__pow2_div.second < __threshold);
}
}
//! @brief Divides the dividend by the precomputed divisor
//! @param[in] __dividend The dividend value, must be non-negative
//! @param[in] __divisor1 The precomputed divisor
//! @return The quotient of the division
template <typename _Lhs>
[[nodiscard]] _CCCL_API friend ::cuda::std::common_type_t<_Tp, _Lhs>
operator/(_Lhs __dividend, fast_mod_div __divisor1) noexcept
{
using ::cuda::std::is_same_v;
using ::cuda::std::is_signed_v;
using ::cuda::std::is_unsigned_v;
static_assert(::cuda::std::__cccl_is_integer_v<_Lhs>, "cuda::fast_mod_div: T is required to be an integer type");
static_assert(
::cuda::std::cmp_less_equal(::cuda::std::numeric_limits<_Lhs>::max(), ::cuda::std::numeric_limits<_Tp>::max()),
"cuda::fast_mod_div: dividend type must be less than or equal to divisor type");
if constexpr (is_signed_v<_Lhs>)
{
_CCCL_ASSERT(__dividend >= 0, "dividend must be non-negative");
}
using __common_t = ::cuda::std::common_type_t<_Tp, _Lhs>;
using __ucommon_t = ::cuda::std::make_unsigned_t<__common_t>;
using __unsigned_lhs_t = ::cuda::std::make_unsigned_t<_Lhs>;
const auto __div = __divisor1.__divisor; // cannot use structure binding because of clang-14
const auto __mul = __divisor1.__multiplier;
const auto __shift_ = __divisor1.__shift;
auto __udividend = static_cast<__unsigned_lhs_t>(__dividend);
if constexpr (is_unsigned_v<_Tp>)
{
if (__mul == 0) // divisor is a power of two
{
return static_cast<__common_t>(static_cast<__ucommon_t>(__udividend) >> __shift_);
}
// if dividend is a signed type, overflow is not possible
if (is_signed_v<_Lhs> || __udividend != ::cuda::std::numeric_limits<__unsigned_lhs_t>::max()) // avoid overflow
{
__udividend += static_cast<__unsigned_lhs_t>(__divisor1.__add);
}
}
else if (!_DivisorIsNeverOne && __div == 1)
{
return static_cast<__common_t>(__dividend);
}
const auto __higher_bits = ::cuda::mul_hi(static_cast<__ucommon_t>(__udividend), static_cast<__ucommon_t>(__mul));
const auto __quotient = static_cast<__common_t>(__higher_bits >> __shift_);
_CCCL_ASSERT(__quotient == static_cast<__common_t>(__dividend / __div), "wrong __quotient");
return __quotient;
}
//! @brief Computes the remainder of dividing the dividend by the precomputed divisor
//! @param[in] __dividend The dividend value
//! @param[in] __divisor1 The precomputed divisor
//! @return The remainder of the division
template <typename _Lhs>
[[nodiscard]] _CCCL_API friend ::cuda::std::common_type_t<_Tp, _Lhs>
operator%(_Lhs __dividend, fast_mod_div __divisor1) noexcept
{
return __dividend - (__dividend / __divisor1) * __divisor1.__divisor;
}
//! @brief Converts to the underlying divisor value
//! @return The divisor value
[[nodiscard]] _CCCL_API operator _Tp() const noexcept
{
return static_cast<_Tp>(__divisor);
}
private:
//! @brief Computes {2^power / divisor, 2^power % divisor}
//!
//! @param[in] __power The exponent, in the range [0, 2*N) where N is the bit-width of \c __unsigned_t
//! @param[in] __divisor The divisor, must be positive
//! @return A pair of (quotient, remainder)
[[nodiscard]] _CCCL_API static ::cuda::std::pair<__unsigned_t, __unsigned_t>
__divmod_pow2(int __power, __unsigned_t __divisor) noexcept
{
constexpr int __num_bits = ::cuda::std::__num_bits_v<__unsigned_t>;
_CCCL_ASSERT(__power >= 0, "power must be non-negative");
_CCCL_ASSERT(__power < 2 * __num_bits, "power must be less than 2 * N");
_CCCL_ASSERT(__divisor > 0, "divisor must be positive");
__unsigned_t __quotient = 0;
__unsigned_t __remainder = 0;
// Algorithm: restoring binary division
// Reference: https://marz.utk.edu/my-courses/cosc130/lectures/binary-arithmetic/
// The dividend 2^power may exceed the range of __unsigned_t. The algorithm processes the dividend bit-by-bit from
// MSB to LSB without materializing it.
// Since 2^power has exactly one set bit, the loop contributes to 1 only when the current bit index equals __power,
// and 0 otherwise.
// At each iteration, the remainder is shifted left, the next dividend bit is appended.
// If the result is >= __divisor, the divisor is subtracted and 1-bit is recorded in the quotient.
for (int __bit = __power; __bit >= 0; --__bit)
{
// __carry_flag only matters for unsigned types with large divisors (MSB set: >= 2^(N-1))
const bool __carry_flag = (__remainder >> (__num_bits - 1)) != 0; // __remainder / 2^N != 0
__remainder <<= 1;
__remainder |= (__bit == __power); // append the remainder bit
const bool __quotient_bit = __carry_flag || (__remainder >= __divisor);
__quotient <<= 1; // shift
__quotient |= unsigned{__quotient_bit}; // append the quotient bit
if (__quotient_bit)
{
__remainder -= __divisor;
}
}
return {__quotient, __remainder};
}
_Tp __divisor = 1;
__unsigned_t __multiplier = 0;
unsigned __add = 0;
int __shift = 0;
};
/***********************************************************************************************************************
* Non-member functions
**********************************************************************************************************************/
//! @brief Computes both quotient and remainder of dividing the dividend by the precomputed divisor
//! @param[in] __dividend The dividend value
//! @param[in] __divisor The precomputed divisor
//! @return A pair of (quotient, remainder)
template <typename _Tp, typename _Lhs, bool _DivisorIsNeverOne>
[[nodiscard]] _CCCL_API ::cuda::std::pair<_Tp, _Lhs>
div(_Tp __dividend, fast_mod_div<_Lhs, _DivisorIsNeverOne> __divisor) noexcept
{
const auto __quotient = __dividend / __divisor;
const auto __remainder = __dividend - __quotient * __divisor;
return {__quotient, __remainder};
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_FAST_MODULO_DIVISION_H

View File

@@ -0,0 +1,203 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_ILOG_H
#define _CUDA___CMATH_ILOG_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__bit/has_single_bit.h>
#include <cuda/std/__bit/integral.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__limits/numeric_limits.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/array>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr int ilog2(const _Tp __t) noexcept
{
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
_CCCL_ASSERT(__t > 0, "ilog2() argument must be strictly positive");
auto __log2_approx = ::cuda::std::__bit_log2(static_cast<_Up>(__t));
_CCCL_ASSUME(__log2_approx <= ::cuda::std::numeric_limits<_Tp>::digits);
return __log2_approx;
}
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr int ceil_ilog2(const _Tp __t) noexcept
{
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
return ::cuda::ilog2(__t) + !::cuda::std::has_single_bit(static_cast<_Up>(__t));
}
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL ::cuda::std::array<::cuda::std::uint32_t, 10> __power_of_10_32bit() noexcept
{
return {10,
100,
1'000,
10'000,
100'000,
1'000'000,
10'000'000,
100'000'000,
1'000'000'000,
::cuda::std::numeric_limits<::cuda::std::uint32_t>::max()};
}
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL ::cuda::std::array<::cuda::std::uint64_t, 20> __power_of_10_64bit() noexcept
{
return {
10,
100,
1'000,
10'000,
100'000,
1'000'000,
10'000'000,
100'000'000,
1'000'000'000,
10'000'000'000,
100'000'000'000,
1'000'000'000'000,
10'000'000'000'000,
100'000'000'000'000,
1'000'000'000'000'000,
10'000'000'000'000'000,
100'000'000'000'000'000,
1'000'000'000'000'000'000,
10'000'000'000'000'000'000ull,
::cuda::std::numeric_limits<::cuda::std::uint64_t>::max()};
}
#if _CCCL_HAS_INT128()
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL ::cuda::std::array<__uint128_t, 39> __power_of_10_128bit() noexcept
{
return {
10,
100,
1'000,
10'000,
100'000,
1'000'000,
10'000'000,
100'000'000,
1'000'000'000,
10'000'000'000,
100'000'000'000,
1'000'000'000'000,
10'000'000'000'000,
100'000'000'000'000,
1'000'000'000'000'000,
10'000'000'000'000'000,
100'000'000'000'000'000,
1'000'000'000'000'000'000,
10'000'000'000'000'000'000ull,
__uint128_t{10'000'000'000'000'000'000ull} * 10,
__uint128_t{10'000'000'000'000'000'000ull} * 100,
__uint128_t{10'000'000'000'000'000'000ull} * 1'000,
__uint128_t{10'000'000'000'000'000'000ull} * 10'000,
__uint128_t{10'000'000'000'000'000'000ull} * 100'000,
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 10'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 100'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 10'000'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 100'000'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 10'000'000'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 100'000'000'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000'000'000,
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000'000'0000,
__uint128_t{10'000'000'000'000'000'000ull} * 10'000'000'000'000'0000,
__uint128_t{10'000'000'000'000'000'000ull} * 100'000'000'000'000'0000,
__uint128_t{10'000'000'000'000'000'000ull} * 1'000'000'000'000'000'0000ull,
::cuda::std::numeric_limits<__uint128_t>::max()};
}
#endif // _CCCL_HAS_INT128()
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr int ilog10(const _Tp __t) noexcept
{
using ::cuda::std::uint32_t;
using ::cuda::std::uint64_t;
_CCCL_ASSERT(__t > 0, "cuda::ilog10() argument must be strictly positive");
constexpr auto __reciprocal_log2_10 = 0.301029995663f; // 1 / log2(10)
const auto __log2 = ::cuda::ilog2(__t) * __reciprocal_log2_10;
auto __log10_approx = static_cast<int>(__log2);
if constexpr (sizeof(_Tp) <= sizeof(uint32_t))
{
_CCCL_ASSERT(__log10_approx < static_cast<int>(::cuda::__power_of_10_32bit().size()), "out of bounds");
if constexpr (::cuda::std::is_same_v<_Tp, uint32_t>)
{
// don't replace +1 with >= because wraparound behavior is needed here
__log10_approx += static_cast<uint32_t>(__t) + 1 > ::cuda::__power_of_10_32bit()[__log10_approx];
}
else
{
__log10_approx += static_cast<uint32_t>(__t) >= ::cuda::__power_of_10_32bit()[__log10_approx];
}
}
else if constexpr (sizeof(_Tp) == sizeof(uint64_t))
{
_CCCL_ASSERT(__log10_approx < static_cast<int>(::cuda::__power_of_10_64bit().size()), "out of bounds");
// +1 is not needed here
__log10_approx += static_cast<uint64_t>(__t) >= ::cuda::__power_of_10_64bit()[__log10_approx];
}
#if _CCCL_HAS_INT128()
else
{
_CCCL_ASSERT(__log10_approx < static_cast<int>(::cuda::__power_of_10_128bit().size()), "out of bounds");
if constexpr (::cuda::std::is_same_v<_Tp, __uint128_t>)
{
// don't replace +1 with >= because wraparound behavior is needed here
__log10_approx += static_cast<__uint128_t>(__t) + 1 > ::cuda::__power_of_10_128bit()[__log10_approx];
}
else
{
__log10_approx += static_cast<__uint128_t>(__t) >= ::cuda::__power_of_10_128bit()[__log10_approx];
}
}
#endif // _CCCL_HAS_INT128()
_CCCL_ASSUME(__log10_approx <= ::cuda::std::numeric_limits<_Tp>::digits / 3); // 2^X < 10^(x/3) -> 8^X < 10^x
return __log10_approx;
}
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr int ceil_ilog10(const _Tp __t) noexcept
{
_CCCL_ASSERT(__t > 0, "cuda::ceil_ilog10() argument must be strictly positive");
return __t == 1 ? 0 : ::cuda::ilog10(static_cast<_Tp>(__t - 1)) + 1;
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_ILOG_H

View File

@@ -0,0 +1,111 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_IPOW_H
#define _CUDA___CMATH_IPOW_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__cmath/ilog.h>
#include <cuda/__cmath/neg.h>
#include <cuda/__cmath/pow2.h>
#include <cuda/__cmath/uabs.h>
#include <cuda/std/__bit/countl.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/is_unsigned.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__utility/cmp.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <class _Tp, class _Ep>
[[nodiscard]] _CCCL_API constexpr _Tp __cccl_ipow_impl_base_pow2(_Tp __b, _Ep __e) noexcept
{
const auto __shift = static_cast<int>(__e - 1) * ::cuda::ilog2(__b);
const auto __lz = ::cuda::std::countl_zero(__b);
return (__shift >= __lz) ? _Tp{0} : (_Tp{__b} << __shift);
}
template <class _Tp, class _Ep>
[[nodiscard]] _CCCL_API constexpr _Tp __cccl_ipow_impl(_Tp __b, _Ep __e) noexcept
{
static_assert(::cuda::std::is_unsigned_v<_Tp>);
if (::cuda::is_power_of_two(__b))
{
return ::cuda::__cccl_ipow_impl_base_pow2(__b, __e);
}
auto __x = __b;
auto __y = _Tp{1};
while (__e > 1)
{
if (__e % 2 == 1)
{
__y *= __x;
--__e;
}
__x *= __x;
__e /= 2;
}
return __x * __y;
}
//! @brief Computes the integer power of a base to an exponent.
//! @param __b The base
//! @param __e The exponent
//! @pre \p __b must be an integer type
//! @pre \p __e must be an integer type
//! @return The result of raising \p __b to the power of \p __e
//! @note The result is undefined if \p __b is 0 and \p __e is negative.
_CCCL_TEMPLATE(class _Tp, class _Ep)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND ::cuda::std::__cccl_is_integer_v<_Ep>)
[[nodiscard]] _CCCL_API constexpr _Tp ipow(_Tp __b, _Ep __e) noexcept
{
_CCCL_ASSERT(__b != _Tp{0} || ::cuda::std::cmp_greater_equal(__e, _Ep{0}),
"cuda::ipow() requires non-negative exponent for base 0");
if (__e == _Ep{0} || __b == _Tp{1})
{
return _Tp{1};
}
else if (::cuda::std::cmp_less(__e, _Ep{0}) || __b == _Tp{0})
{
return _Tp{0};
}
auto __res = ::cuda::__cccl_ipow_impl(::cuda::uabs(__b), ::cuda::std::__to_unsigned_like(__e));
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
if (__b < _Tp{0} && (__e % 2u == 1))
{
__res = cuda::neg(__res);
}
}
return static_cast<_Tp>(__res);
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_IPOW_H

View File

@@ -0,0 +1,80 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_ISQRT_H
#define _CUDA___CMATH_ISQRT_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__bit/integral.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Returns the square root of the given non-negative integer rounded down
//! @param __v The input number
//! @pre \p __v must be an integer type
//! @pre \p __v must be non-negative
//! @return The square root of \p __v rounded down
//! @warning If \p __v is negative, the behavior is undefined
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr _Tp isqrt(_Tp __v) noexcept
{
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
_CCCL_ASSERT(__v >= _Tp{0}, "cuda::isqrt requires non-negative input");
}
if (__v <= 1)
{
return __v;
}
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
_Up __uv = static_cast<_Up>(__v);
_Up __ret{};
_Up __bit = static_cast<_Up>(_Up{1} << ((::cuda::std::bit_width(__uv) - 1) & (~1)));
while (__bit != 0)
{
if (__uv >= __ret + __bit)
{
__uv -= __ret + __bit;
__ret = (__ret >> 1) + __bit;
}
else
{
__ret >>= 1;
}
__bit >>= 2;
}
return static_cast<_Tp>(__ret);
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_ISQRT_H

View File

@@ -0,0 +1,147 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_MUL_HI_H
#define _CUDA___CMATH_MUL_HI_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/make_nbit_int.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__type_traits/num_bits.h>
#include <cuda/std/cstdint>
#if _CCCL_COMPILER(MSVC)
# include <intrin.h>
#endif // _CCCL_COMPILER(MSVC)
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
/***********************************************************************************************************************
* Extract higher bits after multiplication
**********************************************************************************************************************/
template <typename _Tp>
[[nodiscard]] _CCCL_API constexpr _Tp __mul_hi_fallback(_Tp __lhs, _Tp __rhs) noexcept
{
static_assert(::cuda::std::is_unsigned_v<_Tp>, "__mul_hi_fallback: T is required to be a unsigned integer type");
constexpr int __half_bits = ::cuda::std::__num_bits_v<_Tp> / 2;
using __half_bits_t = ::cuda::std::__make_nbit_uint_t<__half_bits>;
const auto __lhs_low = static_cast<__half_bits_t>(__lhs); // 32-bit
const auto __lhs_high = static_cast<__half_bits_t>(__lhs >> __half_bits); // 32-bit
const auto __rhs_low = static_cast<__half_bits_t>(__rhs); // 32-bit
const auto __rhs_high = static_cast<__half_bits_t>(__rhs >> __half_bits); // 32-bit
const auto __po_half = (static_cast<_Tp>(__lhs_low) * __rhs_low) >> __half_bits;
const auto __p1 = static_cast<_Tp>(__lhs_low) * __rhs_high; // 64-bit
const auto __p2 = static_cast<_Tp>(__lhs_high) * __rhs_low; // 64-bit
const auto __p3 = static_cast<_Tp>(__lhs_high) * __rhs_high; // 64-bit
const auto __p1_half = static_cast<__half_bits_t>(__p1); // 32-bit
const auto __p2_half = static_cast<__half_bits_t>(__p2); // 32-bit
const auto __carry = (__po_half + __p1_half + __p2_half) >> __half_bits; // 64-bit
return __p3 + (__p1 >> __half_bits) + (__p2 >> __half_bits) + __carry;
}
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
[[nodiscard]]
_CCCL_API constexpr _Tp mul_hi(_Tp __lhs, _Tp __rhs) noexcept
{
using ::cuda::std::int64_t;
using ::cuda::std::is_signed_v;
#if !_CCCL_TILE_COMPILATION() // nvbug6085239 error: calling a __device__ function from a __tile__ function
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
{
if constexpr (sizeof(_Tp) == sizeof(int))
{
if constexpr (is_signed_v<_Tp>)
{
[[maybe_unused]] const auto __lhs1 = static_cast<int>(__lhs);
[[maybe_unused]] const auto __rhs1 = static_cast<int>(__rhs);
NV_IF_TARGET(NV_IS_DEVICE, (return ::__mulhi(__lhs1, __rhs1);));
}
else // is_unsigned_v<_Tp>
{
[[maybe_unused]] const auto __lhs1 = static_cast<unsigned>(__lhs);
[[maybe_unused]] const auto __rhs1 = static_cast<unsigned>(__rhs);
NV_IF_TARGET(NV_IS_DEVICE, (return ::__umulhi(__lhs1, __rhs1);));
}
}
else if constexpr (sizeof(_Tp) == sizeof(int64_t))
{
if constexpr (is_signed_v<_Tp>)
{
[[maybe_unused]] const auto __lhs1 = static_cast<long long>(__lhs);
[[maybe_unused]] const auto __rhs1 = static_cast<long long>(__rhs);
NV_IF_TARGET(NV_IS_DEVICE, (return ::__mul64hi(__lhs1, __rhs1);));
# if _CCCL_COMPILER(MSVC)
NV_IF_TARGET(NV_IS_HOST, (return ::__mulh(__lhs1, __rhs1);));
# endif // _CCCL_COMPILER(MSVC)
}
else // is_unsigned_v<_Tp>
{
[[maybe_unused]] const auto __lhs1 = static_cast<unsigned long long>(__lhs);
[[maybe_unused]] const auto __rhs1 = static_cast<unsigned long long>(__rhs);
NV_IF_TARGET(NV_IS_DEVICE, (return ::__umul64hi(__lhs1, __rhs1);));
# if _CCCL_COMPILER(MSVC)
NV_IF_TARGET(NV_IS_HOST, (return ::__umulh(__lhs1, __rhs1);));
# endif // _CCCL_COMPILER(MSVC)
}
}
}
#endif // !_CCCL_TILE_COMPILATION()
if constexpr (sizeof(_Tp) < sizeof(int64_t) || (sizeof(_Tp) == sizeof(int64_t) && _CCCL_HAS_INT128()))
{
constexpr auto __bits = ::cuda::std::__num_bits_v<_Tp>;
using __larger_t = ::cuda::std::__make_nbit_int_t<__bits * 2, is_signed_v<_Tp>>;
const auto __ret = (static_cast<__larger_t>(__lhs) * __rhs) >> __bits;
return static_cast<_Tp>(__ret);
}
else // sizeof(_Tp) >= sizeof(int64_t) && !_CCCL_HAS_INT128()
{
if constexpr (is_signed_v<_Tp>)
{
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
const auto __lhs1 = static_cast<_Up>(__lhs);
const auto __rhs1 = static_cast<_Up>(__rhs);
auto __hi = ::cuda::__mul_hi_fallback(__lhs1, __rhs1);
if (__lhs < 0)
{
__hi -= __rhs1;
}
if (__rhs < 0)
{
__hi -= __lhs1;
}
return static_cast<_Tp>(__hi);
}
else
{
return ::cuda::__mul_hi_fallback(__lhs, __rhs);
}
}
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_MULTIPLY_HIGH_HALF_H

View File

@@ -0,0 +1,47 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_NEG_H
#define _CUDA___CMATH_NEG_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Returns the negative value of the input number
//! @param __v The input number
//! @return The signed negative value of \p __v
//! @note This function doesn't cause undefined behavior when negating the minimum value of a signed integer type.
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr _Tp neg(_Tp __v) noexcept
{
return static_cast<_Tp>(~::cuda::std::__to_unsigned_like(__v) + 1);
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_NEG_H

View File

@@ -0,0 +1,74 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_POW2_H
#define _CUDA___CMATH_POW2_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__bit/has_single_bit.h>
#include <cuda/std/__bit/integral.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr bool is_power_of_two(_Tp __t) noexcept
{
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
_CCCL_ASSERT(__t >= _Tp{0}, "cuda::is_power_of_two requires non-negative input");
}
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
return ::cuda::std::has_single_bit(static_cast<_Up>(__t));
}
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr _Tp next_power_of_two(_Tp __t) noexcept
{
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
_CCCL_ASSERT(__t >= _Tp{0}, "cuda::is_power_of_two requires non-negative input");
}
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
return ::cuda::std::bit_ceil(static_cast<_Up>(__t));
}
_CCCL_TEMPLATE(typename _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr _Tp prev_power_of_two(_Tp __t) noexcept
{
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
_CCCL_ASSERT(__t >= _Tp{0}, "cuda::is_power_of_two requires non-negative input");
}
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
return ::cuda::std::bit_floor(static_cast<_Up>(__t));
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_POW2_H

View File

@@ -0,0 +1,102 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_ROUND_DOWN_H
#define _CUDA___CMATH_ROUND_DOWN_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__type_traits/common_type.h>
#include <cuda/std/__type_traits/is_enum.h>
#include <cuda/std/__type_traits/is_integral.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__utility/to_underlying.h>
#include <cuda/std/limits>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Round the number \p __a to the previous multiple of \p __b
//! @param __a The input number
//! @param __b The multiplicand
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, _Up> round_down(const _Tp __a, const _Up __b) noexcept
{
_CCCL_ASSERT(__b > _Up{0}, "cuda::round_down: 'b' must be positive");
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
_CCCL_ASSERT(__a >= _Tp{0}, "cuda::round_down: 'a' must be non negative");
}
using _Common = ::cuda::std::common_type_t<_Tp, _Up>;
using _Prom = decltype(_Tp{} / _Up{});
using _UProm = ::cuda::std::make_unsigned_t<_Prom>;
auto __c1 = static_cast<_UProm>(__a) / static_cast<_UProm>(__b);
return static_cast<_Common>(__c1 * static_cast<_UProm>(__b));
}
//! @brief Round the number \p __a to the previous multiple of \p __b
//! @param __a The input number
//! @param __b The multiplicand
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, ::cuda::std::underlying_type_t<_Up>>
round_down(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::round_down(__a, ::cuda::std::to_underlying(__b));
}
//! @brief Round the number \p __a to the previous multiple of \p __b
//! @param __a The input number
//! @param __b The multiplicand
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, _Up>
round_down(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::round_down(::cuda::std::to_underlying(__a), __b);
}
//! @brief Round the number \p __a to the previous multiple of \p __b
//! @param __a The input number
//! @param __b The multiplicand
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
[[nodiscard]]
_CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, ::cuda::std::underlying_type_t<_Up>>
round_down(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::round_down(::cuda::std::to_underlying(__a), ::cuda::std::to_underlying(__b));
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_ROUND_DOWN_H

View File

@@ -0,0 +1,104 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_ROUND_UP_H
#define _CUDA___CMATH_ROUND_UP_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__cmath/ceil_div.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__type_traits/common_type.h>
#include <cuda/std/__type_traits/is_enum.h>
#include <cuda/std/__type_traits/is_integral.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__utility/to_underlying.h>
#include <cuda/std/limits>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Round the number \p __a to the next multiple of \p __b
//! @param __a The input number
//! @param __b The multiplicand
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, _Up> round_up(const _Tp __a, const _Up __b) noexcept
{
_CCCL_ASSERT(__b > _Up{0}, "cuda::round_up: 'b' must be positive");
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
_CCCL_ASSERT(__a >= _Tp{0}, "cuda::round_up: 'a' must be non negative");
}
using _Common = ::cuda::std::common_type_t<_Tp, _Up>;
using _Prom = decltype(_Tp{} / _Up{});
auto __c = ::cuda::ceil_div(static_cast<_Prom>(__a), static_cast<_Prom>(__b));
_CCCL_ASSERT(static_cast<_Common>(__c) <= ::cuda::std::numeric_limits<_Common>::max() / static_cast<_Common>(__b),
"cuda::round_up: result overflow");
return static_cast<_Common>(static_cast<_Prom>(__c) * static_cast<_Prom>(__b));
}
//! @brief Round the number \p __a to the next multiple of \p __b
//! @param __a The input number
//! @param __b The multiplicand
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_integral_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<_Tp, ::cuda::std::underlying_type_t<_Up>>
round_up(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::round_up(__a, ::cuda::std::to_underlying(__b));
}
//! @brief Round the number \p __a to the next multiple of \p __b
//! @param __a The input number
//! @param __b The multiplicand
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_integral_v<_Up>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, _Up>
round_up(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::round_up(::cuda::std::to_underlying(__a), __b);
}
//! @brief Round the number \p __a to the next multiple of \p __b
//! @param __a The input number
//! @param __b The multiplicand
//! @pre \p __a must be non-negative
//! @pre \p __b must be positive
_CCCL_TEMPLATE(class _Tp, class _Up)
_CCCL_REQUIRES(::cuda::std::is_enum_v<_Tp> _CCCL_AND ::cuda::std::is_enum_v<_Up>)
[[nodiscard]]
_CCCL_API constexpr ::cuda::std::common_type_t<::cuda::std::underlying_type_t<_Tp>, ::cuda::std::underlying_type_t<_Up>>
round_up(const _Tp __a, const _Up __b) noexcept
{
return ::cuda::round_up(::cuda::std::to_underlying(__a), ::cuda::std::to_underlying(__b));
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_ROUND_UP_H

View File

@@ -0,0 +1,134 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_SINCOS_H
#define _CUDA___CMATH_SINCOS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__cmath/trigonometric_functions.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__type_traits/conditional.h>
#include <cuda/std/__type_traits/is_extended_arithmetic.h>
#include <cuda/std/__type_traits/is_integral.h>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/std/__cccl/prologue.h>
#if _CCCL_HAS_BUILTIN(__builtin_sincosf) || _CCCL_COMPILER(GCC)
# define _CCCL_BUILTIN_SINCOSF(...) __builtin_sincosf(__VA_ARGS__)
#endif // _CCCL_HAS_BUILTIN(__builtin_sincosf) || _CCCL_COMPILER(GCC)
#if _CCCL_HAS_BUILTIN(__builtin_sincos) || _CCCL_COMPILER(GCC)
# define _CCCL_BUILTIN_SINCOS(...) __builtin_sincos(__VA_ARGS__)
#endif // _CCCL_HAS_BUILTIN(__builtin_sincos) || _CCCL_COMPILER(GCC)
#if _CCCL_HAS_BUILTIN(__builtin_sincosl) || _CCCL_COMPILER(GCC)
# define _CCCL_BUILTIN_SINCOSL(...) __builtin_sincosl(__VA_ARGS__)
#endif // _CCCL_HAS_BUILTIN(__builtin_sincosl) || _CCCL_COMPILER(GCC)
// clang-cuda crashes if these builtins are used.
#if _CCCL_CUDA_COMPILER(CLANG)
# undef _CCCL_BUILTIN_SINCOSF
# undef _CCCL_BUILTIN_SINCOS
# undef _CCCL_BUILTIN_SINCOSL
#endif // _CCCL_CUDA_COMPILER(CLANG)
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Type returned by \c cuda::sincos.
template <class _Tp>
struct _CCCL_TYPE_VISIBILITY_DEFAULT sincos_result
{
_Tp sin; //!< The sin result.
_Tp cos; //!< The cos result.
};
//! @brief Computes sin and cos operation of a value.
//!
//! @param __v The value.
//!
//! @return The \c cuda::sincos_result with the results of sin and cos operations.
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES(::cuda::std::__is_extended_arithmetic_v<_Tp>)
[[nodiscard]] _CCCL_API auto sincos(_Tp __v) noexcept
-> sincos_result<::cuda::std::conditional_t<::cuda::std::is_integral_v<_Tp>, double, _Tp>>
{
if constexpr (::cuda::std::is_integral_v<_Tp>)
{
return ::cuda::sincos(static_cast<double>(__v));
}
else
{
[[maybe_unused]] sincos_result<_Tp> __ret{};
#if defined(_CCCL_BUILTIN_SINCOSF)
if constexpr (::cuda::std::is_same_v<_Tp, float>)
{
_CCCL_BUILTIN_SINCOSF(__v, &__ret.sin, &__ret.cos);
return __ret;
}
#endif // _CCCL_BUILTIN_SINCOSF
#if defined(_CCCL_BUILTIN_SINCOS)
if constexpr (::cuda::std::is_same_v<_Tp, double>)
{
_CCCL_BUILTIN_SINCOS(__v, &__ret.sin, &__ret.cos);
return __ret;
}
#endif // _CCCL_BUILTIN_SINCOS
#if _CCCL_HAS_LONG_DOUBLE() && defined(_CCCL_BUILTIN_SINCOSL)
if constexpr (::cuda::std::is_same_v<_Tp, long double>)
{
_CCCL_BUILTIN_SINCOSL(__v, &__ret.sin, &__ret.cos);
return __ret;
}
#endif // _CCCL_HAS_LONG_DOUBLE() && _CCCL_BUILTIN_SINCOSL
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
{
if constexpr (::cuda::std::is_same_v<_Tp, float>)
{
NV_IF_TARGET(NV_IS_DEVICE, (::sincosf(__v, &__ret.sin, &__ret.cos); return __ret;))
}
if constexpr (::cuda::std::is_same_v<_Tp, double>)
{
NV_IF_TARGET(NV_IS_DEVICE, (::sincos(__v, &__ret.sin, &__ret.cos); return __ret;))
}
#if _LIBCUDACXX_HAS_NVFP16()
if constexpr (::cuda::std::is_same_v<_Tp, ::__half>)
{
const auto __result_float = ::cuda::sincos(::__half2float(__v));
return {::__float2half(__result_float.sin), ::__float2half(__result_float.cos)};
}
#endif // _LIBCUDACXX_HAS_NVFP16()
#if _LIBCUDACXX_HAS_NVBF16()
if constexpr (::cuda::std::is_same_v<_Tp, ::__nv_bfloat16>)
{
const auto __result_float = ::cuda::sincos(::__bfloat162float(__v));
return {::__float2bfloat16(__result_float.sin), ::__float2bfloat16(__result_float.cos)};
}
#endif // _LIBCUDACXX_HAS_NVBF16()
}
return {::cuda::std::sin(__v), ::cuda::std::cos(__v)};
}
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_SINCOS_H

View File

@@ -0,0 +1,57 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___CMATH_UABS_H
#define _CUDA___CMATH_UABS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__cmath/neg.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/make_unsigned.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Returns the *unsigned* absolute value of the given number.
//! @param __v The input number
//! @pre \p __v must be an integer type
//! @return The unsigned absolute value of \p __v
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES(::cuda::std::__cccl_is_cv_integer_v<_Tp>)
[[nodiscard]] _CCCL_API constexpr ::cuda::std::make_unsigned_t<_Tp> uabs(_Tp __v) noexcept
{
if constexpr (::cuda::std::is_signed_v<_Tp>)
{
using _Up = ::cuda::std::make_unsigned_t<_Tp>;
return (__v < _Tp(0)) ? static_cast<_Up>(::cuda::neg(__v)) : static_cast<_Up>(__v);
}
else
{
return __v;
}
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___CMATH_UABS_H