CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
417 lines
12 KiB
Plaintext
417 lines
12 KiB
Plaintext
//===----------------------------------------------------------------------===//
|
|
//
|
|
// Part of libcu++, the C++ Standard Library for your entire system,
|
|
// under the Apache License v2.0 with LLVM Exceptions.
|
|
// See https://llvm.org/LICENSE.txt for license information.
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
|
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
|
//
|
|
//===----------------------------------------------------------------------===//
|
|
|
|
#ifndef _CUDA_STD_RATIO
|
|
#define _CUDA_STD_RATIO
|
|
|
|
#include <cuda/std/detail/__config>
|
|
|
|
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
|
# pragma GCC system_header
|
|
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
|
# pragma clang system_header
|
|
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
|
# pragma system_header
|
|
#endif // no system header
|
|
|
|
#include <cuda/std/__type_traits/integral_constant.h>
|
|
#include <cuda/std/climits>
|
|
#include <cuda/std/cstdint>
|
|
|
|
#include <cuda/std/__cccl/prologue.h>
|
|
|
|
_CCCL_BEGIN_NAMESPACE_CUDA_STD
|
|
|
|
// __static_gcd
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
struct __static_gcd
|
|
{
|
|
static const intmax_t value = __static_gcd<_Yp, _Xp % _Yp>::value;
|
|
};
|
|
|
|
template <intmax_t _Xp>
|
|
struct __static_gcd<_Xp, 0>
|
|
{
|
|
static const intmax_t value = _Xp;
|
|
};
|
|
|
|
template <>
|
|
struct __static_gcd<0, 0>
|
|
{
|
|
static const intmax_t value = 1;
|
|
};
|
|
|
|
// __static_lcm
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
struct __static_lcm
|
|
{
|
|
static const intmax_t value = _Xp / __static_gcd<_Xp, _Yp>::value * _Yp;
|
|
};
|
|
|
|
template <intmax_t _Xp>
|
|
struct __static_abs
|
|
{
|
|
static const intmax_t value = _Xp < 0 ? -_Xp : _Xp;
|
|
};
|
|
|
|
template <intmax_t _Xp>
|
|
struct __static_sign
|
|
{
|
|
static const intmax_t value = _Xp == 0 ? 0 : (_Xp < 0 ? -1 : 1);
|
|
};
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp, intmax_t = __static_sign<_Yp>::value>
|
|
class __ll_add;
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
class __ll_add<_Xp, _Yp, 1>
|
|
{
|
|
static const intmax_t min = (1LL << (sizeof(intmax_t) * CHAR_BIT - 1)) + 1;
|
|
static const intmax_t max = -min;
|
|
|
|
static_assert(_Xp <= max - _Yp, "overflow in __ll_add");
|
|
|
|
public:
|
|
static const intmax_t value = _Xp + _Yp;
|
|
};
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
class __ll_add<_Xp, _Yp, 0>
|
|
{
|
|
public:
|
|
static const intmax_t value = _Xp;
|
|
};
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
class __ll_add<_Xp, _Yp, -1>
|
|
{
|
|
static const intmax_t min = (1LL << (sizeof(intmax_t) * CHAR_BIT - 1)) + 1;
|
|
static const intmax_t max = -min;
|
|
|
|
static_assert(min - _Yp <= _Xp, "overflow in __ll_add");
|
|
|
|
public:
|
|
static const intmax_t value = _Xp + _Yp;
|
|
};
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp, intmax_t = __static_sign<_Yp>::value>
|
|
class __ll_sub;
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
class __ll_sub<_Xp, _Yp, 1>
|
|
{
|
|
static const intmax_t min = (1LL << (sizeof(intmax_t) * CHAR_BIT - 1)) + 1;
|
|
static const intmax_t max = -min;
|
|
|
|
static_assert(min + _Yp <= _Xp, "overflow in __ll_sub");
|
|
|
|
public:
|
|
static const intmax_t value = _Xp - _Yp;
|
|
};
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
class __ll_sub<_Xp, _Yp, 0>
|
|
{
|
|
public:
|
|
static const intmax_t value = _Xp;
|
|
};
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
class __ll_sub<_Xp, _Yp, -1>
|
|
{
|
|
static const intmax_t min = (1LL << (sizeof(intmax_t) * CHAR_BIT - 1)) + 1;
|
|
static const intmax_t max = -min;
|
|
|
|
static_assert(_Xp <= max + _Yp, "overflow in __ll_sub");
|
|
|
|
public:
|
|
static const intmax_t value = _Xp - _Yp;
|
|
};
|
|
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
class __ll_mul
|
|
{
|
|
static const intmax_t nan = (1LL << (sizeof(intmax_t) * CHAR_BIT - 1));
|
|
static const intmax_t min = nan + 1;
|
|
static const intmax_t max = -min;
|
|
static const intmax_t __a_x = __static_abs<_Xp>::value;
|
|
static const intmax_t __a_y = __static_abs<_Yp>::value;
|
|
|
|
static_assert(_Xp != nan && _Yp != nan && __a_x <= max / __a_y, "overflow in __ll_mul");
|
|
|
|
public:
|
|
static const intmax_t value = _Xp * _Yp;
|
|
};
|
|
|
|
template <intmax_t _Yp>
|
|
class __ll_mul<0, _Yp>
|
|
{
|
|
public:
|
|
static const intmax_t value = 0;
|
|
};
|
|
|
|
template <intmax_t _Xp>
|
|
class __ll_mul<_Xp, 0>
|
|
{
|
|
public:
|
|
static const intmax_t value = 0;
|
|
};
|
|
|
|
template <>
|
|
class __ll_mul<0, 0>
|
|
{
|
|
public:
|
|
static const intmax_t value = 0;
|
|
};
|
|
|
|
// Not actually used but left here in case needed in future maintenance
|
|
template <intmax_t _Xp, intmax_t _Yp>
|
|
class __ll_div
|
|
{
|
|
static const intmax_t nan = (1LL << (sizeof(intmax_t) * CHAR_BIT - 1));
|
|
static const intmax_t min = nan + 1;
|
|
static const intmax_t max = -min;
|
|
|
|
static_assert(_Xp != nan && _Yp != nan && _Yp != 0, "overflow in __ll_div");
|
|
|
|
public:
|
|
static const intmax_t value = _Xp / _Yp;
|
|
};
|
|
|
|
template <intmax_t _Num, intmax_t _Den = 1>
|
|
class _CCCL_TYPE_VISIBILITY_DEFAULT ratio
|
|
{
|
|
static_assert(__static_abs<_Num>::value >= 0, "ratio numerator is out of range");
|
|
static_assert(_Den != 0, "ratio divide by 0");
|
|
static_assert(__static_abs<_Den>::value > 0, "ratio denominator is out of range");
|
|
static constexpr intmax_t __na = __static_abs<_Num>::value;
|
|
static constexpr intmax_t __da = __static_abs<_Den>::value;
|
|
static constexpr intmax_t __s = __static_sign<_Num>::value * __static_sign<_Den>::value;
|
|
static constexpr intmax_t __gcd = __static_gcd<__na, __da>::value;
|
|
|
|
public:
|
|
static constexpr intmax_t num = __s * __na / __gcd;
|
|
static constexpr intmax_t den = __da / __gcd;
|
|
|
|
using type = ratio<num, den>;
|
|
};
|
|
|
|
template <intmax_t _Num, intmax_t _Den>
|
|
constexpr intmax_t ratio<_Num, _Den>::num;
|
|
|
|
template <intmax_t _Num, intmax_t _Den>
|
|
constexpr intmax_t ratio<_Num, _Den>::den;
|
|
|
|
template <class _Tp>
|
|
inline constexpr bool __is_cuda_std_ratio_v = false;
|
|
|
|
template <intmax_t _Num, intmax_t _Den>
|
|
inline constexpr bool __is_cuda_std_ratio_v<ratio<_Num, _Den>> = true;
|
|
|
|
using atto = ratio<1LL, 1000000000000000000LL>;
|
|
using femto = ratio<1LL, 1000000000000000LL>;
|
|
using pico = ratio<1LL, 1000000000000LL>;
|
|
using nano = ratio<1LL, 1000000000LL>;
|
|
using micro = ratio<1LL, 1000000LL>;
|
|
using milli = ratio<1LL, 1000LL>;
|
|
using centi = ratio<1LL, 100LL>;
|
|
using deci = ratio<1LL, 10LL>;
|
|
using deca = ratio<10LL, 1LL>;
|
|
using hecto = ratio<100LL, 1LL>;
|
|
using kilo = ratio<1000LL, 1LL>;
|
|
using mega = ratio<1000000LL, 1LL>;
|
|
using giga = ratio<1000000000LL, 1LL>;
|
|
using tera = ratio<1000000000000LL, 1LL>;
|
|
using peta = ratio<1000000000000000LL, 1LL>;
|
|
using exa = ratio<1000000000000000000LL, 1LL>;
|
|
|
|
template <class _R1, class _R2>
|
|
struct __ratio_multiply
|
|
{
|
|
// private:
|
|
static const intmax_t __gcd_n1_d2 = __static_gcd<_R1::num, _R2::den>::value;
|
|
static const intmax_t __gcd_d1_n2 = __static_gcd<_R1::den, _R2::num>::value;
|
|
|
|
public:
|
|
using type = typename ratio<__ll_mul<_R1::num / __gcd_n1_d2, _R2::num / __gcd_d1_n2>::value,
|
|
__ll_mul<_R2::den / __gcd_n1_d2, _R1::den / __gcd_d1_n2>::value>::type;
|
|
};
|
|
|
|
template <class _R1, class _R2>
|
|
using ratio_multiply = typename __ratio_multiply<_R1, _R2>::type;
|
|
|
|
template <class _R1, class _R2>
|
|
struct __ratio_divide
|
|
{
|
|
// private:
|
|
static const intmax_t __gcd_n1_n2 = __static_gcd<_R1::num, _R2::num>::value;
|
|
static const intmax_t __gcd_d1_d2 = __static_gcd<_R1::den, _R2::den>::value;
|
|
|
|
public:
|
|
using type = typename ratio<__ll_mul<_R1::num / __gcd_n1_n2, _R2::den / __gcd_d1_d2>::value,
|
|
__ll_mul<_R2::num / __gcd_n1_n2, _R1::den / __gcd_d1_d2>::value>::type;
|
|
};
|
|
|
|
template <class _R1, class _R2>
|
|
using ratio_divide = typename __ratio_divide<_R1, _R2>::type;
|
|
|
|
template <class _R1, class _R2>
|
|
struct __ratio_add
|
|
{
|
|
// private:
|
|
static const intmax_t __gcd_n1_n2 = __static_gcd<_R1::num, _R2::num>::value;
|
|
static const intmax_t __gcd_d1_d2 = __static_gcd<_R1::den, _R2::den>::value;
|
|
|
|
public:
|
|
using type =
|
|
typename ratio_multiply<ratio<__gcd_n1_n2, _R1::den / __gcd_d1_d2>,
|
|
ratio<__ll_add<__ll_mul<_R1::num / __gcd_n1_n2, _R2::den / __gcd_d1_d2>::value,
|
|
__ll_mul<_R2::num / __gcd_n1_n2, _R1::den / __gcd_d1_d2>::value>::value,
|
|
_R2::den>>::type;
|
|
};
|
|
|
|
template <class _R1, class _R2>
|
|
using ratio_add = typename __ratio_add<_R1, _R2>::type;
|
|
|
|
template <class _R1, class _R2>
|
|
struct __ratio_subtract
|
|
{
|
|
// private:
|
|
static const intmax_t __gcd_n1_n2 = __static_gcd<_R1::num, _R2::num>::value;
|
|
static const intmax_t __gcd_d1_d2 = __static_gcd<_R1::den, _R2::den>::value;
|
|
|
|
public:
|
|
using type =
|
|
typename ratio_multiply<ratio<__gcd_n1_n2, _R1::den / __gcd_d1_d2>,
|
|
ratio<__ll_sub<__ll_mul<_R1::num / __gcd_n1_n2, _R2::den / __gcd_d1_d2>::value,
|
|
__ll_mul<_R2::num / __gcd_n1_n2, _R1::den / __gcd_d1_d2>::value>::value,
|
|
_R2::den>>::type;
|
|
};
|
|
|
|
template <class _R1, class _R2>
|
|
using ratio_subtract = typename __ratio_subtract<_R1, _R2>::type;
|
|
|
|
// ratio_equal
|
|
|
|
template <class _R1, class _R2>
|
|
struct _CCCL_TYPE_VISIBILITY_DEFAULT ratio_equal : public bool_constant<(_R1::num == _R2::num && _R1::den == _R2::den)>
|
|
{};
|
|
|
|
template <class _R1, class _R2>
|
|
struct _CCCL_TYPE_VISIBILITY_DEFAULT ratio_not_equal : public bool_constant<(!ratio_equal<_R1, _R2>::value)>
|
|
{};
|
|
|
|
// ratio_less
|
|
|
|
template <class _R1,
|
|
class _R2,
|
|
bool _Odd = false,
|
|
intmax_t _Q1 = _R1::num / _R1::den,
|
|
intmax_t _M1 = _R1::num % _R1::den,
|
|
intmax_t _Q2 = _R2::num / _R2::den,
|
|
intmax_t _M2 = _R2::num % _R2::den>
|
|
struct __ratio_less1
|
|
{
|
|
static const bool value = _Odd ? _Q2 < _Q1 : _Q1 < _Q2;
|
|
};
|
|
|
|
template <class _R1, class _R2, bool _Odd, intmax_t _Qp>
|
|
struct __ratio_less1<_R1, _R2, _Odd, _Qp, 0, _Qp, 0>
|
|
{
|
|
static const bool value = false;
|
|
};
|
|
|
|
template <class _R1, class _R2, bool _Odd, intmax_t _Qp, intmax_t _M2>
|
|
struct __ratio_less1<_R1, _R2, _Odd, _Qp, 0, _Qp, _M2>
|
|
{
|
|
static const bool value = !_Odd;
|
|
};
|
|
|
|
template <class _R1, class _R2, bool _Odd, intmax_t _Qp, intmax_t _M1>
|
|
struct __ratio_less1<_R1, _R2, _Odd, _Qp, _M1, _Qp, 0>
|
|
{
|
|
static const bool value = _Odd;
|
|
};
|
|
|
|
template <class _R1, class _R2, bool _Odd, intmax_t _Qp, intmax_t _M1, intmax_t _M2>
|
|
struct __ratio_less1<_R1, _R2, _Odd, _Qp, _M1, _Qp, _M2>
|
|
{
|
|
static const bool value = __ratio_less1<ratio<_R1::den, _M1>, ratio<_R2::den, _M2>, !_Odd>::value;
|
|
};
|
|
|
|
template <class _R1,
|
|
class _R2,
|
|
intmax_t _S1 = __static_sign<_R1::num>::value,
|
|
intmax_t _S2 = __static_sign<_R2::num>::value>
|
|
struct __ratio_less
|
|
{
|
|
static const bool value = _S1 < _S2;
|
|
};
|
|
|
|
template <class _R1, class _R2>
|
|
struct __ratio_less<_R1, _R2, 1LL, 1LL>
|
|
{
|
|
static const bool value = __ratio_less1<_R1, _R2>::value;
|
|
};
|
|
|
|
template <class _R1, class _R2>
|
|
struct __ratio_less<_R1, _R2, -1LL, -1LL>
|
|
{
|
|
static const bool value = __ratio_less1<ratio<-_R2::num, _R2::den>, ratio<-_R1::num, _R1::den>>::value;
|
|
};
|
|
|
|
template <class _R1, class _R2>
|
|
struct _CCCL_TYPE_VISIBILITY_DEFAULT ratio_less : public bool_constant<(__ratio_less<_R1, _R2>::value)>
|
|
{};
|
|
|
|
template <class _R1, class _R2>
|
|
struct _CCCL_TYPE_VISIBILITY_DEFAULT ratio_less_equal : public bool_constant<(!ratio_less<_R2, _R1>::value)>
|
|
{};
|
|
|
|
template <class _R1, class _R2>
|
|
struct _CCCL_TYPE_VISIBILITY_DEFAULT ratio_greater : public bool_constant<(ratio_less<_R2, _R1>::value)>
|
|
{};
|
|
|
|
template <class _R1, class _R2>
|
|
struct _CCCL_TYPE_VISIBILITY_DEFAULT ratio_greater_equal : public bool_constant<(!ratio_less<_R1, _R2>::value)>
|
|
{};
|
|
|
|
template <class _R1, class _R2>
|
|
struct __ratio_gcd
|
|
{
|
|
using type = ratio<__static_gcd<_R1::num, _R2::num>::value, __static_lcm<_R1::den, _R2::den>::value>;
|
|
};
|
|
|
|
template <class _R1, class _R2>
|
|
inline constexpr bool ratio_equal_v = ratio_equal<_R1, _R2>::value;
|
|
|
|
template <class _R1, class _R2>
|
|
inline constexpr bool ratio_not_equal_v = ratio_not_equal<_R1, _R2>::value;
|
|
|
|
template <class _R1, class _R2>
|
|
inline constexpr bool ratio_less_v = ratio_less<_R1, _R2>::value;
|
|
|
|
template <class _R1, class _R2>
|
|
inline constexpr bool ratio_less_equal_v = ratio_less_equal<_R1, _R2>::value;
|
|
|
|
template <class _R1, class _R2>
|
|
inline constexpr bool ratio_greater_v = ratio_greater<_R1, _R2>::value;
|
|
|
|
template <class _R1, class _R2>
|
|
inline constexpr bool ratio_greater_equal_v = ratio_greater_equal<_R1, _R2>::value;
|
|
|
|
_CCCL_END_NAMESPACE_CUDA_STD
|
|
|
|
#include <cuda/std/__cccl/epilogue.h>
|
|
|
|
#endif // _CUDA_STD_RATIO
|