[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,126 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_CAPACITY_CUH
#define _CUDAX___CUCO_CAPACITY_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__cmath/ceil_div.h>
#include <cuda/__numeric/mul_overflow.h>
#include <cuda/__utility/in_range.h>
#include <cuda/std/__algorithm/max.h>
#include <cuda/std/__cmath/rounding_functions.h>
#include <cuda/std/__exception/exception_macros.h>
#include <cuda/std/__limits/numeric_limits.h>
#include <cuda/std/cstdint>
#include <cuda/experimental/__cuco/detail/prime.cuh>
#include <cuda/experimental/__cuco/probing_scheme.cuh>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief Rounds a requested capacity up to the smallest valid capacity for the given probing scheme
//! and bucket size.
//!
//! The probe stride is `_ProbingScheme::cg_size * _BucketSize`. For linear probing the result is a
//! multiple of the stride; for double hashing the probe cycle count `capacity / stride` is
//! additionally prime. The function is idempotent: applying it to an already valid capacity returns
//! the same value.
//!
//! @tparam _ProbingScheme Probing scheme type
//! @tparam _BucketSize Number of slots per bucket
//! @tparam _SizeType Size type
//!
//! @param __requested Requested capacity
//!
//! @return The smallest valid capacity that is greater than or equal to `__requested`
template <class _ProbingScheme, int _BucketSize, class _SizeType>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr _SizeType make_valid_capacity(_SizeType __requested)
{
constexpr auto __stride = _SizeType{_ProbingScheme::cg_size * _BucketSize};
const auto __cycles = ::cuda::ceil_div(::cuda::std::max(__requested, _SizeType{1}), __stride);
_SizeType __capacity{};
if constexpr (is_double_hashing_v<_ProbingScheme>)
{
const auto __prime = detail::__next_prime(static_cast<::cuda::std::uint64_t>(__cycles));
if (::cuda::mul_overflow(__capacity, __prime, __stride))
{
_CCCL_THROW(::std::logic_error, "Invalid input capacity");
}
}
else
{
const auto __num_buckets = __cycles + _SizeType{__requested == 0};
if (::cuda::mul_overflow(__capacity, __num_buckets, __stride))
{
_CCCL_THROW(::std::logic_error, "Invalid input capacity");
}
}
return __capacity;
}
//! @brief Rounds a requested capacity up to a valid capacity for a desired load factor.
//!
//! @tparam _ProbingScheme Probing scheme type
//! @tparam _BucketSize Number of slots per bucket
//! @tparam _SizeType Size type
//!
//! @param __requested Requested element count
//! @param __load_factor Desired load factor in (0, 1]
//!
//! @return The smallest valid capacity that fits `__requested` elements at `__load_factor`
template <class _ProbingScheme, int _BucketSize, class _SizeType>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr _SizeType make_valid_capacity(_SizeType __requested, double __load_factor)
{
if (__load_factor <= 0. || !::cuda::in_range(__load_factor, 0., 1.))
{
_CCCL_THROW(::std::logic_error, "Desired load factor must be in the range (0, 1]");
}
const auto __scaled = ::cuda::std::ceil(static_cast<double>(__requested) / __load_factor);
if (__scaled > static_cast<double>(::cuda::std::numeric_limits<_SizeType>::max()))
{
_CCCL_THROW(::std::logic_error,
"Invalid load factor: requested capacity divided by load factor exceeds the maximum representable "
"value");
}
return make_valid_capacity<_ProbingScheme, _BucketSize>(static_cast<_SizeType>(__scaled));
}
//! @brief Returns whether `__capacity` is already a valid capacity for the given probing scheme and
//! bucket size.
//!
//! @tparam _ProbingScheme Probing scheme type
//! @tparam _BucketSize Number of slots per bucket
//! @tparam _SizeType Size type
//!
//! @param __capacity Capacity to test
//!
//! @return `true` if `__capacity` needs no rounding
template <class _ProbingScheme, int _BucketSize, class _SizeType>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr bool is_valid_capacity(_SizeType __capacity)
{
return make_valid_capacity<_ProbingScheme, _BucketSize>(__capacity) == __capacity;
}
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_CAPACITY_CUH

View File

@@ -0,0 +1,59 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_BITWISE_COMPARE_CUH
#define _CUDAX___CUCO_DETAIL_BITWISE_COMPARE_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__type_traits/is_bitwise_comparable.h>
#include <cuda/std/__bit/bit_cast.h>
#include <cuda/std/__limits/numeric_limits.h>
#include <cuda/std/__type_traits/make_nbit_int.h>
#include <cuda/std/array>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::detail
{
//! @brief Bitwise equality comparison.
//!
//! @tparam _Tp Value type
template <class _Tp>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr bool __bitwise_compare(const _Tp& __lhs, const _Tp& __rhs)
{
static_assert(::cuda::is_bitwise_comparable_v<_Tp>,
"Bitwise compared objects must have unique object representations or be explicitly declared safe.");
if constexpr (sizeof(_Tp) <= sizeof(::cuda::std::uint64_t)
|| (sizeof(_Tp) == 2 * sizeof(::cuda::std::uint64_t) && _CCCL_HAS_INT128()))
{
using _Up = ::cuda::std::__make_nbit_uint_t<sizeof(_Tp) * ::cuda::std::numeric_limits<unsigned char>::digits>;
return ::cuda::std::bit_cast<_Up>(__lhs) == ::cuda::std::bit_cast<_Up>(__rhs);
}
else
{
using _Array = ::cuda::std::array<::cuda::std::uint64_t, sizeof(_Tp) / sizeof(::cuda::std::uint64_t)>;
return ::cuda::std::bit_cast<_Array>(__lhs) == ::cuda::std::bit_cast<_Array>(__rhs);
}
}
} // namespace cuda::experimental::cuco::detail
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_BITWISE_COMPARE_CUH

View File

@@ -0,0 +1,131 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_EQUAL_WRAPPER_CUH
#define _CUDAX___CUCO_DETAIL_EQUAL_WRAPPER_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/cstdint>
#include <cuda/experimental/__cuco/detail/bitwise_compare.cuh>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::detail
{
//! @brief Enum of equality comparison results.
enum class __equal_result : ::cuda::std::int8_t
{
__unequal,
__equal,
__empty,
__available,
};
//! @brief Enum indicating whether the operation is an insert.
enum class __is_insert : ::cuda::std::int8_t
{
__yes,
__no
};
//! @brief Key equality wrapper.
//!
//! @tparam _Tp Right-hand side element type
//! @tparam _Equal Equality callable
//! @tparam _AllowsDuplicates Duplicate key flag
template <class _Tp, class _Equal, bool _AllowsDuplicates>
struct __equal_wrapper
{
_Tp __empty_sentinel;
_Tp __erased_sentinel;
_Equal __equal;
//! @brief Equality wrapper constructor.
//!
//! @param __empty Empty sentinel value
//! @param __erased Erased sentinel value
//! @param __eq Equality binary callable
_CCCL_HOST_DEVICE_API constexpr __equal_wrapper(_Tp __empty, _Tp __erased, const _Equal& __eq) noexcept
: __empty_sentinel{__empty}
, __erased_sentinel{__erased}
, __equal{__eq}
{}
#if _CCCL_CUDA_COMPILATION()
//! @brief Equality check with the given equality callable.
//!
//! @tparam _Lhs Left-hand side element type
//! @tparam _Rhs Right-hand side element type
//!
//! @param __lhs Left-hand side element to check equality
//! @param __rhs Right-hand side element to check equality
//!
//! @return `__equal` if `__lhs` and `__rhs` are equivalent, `__unequal` otherwise
template <class _Lhs, class _Rhs>
[[nodiscard]] _CCCL_DEVICE_API constexpr __equal_result __equal_to(const _Lhs& __lhs, const _Rhs& __rhs) const noexcept
{
return __equal(__lhs, __rhs) ? __equal_result::__equal : __equal_result::__unequal;
}
//! @brief Order-sensitive equality operator.
//!
//! @note This function always compares the right-hand side element against sentinel values first
//! then performs an equality check with the given `__equal` callable, i.e., `__equal(__lhs, __rhs)`.
//! @note Container (like set or map) slots MUST always be on the right-hand side.
//!
//! @tparam _IsInsert Flag indicating whether it's an insert equality check or not. Insert probing
//! stops when it's an empty or erased slot while query probing stops only when it's empty.
//! @tparam _Lhs Left-hand side element type
//! @tparam _Rhs Right-hand side element type
//!
//! @param __lhs Left-hand side element to check equality
//! @param __rhs Right-hand side element to check equality
//!
//! @return Three-way equality comparison result
template <__is_insert _IsInsert, class _Lhs, class _Rhs>
[[nodiscard]] _CCCL_DEVICE_API constexpr __equal_result operator()(const _Lhs& __lhs, const _Rhs& __rhs) const noexcept
{
if constexpr (_IsInsert == __is_insert::__yes)
{
if (detail::__bitwise_compare(__rhs, __empty_sentinel) || detail::__bitwise_compare(__rhs, __erased_sentinel))
{
return __equal_result::__available;
}
else if constexpr (_AllowsDuplicates)
{
return __equal_result::__unequal;
}
else
{
return __equal_to(__lhs, __rhs);
}
}
else
{
return detail::__bitwise_compare(__rhs, __empty_sentinel) ? __equal_result::__empty : __equal_to(__lhs, __rhs);
}
}
#endif // _CCCL_CUDA_COMPILATION()
};
} // namespace cuda::experimental::cuco::detail
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_EQUAL_WRAPPER_CUH

View File

@@ -0,0 +1,848 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
/* MurmurHash3_32 implementation from
* https://github.com/aappleby/smhasher/blob/master/src/MurmurHash3.cpp
* -----------------------------------------------------------------------------
* MurmurHash3 was written by Austin Appleby, and is placed in the public domain. The author
* hereby disclaims copyright to this source code.
*
* Note - The x86 and x64 versions do _not_ produce the same results, as the algorithms are
* optimized for their respective platforms. You can still compile and run any of them on any
* platform, but your performance with the non-native version will be less than optimal.
*/
#ifndef _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_MURMURHASH3_CUH
#define _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_MURMURHASH3_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__utility/static_for.h>
#include <cuda/std/__bit/bit_cast.h>
#include <cuda/std/__bit/rotate.h>
#include <cuda/std/array>
#include <cuda/std/cstddef>
#include <cuda/std/cstdint>
#include <cuda/std/span>
#include <cuda/experimental/__cuco/detail/hash_functions/utils.cuh>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
template <typename _Key>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint32_t
__fmix32(_Key __key, ::cuda::std::uint32_t __seed = 0) noexcept
{
static_assert(sizeof(_Key) == 4, "Key type must be 4 bytes in size.");
auto __h = ::cuda::std::bit_cast<::cuda::std::uint32_t>(__key) ^ __seed;
__h ^= __h >> 16;
__h *= 0x85ebca6b;
__h ^= __h >> 13;
__h *= 0xc2b2ae35;
__h ^= __h >> 16;
return __h;
}
#if _CCCL_HAS_INT128()
template <typename _Key>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint64_t
__fmix64(_Key __key, ::cuda::std::uint64_t __seed = 0) noexcept
{
static_assert(sizeof(_Key) == 8, "Key type must be 8 bytes in size.");
auto __h = ::cuda::std::bit_cast<::cuda::std::uint64_t>(__key) ^ __seed;
__h ^= __h >> 33;
__h *= 0xff51afd7ed558ccdULL;
__h ^= __h >> 33;
__h *= 0xc4ceb9fe1a85ec53ULL;
__h ^= __h >> 33;
return __h;
}
#endif // _CCCL_HAS_INT128()
//! @brief A `MurmurHash3_32` hash function to hash the given argument on host and device.
//!
//! @tparam _Key The type of the values to hash
template <typename _Key>
struct _MurmurHash3_32
{
static constexpr ::cuda::std::uint32_t __c1 = 0xcc9e2d51;
static constexpr ::cuda::std::uint32_t __c2 = 0x1b873593;
static constexpr ::cuda::std::uint32_t __block_size = 4;
static constexpr ::cuda::std::uint32_t __chunk_size = 4;
_CCCL_HOST_DEVICE_API constexpr _MurmurHash3_32(::cuda::std::uint32_t __seed = 0)
: __seed_{__seed}
{}
//! @brief Returns a hash value for its argument, as a value of type `::cuda::std::uint32_t`.
//! @param __key The input argument to hash
//! @return The resulting hash value
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint32_t operator()(const _Key& __key) const noexcept
{
using _Holder = _Byte_holder<sizeof(_Key), __chunk_size, __block_size, false, ::cuda::std::uint32_t>;
return __compute_hash(::cuda::std::bit_cast<_Holder>(__key));
}
//! @brief Returns a hash value for its argument, as a value of type `::cuda::std::uint32_t`.
//! @tparam _Extent The extent type
//! @param __keys span of keys to hash
//! @return The resulting hash value
template <size_t _Extent>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint64_t
operator()(::cuda::std::span<_Key, _Extent> __keys) const noexcept
{
return __compute_hash_span(__keys);
}
private:
template <class _Holder>
[[nodiscard]] _CCCL_HOST_DEVICE_API ::cuda::std::uint32_t __compute_hash(_Holder __holder) const noexcept
{
::cuda::std::uint32_t __h1 = __seed_;
//----------
// body
if constexpr (_Holder::__num_blocks > 0)
{
::cuda::static_for<_Holder::__num_blocks>([&](auto __i) {
::cuda::std::uint32_t __k1 = __holder.__blocks[__i];
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 15);
__k1 *= __c2;
__h1 ^= __k1;
__h1 = ::cuda::std::rotl(__h1, 13);
__h1 = __h1 * 5 + 0xe6546b64;
});
}
//----------
// tail
if constexpr (_Holder::__tail_size > 0)
{
::cuda::std::uint32_t __k1 = 0;
switch (__holder.__tail_size)
{
case 3:
__k1 ^= ::cuda::std::to_integer<::cuda::std::uint32_t>(__holder.__bytes[2]) << 16;
[[fallthrough]];
case 2:
__k1 ^= ::cuda::std::to_integer<::cuda::std::uint32_t>(__holder.__bytes[1]) << 8;
[[fallthrough]];
case 1:
__k1 ^= ::cuda::std::to_integer<::cuda::std::uint32_t>(__holder.__bytes[0]);
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 15);
__k1 *= __c2;
__h1 ^= __k1;
};
}
//----------
// finalization
__h1 ^= ::cuda::std::uint32_t{sizeof(_Holder)};
__h1 = ::cuda::experimental::cuco::__fmix32(__h1);
return __h1;
}
[[nodiscard]] _CCCL_HOST_DEVICE_API ::cuda::std::uint32_t
__compute_hash_span(::cuda::std::span<const _Key> __keys) const noexcept
{
const auto __bytes = ::cuda::std::as_bytes(__keys).data();
const auto __size = __keys.size_bytes();
const auto __nblocks = __size / __block_size;
::cuda::std::uint32_t __h1 = __seed_;
//----------
// body
for (::cuda::std::remove_const_t<decltype(__nblocks)> __i = 0; __i < __nblocks; __i++)
{
::cuda::std::uint32_t __k1 = ::cuda::experimental::cuco::__load_chunk<::cuda::std::uint32_t>(__bytes, __i);
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 15);
__k1 *= __c2;
__h1 ^= __k1;
__h1 = ::cuda::std::rotl(__h1, 13);
__h1 = __h1 * 5 + 0xe6546b64;
}
//----------
// tail
::cuda::std::uint32_t __k1 = 0;
switch (__size % 4)
{
case 3:
__k1 ^= ::cuda::std::to_integer<::cuda::std::uint32_t>(__bytes[__nblocks * __block_size + 2]) << 16;
[[fallthrough]];
case 2:
__k1 ^= ::cuda::std::to_integer<::cuda::std::uint32_t>(__bytes[__nblocks * __block_size + 1]) << 8;
[[fallthrough]];
case 1:
__k1 ^= ::cuda::std::to_integer<::cuda::std::uint32_t>(__bytes[__nblocks * __block_size + 0]);
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 15);
__k1 *= __c2;
__h1 ^= __k1;
};
//----------
// finalization
__h1 ^= __size;
__h1 = ::cuda::experimental::cuco::__fmix32(__h1);
return __h1;
}
::cuda::std::uint32_t __seed_;
};
#if _CCCL_HAS_INT128()
template <typename _Key>
struct _MurmurHash3_x86_128
{
private:
static constexpr ::cuda::std::uint32_t __c1 = 0x239b961b;
static constexpr ::cuda::std::uint32_t __c2 = 0xab0e9789;
static constexpr ::cuda::std::uint32_t __c3 = 0x38b34ae5;
static constexpr ::cuda::std::uint32_t __c4 = 0xa1e38b93;
static constexpr ::cuda::std::uint32_t __block_size = 4;
static constexpr ::cuda::std::uint32_t __chunk_size = 16;
public:
_CCCL_HOST_DEVICE_API constexpr _MurmurHash3_x86_128(::cuda::std::uint32_t __seed = 0)
: __seed_{__seed}
{}
//! @brief Returns a hash value for its argument, as a value of type `__uint128_t`.
//! @param __key The input argument to hash
//! @return The resulting hash value
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __uint128_t operator()(const _Key& __key) const noexcept
{
using _Holder = _Byte_holder<sizeof(_Key), __chunk_size, __block_size, false, ::cuda::std::uint32_t>;
return __compute_hash(::cuda::std::bit_cast<_Holder>(__key));
}
//! @brief Returns a hash value for its argument, as a value of type `__uint128_t`.
//! @tparam _Extent The extent type
//! @param __keys span of keys to hash
//! @return The resulting hash value
template <size_t _Extent>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __uint128_t
operator()(::cuda::std::span<_Key, _Extent> __keys) const noexcept
{
return __compute_hash_span(__keys);
}
private:
template <class _Holder>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __uint128_t __compute_hash(_Holder __holder) const noexcept
{
::cuda::std::array<::cuda::std::uint32_t, 4> __h{__seed_, __seed_, __seed_, __seed_};
const auto __size = ::cuda::std::uint32_t{sizeof(_Holder)};
if constexpr (_Holder::__num_chunks > 0)
{
::cuda::static_for<_Holder::__num_chunks>([&](auto __i) {
::cuda::std::uint32_t __k1 = __holder.__blocks[4 * __i];
::cuda::std::uint32_t __k2 = __holder.__blocks[4 * __i + 1];
::cuda::std::uint32_t __k3 = __holder.__blocks[4 * __i + 2];
::cuda::std::uint32_t __k4 = __holder.__blocks[4 * __i + 3];
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 15);
__k1 *= __c2;
__h[0] ^= __k1;
__h[0] = ::cuda::std::rotl(__h[0], 19);
__h[0] += __h[1];
__h[0] = __h[0] * 5 + 0x561ccd1b;
__k2 *= __c2;
__k2 = ::cuda::std::rotl(__k2, 16);
__k2 *= __c3;
__h[1] ^= __k2;
__h[1] = ::cuda::std::rotl(__h[1], 17);
__h[1] += __h[2];
__h[1] = __h[1] * 5 + 0x0bcaa747;
__k3 *= __c3;
__k3 = ::cuda::std::rotl(__k3, 17);
__k3 *= __c4;
__h[2] ^= __k3;
__h[2] = ::cuda::std::rotl(__h[2], 15);
__h[2] += __h[3];
__h[2] = __h[2] * 5 + 0x96cd1c35;
__k4 *= __c4;
__k4 = ::cuda::std::rotl(__k4, 18);
__k4 *= __c1;
__h[3] ^= __k4;
__h[3] = ::cuda::std::rotl(__h[3], 13);
__h[3] += __h[0];
__h[3] = __h[3] * 5 + 0x32ac3b17;
});
}
// tail
if constexpr (_Holder::__tail_size > 0)
{
::cuda::std::uint32_t __k1 = 0;
::cuda::std::uint32_t __k2 = 0;
::cuda::std::uint32_t __k3 = 0;
::cuda::std::uint32_t __k4 = 0;
const auto __tail = __holder.__bytes;
switch (__size % __chunk_size)
{
case 15:
__k4 ^= static_cast<::cuda::std::uint32_t>(__tail[14]) << 16;
[[fallthrough]];
case 14:
__k4 ^= static_cast<::cuda::std::uint32_t>(__tail[13]) << 8;
[[fallthrough]];
case 13:
__k4 ^= static_cast<::cuda::std::uint32_t>(__tail[12]) << 0;
__k4 *= __c4;
__k4 = ::cuda::std::rotl(__k4, 18);
__k4 *= __c1;
__h[3] ^= __k4;
[[fallthrough]];
case 12:
__k3 ^= static_cast<::cuda::std::uint32_t>(__tail[11]) << 24;
[[fallthrough]];
case 11:
__k3 ^= static_cast<::cuda::std::uint32_t>(__tail[10]) << 16;
[[fallthrough]];
case 10:
__k3 ^= static_cast<::cuda::std::uint32_t>(__tail[9]) << 8;
[[fallthrough]];
case 9:
__k3 ^= static_cast<::cuda::std::uint32_t>(__tail[8]) << 0;
__k3 *= __c3;
__k3 = ::cuda::std::rotl(__k3, 17);
__k3 *= __c4;
__h[2] ^= __k3;
[[fallthrough]];
case 8:
__k2 ^= static_cast<::cuda::std::uint32_t>(__tail[7]) << 24;
[[fallthrough]];
case 7:
__k2 ^= static_cast<::cuda::std::uint32_t>(__tail[6]) << 16;
[[fallthrough]];
case 6:
__k2 ^= static_cast<::cuda::std::uint32_t>(__tail[5]) << 8;
[[fallthrough]];
case 5:
__k2 ^= static_cast<::cuda::std::uint32_t>(__tail[4]) << 0;
__k2 *= __c2;
__k2 = ::cuda::std::rotl(__k2, 16);
__k2 *= __c3;
__h[1] ^= __k2;
[[fallthrough]];
case 4:
__k1 ^= static_cast<::cuda::std::uint32_t>(__tail[3]) << 24;
[[fallthrough]];
case 3:
__k1 ^= static_cast<::cuda::std::uint32_t>(__tail[2]) << 16;
[[fallthrough]];
case 2:
__k1 ^= static_cast<::cuda::std::uint32_t>(__tail[1]) << 8;
[[fallthrough]];
case 1:
__k1 ^= static_cast<::cuda::std::uint32_t>(__tail[0]) << 0;
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 15);
__k1 *= __c2;
__h[0] ^= __k1;
};
}
// finalization
__h[0] ^= __size;
__h[1] ^= __size;
__h[2] ^= __size;
__h[3] ^= __size;
__h[0] += __h[1];
__h[0] += __h[2];
__h[0] += __h[3];
__h[1] += __h[0];
__h[2] += __h[0];
__h[3] += __h[0];
__h[0] = ::cuda::experimental::cuco::__fmix32(__h[0]);
__h[1] = ::cuda::experimental::cuco::__fmix32(__h[1]);
__h[2] = ::cuda::experimental::cuco::__fmix32(__h[2]);
__h[3] = ::cuda::experimental::cuco::__fmix32(__h[3]);
__h[0] += __h[1];
__h[0] += __h[2];
__h[0] += __h[3];
__h[1] += __h[0];
__h[2] += __h[0];
__h[3] += __h[0];
return ::cuda::std::bit_cast<__uint128_t>(__h);
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __uint128_t
__compute_hash_span(::cuda::std::span<const _Key> __keys) const noexcept
{
const auto __bytes = ::cuda::std::as_bytes(__keys).data();
const auto __size = __keys.size_bytes();
const auto __nchunks = __size / __chunk_size;
::cuda::std::array<::cuda::std::uint32_t, 4> __h{__seed_, __seed_, __seed_, __seed_};
// body
for (::cuda::std::remove_const_t<decltype(__nchunks)> __i = 0; __size >= __chunk_size && __i < __nchunks; ++__i)
{
::cuda::std::uint32_t __k1 = ::cuda::experimental::cuco::__load_chunk<::cuda::std::uint32_t>(__bytes, 4 * __i);
::cuda::std::uint32_t __k2 =
::cuda::experimental::cuco::__load_chunk<::cuda::std::uint32_t>(__bytes, 4 * __i + 1);
::cuda::std::uint32_t __k3 =
::cuda::experimental::cuco::__load_chunk<::cuda::std::uint32_t>(__bytes, 4 * __i + 2);
::cuda::std::uint32_t __k4 =
::cuda::experimental::cuco::__load_chunk<::cuda::std::uint32_t>(__bytes, 4 * __i + 3);
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 15);
__k1 *= __c2;
__h[0] ^= __k1;
__h[0] = ::cuda::std::rotl(__h[0], 19);
__h[0] += __h[1];
__h[0] = __h[0] * 5 + 0x561ccd1b;
__k2 *= __c2;
__k2 = ::cuda::std::rotl(__k2, 16);
__k2 *= __c3;
__h[1] ^= __k2;
__h[1] = ::cuda::std::rotl(__h[1], 17);
__h[1] += __h[2];
__h[1] = __h[1] * 5 + 0x0bcaa747;
__k3 *= __c3;
__k3 = ::cuda::std::rotl(__k3, 17);
__k3 *= __c4;
__h[2] ^= __k3;
__h[2] = ::cuda::std::rotl(__h[2], 15);
__h[2] += __h[3];
__h[2] = __h[2] * 5 + 0x96cd1c35;
__k4 *= __c4;
__k4 = ::cuda::std::rotl(__k4, 18);
__k4 *= __c1;
__h[3] ^= __k4;
__h[3] = ::cuda::std::rotl(__h[3], 13);
__h[3] += __h[0];
__h[3] = __h[3] * 5 + 0x32ac3b17;
}
// tail
::cuda::std::uint32_t __k1 = 0;
::cuda::std::uint32_t __k2 = 0;
::cuda::std::uint32_t __k3 = 0;
::cuda::std::uint32_t __k4 = 0;
const auto __tail = __bytes + __nchunks * __chunk_size;
switch (__size % __chunk_size)
{
case 15:
__k4 ^= static_cast<::cuda::std::uint32_t>(__tail[14]) << 16;
[[fallthrough]];
case 14:
__k4 ^= static_cast<::cuda::std::uint32_t>(__tail[13]) << 8;
[[fallthrough]];
case 13:
__k4 ^= static_cast<::cuda::std::uint32_t>(__tail[12]) << 0;
__k4 *= __c4;
__k4 = ::cuda::std::rotl(__k4, 18);
__k4 *= __c1;
__h[3] ^= __k4;
[[fallthrough]];
case 12:
__k3 ^= static_cast<::cuda::std::uint32_t>(__tail[11]) << 24;
[[fallthrough]];
case 11:
__k3 ^= static_cast<::cuda::std::uint32_t>(__tail[10]) << 16;
[[fallthrough]];
case 10:
__k3 ^= static_cast<::cuda::std::uint32_t>(__tail[9]) << 8;
[[fallthrough]];
case 9:
__k3 ^= static_cast<::cuda::std::uint32_t>(__tail[8]) << 0;
__k3 *= __c3;
__k3 = ::cuda::std::rotl(__k3, 17);
__k3 *= __c4;
__h[2] ^= __k3;
[[fallthrough]];
case 8:
__k2 ^= static_cast<::cuda::std::uint32_t>(__tail[7]) << 24;
[[fallthrough]];
case 7:
__k2 ^= static_cast<::cuda::std::uint32_t>(__tail[6]) << 16;
[[fallthrough]];
case 6:
__k2 ^= static_cast<::cuda::std::uint32_t>(__tail[5]) << 8;
[[fallthrough]];
case 5:
__k2 ^= static_cast<::cuda::std::uint32_t>(__tail[4]) << 0;
__k2 *= __c2;
__k2 = ::cuda::std::rotl(__k2, 16);
__k2 *= __c3;
__h[1] ^= __k2;
[[fallthrough]];
case 4:
__k1 ^= static_cast<::cuda::std::uint32_t>(__tail[3]) << 24;
[[fallthrough]];
case 3:
__k1 ^= static_cast<::cuda::std::uint32_t>(__tail[2]) << 16;
[[fallthrough]];
case 2:
__k1 ^= static_cast<::cuda::std::uint32_t>(__tail[1]) << 8;
[[fallthrough]];
case 1:
__k1 ^= static_cast<::cuda::std::uint32_t>(__tail[0]) << 0;
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 15);
__k1 *= __c2;
__h[0] ^= __k1;
};
// finalization
__h[0] ^= __size;
__h[1] ^= __size;
__h[2] ^= __size;
__h[3] ^= __size;
__h[0] += __h[1];
__h[0] += __h[2];
__h[0] += __h[3];
__h[1] += __h[0];
__h[2] += __h[0];
__h[3] += __h[0];
__h[0] = ::cuda::experimental::cuco::__fmix32(__h[0]);
__h[1] = ::cuda::experimental::cuco::__fmix32(__h[1]);
__h[2] = ::cuda::experimental::cuco::__fmix32(__h[2]);
__h[3] = ::cuda::experimental::cuco::__fmix32(__h[3]);
__h[0] += __h[1];
__h[0] += __h[2];
__h[0] += __h[3];
__h[1] += __h[0];
__h[2] += __h[0];
__h[3] += __h[0];
return ::cuda::std::bit_cast<__uint128_t>(__h);
}
private:
::cuda::std::uint32_t __seed_;
};
template <typename _Key>
struct _MurmurHash3_x64_128
{
private:
static constexpr ::cuda::std::uint64_t __c1 = 0x87c37b91114253d5ull;
static constexpr ::cuda::std::uint64_t __c2 = 0x4cf5ad432745937full;
static constexpr ::cuda::std::uint32_t __block_size = 8;
static constexpr ::cuda::std::uint32_t __chunk_size = 16;
public:
_CCCL_HOST_DEVICE_API constexpr _MurmurHash3_x64_128(::cuda::std::uint64_t __seed = 0)
: __seed_{__seed}
{}
//! @brief Returns a hash value for its argument, as a value of type `__uint128_t`.
//! @param __key The input argument to hash
//! @return The resulting hash value
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __uint128_t operator()(const _Key& __key) const noexcept
{
using _Holder = _Byte_holder<sizeof(_Key), __chunk_size, __block_size, false, ::cuda::std::uint64_t>;
return __compute_hash(::cuda::std::bit_cast<_Holder>(__key));
}
//! @brief Returns a hash value for its argument, as a value of type `__uint128_t`.
//! @tparam _Extent The extent type
//! @param __keys span of keys to hash
//! @return The resulting hash value
template <size_t _Extent>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __uint128_t
operator()(::cuda::std::span<_Key, _Extent> __keys) const noexcept
{
return __compute_hash_span(__keys);
}
private:
template <class _Holder>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __uint128_t __compute_hash(_Holder __holder) const noexcept
{
::cuda::std::array<::cuda::std::uint64_t, 2> __h{__seed_, __seed_};
const auto __size = ::cuda::std::uint64_t{sizeof(_Holder)};
if constexpr (_Holder::__num_chunks > 0)
{
::cuda::static_for<_Holder::__num_chunks>([&](auto __i) {
::cuda::std::uint64_t __k1 = __holder.__blocks[2 * __i];
::cuda::std::uint64_t __k2 = __holder.__blocks[2 * __i + 1];
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 31);
__k1 *= __c2;
__h[0] ^= __k1;
__h[0] = ::cuda::std::rotl(__h[0], 27);
__h[0] += __h[1];
__h[0] = __h[0] * 5 + 0x52dce729;
__k2 *= __c2;
__k2 = ::cuda::std::rotl(__k2, 33);
__k2 *= __c1;
__h[1] ^= __k2;
__h[1] = ::cuda::std::rotl(__h[1], 31);
__h[1] += __h[0];
__h[1] = __h[1] * 5 + 0x38495ab5;
});
}
// tail
if constexpr (_Holder::__tail_size > 0)
{
::cuda::std::uint64_t __k1 = 0;
::cuda::std::uint64_t __k2 = 0;
const auto __tail = __holder.__bytes;
switch (__size % __chunk_size)
{
case 15:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[14]) << 48;
[[fallthrough]];
case 14:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[13]) << 40;
[[fallthrough]];
case 13:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[12]) << 32;
[[fallthrough]];
case 12:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[11]) << 24;
[[fallthrough]];
case 11:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[10]) << 16;
[[fallthrough]];
case 10:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[9]) << 8;
[[fallthrough]];
case 9:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[8]) << 0;
__k2 *= __c2;
__k2 = ::cuda::std::rotl(__k2, 33);
__k2 *= __c1;
__h[1] ^= __k2;
[[fallthrough]];
case 8:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[7]) << 56;
[[fallthrough]];
case 7:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[6]) << 48;
[[fallthrough]];
case 6:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[5]) << 40;
[[fallthrough]];
case 5:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[4]) << 32;
[[fallthrough]];
case 4:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[3]) << 24;
[[fallthrough]];
case 3:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[2]) << 16;
[[fallthrough]];
case 2:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[1]) << 8;
[[fallthrough]];
case 1:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[0]) << 0;
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 31);
__k1 *= __c2;
__h[0] ^= __k1;
}
}
// finalization
__h[0] ^= __size;
__h[1] ^= __size;
__h[0] += __h[1];
__h[1] += __h[0];
__h[0] = ::cuda::experimental::cuco::__fmix64(__h[0]);
__h[1] = ::cuda::experimental::cuco::__fmix64(__h[1]);
__h[0] += __h[1];
__h[1] += __h[0];
return ::cuda::std::bit_cast<__uint128_t>(__h);
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __uint128_t
__compute_hash_span(::cuda::std::span<const _Key> __keys) const noexcept
{
const auto __bytes = ::cuda::std::as_bytes(__keys).data();
const auto __size = __keys.size_bytes();
const auto __nchunks = __size / __chunk_size;
::cuda::std::array<::cuda::std::uint64_t, 2> __h{__seed_, __seed_};
// body
for (::cuda::std::remove_const_t<decltype(__nchunks)> __i = 0; __size >= __chunk_size && __i < __nchunks; ++__i)
{
::cuda::std::uint64_t __k1 = ::cuda::experimental::cuco::__load_chunk<::cuda::std::uint64_t>(__bytes, 2 * __i);
::cuda::std::uint64_t __k2 =
::cuda::experimental::cuco::__load_chunk<::cuda::std::uint64_t>(__bytes, 2 * __i + 1);
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 31);
__k1 *= __c2;
__h[0] ^= __k1;
__h[0] = ::cuda::std::rotl(__h[0], 27);
__h[0] += __h[1];
__h[0] = __h[0] * 5 + 0x52dce729;
__k2 *= __c2;
__k2 = ::cuda::std::rotl(__k2, 33);
__k2 *= __c1;
__h[1] ^= __k2;
__h[1] = ::cuda::std::rotl(__h[1], 31);
__h[1] += __h[0];
__h[1] = __h[1] * 5 + 0x38495ab5;
}
// tail
::cuda::std::uint64_t __k1 = 0;
::cuda::std::uint64_t __k2 = 0;
const auto __tail = __bytes + __nchunks * __chunk_size;
switch (__size % __chunk_size)
{
case 15:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[14]) << 48;
[[fallthrough]];
case 14:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[13]) << 40;
[[fallthrough]];
case 13:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[12]) << 32;
[[fallthrough]];
case 12:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[11]) << 24;
[[fallthrough]];
case 11:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[10]) << 16;
[[fallthrough]];
case 10:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[9]) << 8;
[[fallthrough]];
case 9:
__k2 ^= static_cast<::cuda::std::uint64_t>(__tail[8]) << 0;
__k2 *= __c2;
__k2 = ::cuda::std::rotl(__k2, 33);
__k2 *= __c1;
__h[1] ^= __k2;
[[fallthrough]];
case 8:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[7]) << 56;
[[fallthrough]];
case 7:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[6]) << 48;
[[fallthrough]];
case 6:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[5]) << 40;
[[fallthrough]];
case 5:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[4]) << 32;
[[fallthrough]];
case 4:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[3]) << 24;
[[fallthrough]];
case 3:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[2]) << 16;
[[fallthrough]];
case 2:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[1]) << 8;
[[fallthrough]];
case 1:
__k1 ^= static_cast<::cuda::std::uint64_t>(__tail[0]) << 0;
__k1 *= __c1;
__k1 = ::cuda::std::rotl(__k1, 31);
__k1 *= __c2;
__h[0] ^= __k1;
};
// finalization
__h[0] ^= __size;
__h[1] ^= __size;
__h[0] += __h[1];
__h[1] += __h[0];
__h[0] = ::cuda::experimental::cuco::__fmix64(__h[0]);
__h[1] = ::cuda::experimental::cuco::__fmix64(__h[1]);
__h[0] += __h[1];
__h[1] += __h[0];
return ::cuda::std::bit_cast<__uint128_t>(__h);
}
private:
::cuda::std::uint64_t __seed_;
};
#endif // _CCCL_HAS_INT128()
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_MURMURHASH3_CUH

View File

@@ -0,0 +1,150 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_UTILS_CUH
#define _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_UTILS_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__memory/is_aligned.h>
#include <cuda/std/__cstring/memcpy.h>
#include <cuda/std/__memory/assume_aligned.h>
#include <cuda/std/cstddef>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief Loads a chunk of type _Tp from a byte pointer at a given index, handling alignment
//!
//! @tparam _Tp The type of the chunk to load (must be 4 or 8 bytes)
//! @tparam _Extent The index type
//! @param __bytes Pointer to the byte array
//! @param __index The index of the chunk to load
//! @return The loaded chunk of type _Tp
template <typename _Tp, typename _Extent>
[[nodiscard]] _CCCL_HOST_DEVICE_API _Tp __load_chunk(::cuda::std::byte const* const __bytes, _Extent __index) noexcept
{
static_assert(sizeof(_Tp) == 4 || sizeof(_Tp) == 8, "__load_chunk must be used with types of size 4 or 8 bytes");
const auto __ptr = __bytes + __index * sizeof(_Tp);
_Tp __chunk;
if constexpr (alignof(_Tp) == 8)
{
if (::cuda::is_aligned(__ptr, 8))
{
::cuda::std::memcpy(&__chunk, ::cuda::std::assume_aligned<8>(__ptr), sizeof(_Tp));
return __chunk;
}
}
if (::cuda::is_aligned(__ptr, 4))
{
::cuda::std::memcpy(&__chunk, ::cuda::std::assume_aligned<4>(__ptr), sizeof(_Tp));
}
else if (::cuda::is_aligned(__ptr, 2))
{
::cuda::std::memcpy(&__chunk, ::cuda::std::assume_aligned<2>(__ptr), sizeof(_Tp));
}
else
{
::cuda::std::memcpy(&__chunk, __ptr, sizeof(_Tp));
}
return __chunk;
}
//! @brief Type erased holder of all the bytes
//!
//! @tparam _KeySize The size of the key in bytes
//! @tparam _ChunkSize The size of a chunk in bytes
//! @tparam _BlockSize The size of a block in bytes (same as sizeof(_BlockT))
//! @tparam _UseTailBlock Whether to use a tail block for the last bytes
//! @tparam _BlockT The type of the block
//! @tparam _HasBlocksOrChunks Whether the key size is larger than the chunk size or block size
//! @tparam _HasTail Whether the key size is larger than the block size
//!
//! @note _UseTailBlock is true for xxhash and false for murmurhash, as xxhash consider's tail as blocks for the last
//! bytes, where as murmurhash considers the tail as a bytes
template <size_t _KeySize,
size_t _ChunkSize,
size_t _BlockSize,
bool _UseTailBlock,
typename _BlockT,
bool _HasBlocksOrChunks = _UseTailBlock ? (_KeySize >= _BlockSize) : (_KeySize >= _ChunkSize),
bool _HasTail = _UseTailBlock ? ((_KeySize % _BlockSize) != 0) : ((_KeySize % _ChunkSize) != 0)>
struct _Byte_holder
{
//! The number of trailing bytes that do not fit into a _BlockT
static constexpr size_t __tail_size = _UseTailBlock ? _KeySize % _BlockSize : _KeySize % _ChunkSize;
//! The number of `_ChunkSize` chunks
static constexpr size_t __num_chunks = _KeySize / _ChunkSize;
//! The number of `_BlockSize` blocks in a `_ChunkSize` chunk
static constexpr size_t __blocks_per_chunk = _ChunkSize / _BlockSize;
//! The number of `_BlockSize` blocks
static constexpr size_t __num_blocks = _UseTailBlock ? _KeySize / _BlockSize : __num_chunks * __blocks_per_chunk;
_BlockT __blocks[__num_blocks];
::cuda::std::byte __bytes[__tail_size];
};
//! @brief Type erased holder of small types < _BlockSize
template <size_t _KeySize, size_t _ChunkSize, size_t _BlockSize, bool _UseTailBlock, typename _BlockT>
struct _Byte_holder<_KeySize, _ChunkSize, _BlockSize, _UseTailBlock, _BlockT, false, true>
{
//! The number of trailing bytes that do not fit into a _BlockT
static constexpr size_t __tail_size = _UseTailBlock ? _KeySize % _BlockSize : _KeySize % _ChunkSize;
//! The number of `_ChunkSize` chunks
static constexpr size_t __num_chunks = _KeySize / _ChunkSize;
//! The number of `_BlockSize` blocks in a `_ChunkSize` chunk
static constexpr size_t __blocks_per_chunk = _ChunkSize / _BlockSize;
//! The number of `_BlockSize` blocks in a `_ChunkSize` chunk
static constexpr size_t __num_blocks = _UseTailBlock ? _KeySize / _BlockSize : __num_chunks * __blocks_per_chunk;
::cuda::std::byte __bytes[__tail_size];
};
//! @brief Type erased holder of types without trailing bytes
template <size_t _KeySize, size_t _ChunkSize, size_t _BlockSize, bool _UseTailBlock, typename _BlockT>
struct _Byte_holder<_KeySize, _ChunkSize, _BlockSize, _UseTailBlock, _BlockT, true, false>
{
//! The number of trailing bytes that do not fit into a _BlockT
static constexpr size_t __tail_size = _UseTailBlock ? _KeySize % _BlockSize : _KeySize % _ChunkSize;
//! The number of `_ChunkSize` chunks
static constexpr size_t __num_chunks = _KeySize / _ChunkSize;
//! The number of `_BlockSize` blocks in a `_ChunkSize` chunk
static constexpr size_t __blocks_per_chunk = _ChunkSize / _BlockSize;
//! The number of `_BlockSize` blocks
static constexpr size_t __num_blocks = _UseTailBlock ? _KeySize / _BlockSize : __num_chunks * __blocks_per_chunk;
_BlockT __blocks[__num_blocks];
};
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_UTILS_CUH

View File

@@ -0,0 +1,426 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
/*
* `_XXHash_32` and `_XXHash_64` implementation from
* https://github.com/Cyan4973/xxHash
* -----------------------------------------------------------------------------
* xxHash - Extremely Fast Hash algorithm
* Header File
* Copyright (C) 2012-2021 Yann Collet
*
* BSD 2-Clause License (https://www.opensource.org/licenses/bsd-license.php)
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are
* met:
*
* * Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* * Redistributions in binary form must reproduce the above
* copyright notice, this list of conditions and the following disclaimer
* in the documentation and/or other materials provided with the
* distribution.
*
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
* "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
* LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
* A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
* OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
* SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
* LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
* DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
* THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*/
#ifndef _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_XXHASH_CUH
#define _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_XXHASH_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__utility/static_for.h>
#include <cuda/std/__bit/bit_cast.h>
#include <cuda/std/__bit/rotate.h>
#include <cuda/std/array>
#include <cuda/std/cstddef>
#include <cuda/std/cstdint>
#include <cuda/std/span>
#include <cuda/experimental/__cuco/detail/hash_functions/utils.cuh>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief A `_XXHash_32` hash function to hash the given argument on host and device.
//!
//! @tparam Key The type of the values to hash
template <typename _Key>
struct _XXHash_32
{
private:
static constexpr ::cuda::std::uint32_t __prime1 = 0x9e3779b1u;
static constexpr ::cuda::std::uint32_t __prime2 = 0x85ebca77u;
static constexpr ::cuda::std::uint32_t __prime3 = 0xc2b2ae3du;
static constexpr ::cuda::std::uint32_t __prime4 = 0x27d4eb2fu;
static constexpr ::cuda::std::uint32_t __prime5 = 0x165667b1u;
static constexpr ::cuda::std::uint32_t __block_size = 4;
static constexpr ::cuda::std::uint32_t __chunk_size = 16;
public:
//! @brief Constructs a XXH32 hash function with the given `seed`.
//! @param seed A custom number to randomize the resulting hash value
_CCCL_HOST_DEVICE_API constexpr _XXHash_32(::cuda::std::uint32_t __seed = 0)
: __seed_{__seed}
{}
//! @brief Returns a hash value for its argument, as a value of type `::cuda::std::uint32_t`.
//! @param __key The input argument to hash
//! @return The resulting hash value for `__key`
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint32_t operator()(const _Key& __key) const noexcept
{
using _Holder = _Byte_holder<sizeof(_Key), __chunk_size, __block_size, true, ::cuda::std::uint32_t>;
// explicit copy to avoid emitting a bunch of LDG.8 instructions
const _Key __copy{__key};
return __compute_hash(::cuda::std::bit_cast<_Holder>(__copy));
}
//! @brief Returns a hash value for its argument, as a value of type `::cuda::std::uint32_t`.
//! @tparam _Extent The extent type
//! @param __keys span of keys to hash
//! @return The resulting hash value
template <size_t _Extent>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint32_t
operator()(::cuda::std::span<_Key, _Extent> __keys) const noexcept
{
return __compute_hash_span(__keys);
}
private:
//! @brief Returns a hash value for its argument, as a value of type `::cuda::std::uint32_t`.
//!
//! @tparam _Extent The extent type
//! @param __holder The input argument to hash in form of a byte holder
//! @return The resulting hash value
template <class _Holder>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint32_t __compute_hash(_Holder __holder) const noexcept
{
::cuda::std::uint32_t __offset = 0;
::cuda::std::uint32_t __h32 = {};
// process data in 16-byte chunks
if constexpr (_Holder::__num_chunks > 0)
{
::cuda::std::array<::cuda::std::uint32_t, 4> __v;
__v[0] = __seed_ + __prime1 + __prime2;
__v[1] = __seed_ + __prime2;
__v[2] = __seed_;
__v[3] = __seed_ - __prime1;
for (::cuda::std::uint32_t __i = 0; __i < _Holder::__num_chunks; ++__i)
{
::cuda::static_for<4>([&](auto i) {
__v[i] += __holder.__blocks[__offset++] * __prime2;
__v[i] = ::cuda::std::rotl(__v[i], 13);
__v[i] *= __prime1;
});
}
__h32 = ::cuda::std::rotl(__v[0], 1) + ::cuda::std::rotl(__v[1], 7) + ::cuda::std::rotl(__v[2], 12)
+ ::cuda::std::rotl(__v[3], 18);
}
else
{
__h32 = __seed_ + __prime5;
}
__h32 += ::cuda::std::uint32_t{sizeof(_Holder)};
// remaining data can be processed in 4-byte chunks
if constexpr (_Holder::__num_blocks % __chunk_size > 0)
{
for (; __offset < _Holder::__num_blocks; ++__offset)
{
__h32 += __holder.__blocks[__offset] * __prime3;
__h32 = ::cuda::std::rotl(__h32, 17) * __prime4;
}
}
// the following loop is only needed if the size of the key is not a multiple of the block size
if constexpr (_Holder::__tail_size > 0)
{
for (::cuda::std::uint32_t __i = 0; __i < _Holder::__tail_size; ++__i)
{
__h32 += (static_cast<::cuda::std::uint32_t>(__holder.__bytes[__i])) * __prime5;
__h32 = ::cuda::std::rotl(__h32, 11) * __prime1;
}
}
return __finalize(__h32);
}
//! @brief Returns a hash value for its argument, as a value of type `::cuda::std::uint32_t`.
//!
//! @tparam _Extent The extent type
//! @param __holder The input argument to hash in form of a span
//! @return The resulting hash value
[[nodiscard]] _CCCL_HOST_DEVICE_API ::cuda::std::uint32_t
__compute_hash_span(::cuda::std::span<_Key> __keys) const noexcept
{
auto __bytes = ::cuda::std::as_bytes(__keys).data();
const auto __size = __keys.size_bytes();
::cuda::std::uint32_t __offset = 0;
::cuda::std::uint32_t __h32 = {};
// data can be processed in 16-byte chunks
if (__size >= 16)
{
const auto __limit = __size - 16;
::cuda::std::array<::cuda::std::uint32_t, 4> __v;
__v[0] = __seed_ + __prime1 + __prime2;
__v[1] = __seed_ + __prime2;
__v[2] = __seed_;
__v[3] = __seed_ - __prime1;
for (; __offset <= __limit; __offset += 16)
{
// pipeline 4*4byte computations
const auto __pipeline_offset = __offset / 4;
::cuda::static_for<4>([&](auto i) {
__v[i] += ::cuda::experimental::cuco::__load_chunk<::cuda::std::uint32_t>(__bytes, __pipeline_offset + i)
* __prime2;
__v[i] = ::cuda::std::rotl(__v[i], 13);
__v[i] *= __prime1;
});
}
__h32 = ::cuda::std::rotl(__v[0], 1) + ::cuda::std::rotl(__v[1], 7) + ::cuda::std::rotl(__v[2], 12)
+ ::cuda::std::rotl(__v[3], 18);
}
else
{
__h32 = __seed_ + __prime5;
}
__h32 += __size;
// remaining data can be processed in 4-byte chunks
if ((__size % 16) >= 4)
{
_CCCL_PRAGMA_UNROLL(4)
for (; __offset <= __size - 4; __offset += 4)
{
__h32 += ::cuda::experimental::cuco::__load_chunk<::cuda::std::uint32_t>(__bytes, __offset / 4) * __prime3;
__h32 = ::cuda::std::rotl(__h32, 17) * __prime4;
}
}
// the following loop is only needed if the size of the key is not a multiple of the block size
if (__size % 4)
{
while (__offset < __size)
{
__h32 += (::cuda::std::to_integer<::cuda::std::uint32_t>(__bytes[__offset]) & 255) * __prime5;
__h32 = ::cuda::std::rotl(__h32, 11) * __prime1;
++__offset;
}
}
return __finalize(__h32);
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint32_t
__finalize(::cuda::std::uint32_t __h) const noexcept
{
__h ^= __h >> 15;
__h *= __prime2;
__h ^= __h >> 13;
__h *= __prime3;
__h ^= __h >> 16;
return __h;
}
::cuda::std::uint32_t __seed_;
};
//! @brief A `XXHash_64` hash function to hash the given argument on host and device.
//!
//! @tparam _Key The type of the values to hash
template <typename _Key>
struct _XXHash_64
{
private:
static constexpr ::cuda::std::uint64_t __prime1 = 11400714785074694791ull;
static constexpr ::cuda::std::uint64_t __prime2 = 14029467366897019727ull;
static constexpr ::cuda::std::uint64_t __prime3 = 1609587929392839161ull;
static constexpr ::cuda::std::uint64_t __prime4 = 9650029242287828579ull;
static constexpr ::cuda::std::uint64_t __prime5 = 2870177450012600261ull;
public:
//! @brief Constructs a XXH64 hash function with the given `seed`.
//!
//! @param seed A custom number to randomize the resulting hash value
_CCCL_HOST_DEVICE_API constexpr _XXHash_64(::cuda::std::uint64_t __seed = 0)
: __seed_{__seed}
{}
//! @brief Returns a hash value for its argument, as a value of type `result_type`.
//!
//! @param _Key The input argument to hash
//! @return The resulting hash value for `key`
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint64_t operator()(const _Key& __key) const noexcept
{
if constexpr (sizeof(_Key) <= 16)
{
const _Key __copy{__key};
return __compute_hash_span(::cuda::std::span<const _Key, 1>{&__copy, 1});
}
else
{
return __compute_hash_span(::cuda::std::span<const _Key, 1>{&__key, 1});
}
}
//! @brief Returns a hash value for its argument, as a value of type `::cuda::std::uint64_t`.
//!
//! @tparam _Extent The extent type
//! @param __keys span of keys to hash
//! @return The resulting hash value
template <size_t _Extent>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint64_t
operator()(::cuda::std::span<_Key, _Extent> __keys) const noexcept
{
return __compute_hash_span(__keys);
}
private:
//! @brief Returns a hash value for its argument, as a value of type `::cuda::std::uint64_t`.
//!
//! @tparam _Extent The extent type
//! @param __keys span of keys to hash
//! @return The resulting hash value
[[nodiscard]] _CCCL_HOST_DEVICE_API ::cuda::std::uint64_t
__compute_hash_span(::cuda::std::span<const _Key> __keys) const noexcept
{
auto __bytes = ::cuda::std::as_bytes(__keys).data();
const auto __size = __keys.size_bytes();
size_t __offset = 0;
::cuda::std::uint64_t __h64 = {};
// process data in 32-byte chunks
if (__size >= 32)
{
const auto __limit = __size - 32;
::cuda::std::array<::cuda::std::uint64_t, 4> __v;
__v[0] = __seed_ + __prime1 + __prime2;
__v[1] = __seed_ + __prime2;
__v[2] = __seed_;
__v[3] = __seed_ - __prime1;
for (; __offset <= __limit; __offset += 32)
{
// pipeline 4*8byte computations
const auto __pipeline_offset = __offset / 8;
::cuda::static_for<4>([&](auto i) {
__v[i] += ::cuda::experimental::cuco::__load_chunk<::cuda::std::uint64_t>(__bytes, __pipeline_offset + i)
* __prime2;
__v[i] = ::cuda::std::rotl(__v[i], 31);
__v[i] *= __prime1;
});
}
__h64 = ::cuda::std::rotl(__v[0], 1) + ::cuda::std::rotl(__v[1], 7) + ::cuda::std::rotl(__v[2], 12)
+ ::cuda::std::rotl(__v[3], 18);
::cuda::static_for<4>([&](auto i) {
__v[i] *= __prime2;
__v[i] = ::cuda::std::rotl(__v[i], 31);
__v[i] *= __prime1;
__h64 ^= __v[i];
__h64 = __h64 * __prime1 + __prime4;
});
}
else
{
__h64 = __seed_ + __prime5;
}
__h64 += __size;
// remaining data can be processed in 8-byte chunks
if ((__size % 32) >= 8)
{
_CCCL_PRAGMA_UNROLL(4)
for (; __offset <= __size - 8; __offset += 8)
{
::cuda::std::uint64_t __k1 =
::cuda::experimental::cuco::__load_chunk<::cuda::std::uint64_t>(__bytes, __offset / 8) * __prime2;
__k1 = ::cuda::std::rotl(__k1, 31) * __prime1;
__h64 ^= __k1;
__h64 = ::cuda::std::rotl(__h64, 27) * __prime1 + __prime4;
}
}
// remaining data can be processed in 4-byte chunks
if ((__size % 8) >= 4)
{
for (; __offset <= __size - 4; __offset += 4)
{
__h64 ^= (::cuda::experimental::cuco::__load_chunk<::cuda::std::uint32_t>(__bytes, __offset / 4)) * __prime1;
__h64 = ::cuda::std::rotl(__h64, 23) * __prime2 + __prime3;
}
}
// the following loop is only needed if the size of the key is not a multiple of a previous
// block size
if (__size % 4)
{
while (__offset < __size)
{
__h64 ^= (::cuda::std::to_integer<::cuda::std::uint32_t>(__bytes[__offset])) * __prime5;
__h64 = ::cuda::std::rotl(__h64, 11) * __prime1;
++__offset;
}
}
return __finalize(__h64);
}
// avalanche helper
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint64_t __finalize(std::uint64_t __h) const noexcept
{
__h ^= __h >> 33;
__h *= __prime2;
__h ^= __h >> 29;
__h *= __prime3;
__h ^= __h >> 32;
return __h;
}
::cuda::std::uint64_t __seed_;
};
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_HASH_FUNCTIONS_XXHASH_CUH

View File

@@ -0,0 +1,126 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_HYPERLOGLOG_DEFAULT_POLICY_CUH
#define _CUDAX___CUCO_DETAIL_HYPERLOGLOG_DEFAULT_POLICY_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__bit/countl.h>
#include <cuda/std/__limits/numeric_limits.h>
#include <cuda/std/__type_traits/is_unsigned.h>
#include <cuda/std/__utility/declval.h>
#include <cuda/std/cstdint>
#include <cuda/experimental/__cuco/detail/hyperloglog/finalizer.cuh>
#include <cuda/experimental/__cuco/hash_functions.cuh>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief Default policy for `cuda::experimental::cuco::hyperloglog`.
//!
//! Bundles the three customization points of the HLL pipeline -- hash function, bit slicing,
//! and finalizer -- into a single policy. This default reproduces the behaviour shipped by
//! `cuCollections::hyperloglog`: MSB-indexed register selection, padded leading-zero count for
//! rho, and HyperLogLog++ bias correction. Custom policies (e.g. for binary interop with
//! third-party sketch libraries) can be supplied via the `_Policy` template parameter on
//! `hyperloglog` and `hyperloglog_ref`.
//!
//! @tparam _Key The item type the sketch counts.
//! @tparam _Algo The hash algorithm. Defaults to xxhash_64.
template <class _Key, hash_algorithm _Algo = hash_algorithm::xxhash_64>
struct default_hll_policy
{
using hasher = hash<_Key, _Algo>;
using hash_result_type = decltype(::cuda::std::declval<hasher>()(::cuda::std::declval<_Key>()));
using register_type = ::cuda::std::int32_t;
static_assert(::cuda::std::is_unsigned_v<hash_result_type>, "HyperLogLog requires an unsigned hash value type");
static_assert(::cuda::std::numeric_limits<hash_result_type>::digits == 32
|| ::cuda::std::numeric_limits<hash_result_type>::digits == 64,
"HyperLogLog requires a 32-bit or 64-bit hash value type");
hasher hasher_{};
//! @brief Returns the underlying hash functor.
//!
//! @return The hash functor.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr hasher hash_function() const noexcept
{
return hasher_;
}
//! @brief Hashes an item.
//!
//! @param[in] __k The item to hash.
//! @return The hash value of `__k`.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr hash_result_type hash(const _Key& __k) const noexcept
{
return hasher_(__k);
}
//! @brief Extracts the register index from the hash.
//!
//! @note Index is taken from the high `__precision` bits of the hash, matching Apache Spark's
//! HyperLogLog++ convention.
//!
//! @param[in] __h The hash value.
//! @param[in] __precision The HLL precision parameter.
//! @return The register index in `[0, 2^__precision)`.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint32_t
register_index(hash_result_type __h, ::cuda::std::int32_t __precision) const noexcept
{
constexpr auto __hash_bits = ::cuda::std::numeric_limits<hash_result_type>::digits;
return static_cast<::cuda::std::uint32_t>(__h >> (__hash_bits - __precision));
}
//! @brief Computes rho (1 + leading zeros of the rho source) from the hash.
//!
//! @note A one-bit padding bounds the leading-zero count at `hash_bits - __precision`,
//! preventing rho overflow when the low `hash_bits - __precision` bits of the hash are zero.
//!
//! @param[in] __h The hash value.
//! @param[in] __precision The HLL precision parameter.
//! @return rho, in `[1, hash_bits - __precision + 1]`.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint8_t
register_value(hash_result_type __h, ::cuda::std::int32_t __precision) const noexcept
{
const auto __w_padding = hash_result_type{1} << static_cast<hash_result_type>(__precision - 1);
return static_cast<::cuda::std::uint8_t>(::cuda::std::countl_zero((__h << __precision) | __w_padding) + 1);
}
//! @brief Finalizes the GPU reduction into a cardinality estimate using the HyperLogLog++
//! bias-corrected estimator.
//!
//! @param[in] __z Sum of `2^-register[i]` across all registers.
//! @param[in] __v Count of zero registers.
//! @param[in] __precision HLL precision parameter.
//! @return The bias-corrected cardinality estimate.
[[nodiscard]] static _CCCL_HOST_DEVICE_API constexpr double
finalize(double __z, ::cuda::std::int32_t __v, ::cuda::std::int32_t __precision) noexcept
{
return __hyperloglog_ns::hllpp_finalizer{__precision}(__z, __v);
}
};
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_HYPERLOGLOG_DEFAULT_POLICY_CUH

View File

@@ -0,0 +1,189 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_HYPERLOGLOG_FINALIZER_CUH
#define _CUDAX___CUCO_DETAIL_HYPERLOGLOG_FINALIZER_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__utility/in_range.h>
#include <cuda/std/__algorithm/max.h>
#include <cuda/std/__algorithm/min.h>
#include <cuda/std/__cmath/logarithms.h>
#include <cuda/std/__numeric/midpoint.h>
#include <cuda/std/cstdint>
#include <cuda/experimental/__cuco/detail/hyperloglog/tuning.cuh>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::__hyperloglog_ns
{
//! @brief Estimate correction algorithm based on HyperLogLog++.
//!
//! @note Variable names correspond to the definitions given in the HLL++ paper:
//! https://static.googleusercontent.com/media/research.google.com/de//pubs/archive/40671.pdf
//! @note Precision must be >= 4.
//!
class hllpp_finalizer
{
// Note: Most of the types in this implementation are explicit instead of relying on `auto` to
// avoid confusion with the reference implementation.
public:
//! @brief Constructs an HLL finalizer object.
//!
//! @param __precision_ HLL precision parameter
_CCCL_HOST_DEVICE_API constexpr hllpp_finalizer(::cuda::std::int32_t __precision_) noexcept
: __precision{__precision_}
, __m{static_cast<::cuda::std::int32_t>(1u << __precision_)}
{
_CCCL_ASSERT(::cuda::in_range(__precision_, 4, 18), "Precision must be between 4 and 18");
}
//! @brief Compute the bias-corrected cardinality estimate.
//!
//! @param __z Geometric mean of registers
//! @param __v Number of 0 registers
//!
//! @return Bias-corrected cardinality estimate
[[nodiscard]] _CCCL_HOST_DEVICE_API double operator()(double __z, ::cuda::std::int32_t __v) const noexcept
{
double __e = __alpha_mm() / __z;
if (__v > 0)
{
// Use linear counting for small cardinality estimates.
const double __h = __m * ::cuda::std::log(static_cast<double>(__m) / __v);
// The threshold `2.5 * m` is from the original HLL algorithm.
if (__e <= 2.5 * __m)
{
return __h;
}
if (__precision < 19)
{
__e = (__h <= __hyperloglog_ns::__threshold(__precision)) ? __h : __bias_corrected_estimate(__e);
}
}
else
{
// HLL++ is defined only when p < 19, otherwise we need to fallback to HLL.
if (__precision < 19)
{
__e = __bias_corrected_estimate(__e);
}
}
return __e;
}
private:
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr double __alpha_mm() const noexcept
{
const auto __m2 = static_cast<double>(__m) * __m;
switch (__m)
{
case 16:
return 0.673 * __m2;
case 32:
return 0.697 * __m2;
case 64:
return 0.709 * __m2;
default:
return (0.7213 / (1.0 + 1.079 / __m)) * __m2;
}
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr double __bias_corrected_estimate(double __e) const noexcept
{
return (__e < 5.0 * __m) ? __e - __bias(__e) : __e;
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr double __bias(double __e) const noexcept
{
const auto __anchor_index = __interpolation_anchor_index(__e);
const auto __n = static_cast<::cuda::std::int32_t>(__hyperloglog_ns::__raw_estimate_data_size(__precision));
auto __low = ::cuda::std::max(__anchor_index - __k + 1, ::cuda::std::int32_t{0});
auto __high = ::cuda::std::min(__low + __k, __n);
// Keep moving bounds as long as the (exclusive) high bound is closer to the estimate than
// the lower (inclusive) bound.
while (__high < __n && __distance(__e, __high) < __distance(__e, __low))
{
__low += 1;
__high += 1;
}
const auto __biases = __hyperloglog_ns::__bias_data(__precision);
double __bias_sum = 0.0;
for (::cuda::std::int32_t __i = __low; __i < __high; ++__i)
{
__bias_sum += __biases[__i];
}
return __bias_sum / (__high - __low);
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr double __distance(double __e, ::cuda::std::int32_t __i) const noexcept
{
const auto __diff = __e - __hyperloglog_ns::__raw_estimate_data(__precision)[__i];
return __diff * __diff;
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::int32_t
__interpolation_anchor_index(double __e) const noexcept
{
const auto __estimates = __hyperloglog_ns::__raw_estimate_data(__precision);
const auto __n = static_cast<::cuda::std::int32_t>(__hyperloglog_ns::__raw_estimate_data_size(__precision));
::cuda::std::int32_t __left = 0;
::cuda::std::int32_t __right = __n - 1;
while (__left <= __right)
{
const ::cuda::std::int32_t __mid = ::cuda::std::midpoint(__left, __right);
if (__estimates[__mid] < __e)
{
__left = __mid + 1;
}
else if (__estimates[__mid] > __e)
{
__right = __mid - 1;
}
else
{
// Exact match found, no need to look further
return __mid;
}
}
// At this point, '__left' is the binary-search insertion point. Spark uses the insertion
// point as the anchor index when the exact estimate is not present in the table.
return __left;
}
static constexpr ::cuda::std::int32_t __k = 6; ///< Number of interpolation points to consider
::cuda::std::int32_t __precision; ///< HLL precision parameter
::cuda::std::int32_t __m; ///< Number of registers (2^precision)
};
} // namespace cuda::experimental::cuco::__hyperloglog_ns
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_HYPERLOGLOG_FINALIZER_CUH

View File

@@ -0,0 +1,618 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_HYPERLOGLOG_IMPL_CUH
#define _CUDAX___CUCO_DETAIL_HYPERLOGLOG_IMPL_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__container/buffer.h>
#include <cuda/__memory/is_aligned.h>
#include <cuda/__memory_resource/legacy_pinned_memory_resource.h>
#include <cuda/__runtime/api_wrapper.h>
#include <cuda/__stream/stream_ref.h>
#include <cuda/__utility/in_range.h>
#include <cuda/atomic>
#include <cuda/std/__algorithm/max.h>
#include <cuda/std/__bit/countr.h>
#include <cuda/std/__bit/integral.h>
#include <cuda/std/__cmath/rounding_functions.h>
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/__host_stdlib/stdexcept>
#include <cuda/std/__iterator/concepts.h>
#include <cuda/std/__memory/pointer_traits.h>
#include <cuda/std/span>
#include <cuda/experimental/__cuco/detail/hyperloglog/finalizer.cuh>
#include <cuda/experimental/__cuco/detail/hyperloglog/kernels.cuh>
#include <cuda/experimental/__cuco/detail/utility/strong_type.cuh>
#include <cuda/experimental/__cuco/hash_functions.cuh>
#include <cooperative_groups.h>
#include <cooperative_groups/reduce.h>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
CUDAX_CUCO_DEFINE_STRONG_TYPE(__sketch_size_kb_t, double);
CUDAX_CUCO_DEFINE_STRONG_TYPE(__standard_deviation_t, double);
CUDAX_CUCO_DEFINE_STRONG_TYPE(__precision_t, ::cuda::std::int32_t);
//! @brief A GPU-accelerated utility for approximating the number of distinct items in a multiset.
//!
//! @note This class implements the HyperLogLog/HyperLogLog++ algorithm:
//! https://static.googleusercontent.com/media/research.google.com/de//pubs/archive/40671.pdf.
//!
//! @tparam _Tp Type of items to count
//! @tparam _Scope The scope in which operations will be performed by individual threads
//! @tparam _Policy Policy bundling hash function, bit-slicing rule, and finalizer
template <class _Tp, ::cuda::thread_scope _Scope, class _Policy>
class __hyperloglog_impl
{
using __fp_type = double; ///< Floating point type used for reduction
public:
using __value_type = _Tp; ///< Type of items to count
using __policy_type = _Policy; ///< Policy type
using __hasher = typename _Policy::hasher; ///< Hash function type
using __register_type = typename _Policy::register_type; ///< HLL register type
private:
_Policy __policy; ///< Policy used to hash items, slice the hash, and finalize the estimate
::cuda::std::int32_t __precision; ///< HLL precision parameter
::cuda::std::span<__register_type> __sketch; ///< HLL sketch storage
template <class _Tp_, ::cuda::thread_scope _Scope_, class _Policy_>
friend struct __hyperloglog_impl;
public:
static constexpr auto __thread_scope = _Scope; ///< CUDA thread scope
template <::cuda::thread_scope _NewScope>
using __rebind_scope = __hyperloglog_impl<_Tp, _NewScope, _Policy>; ///< Ref type with different thread scope
//! @brief Constructs a non-owning `__hyperloglog_impl` object.
//!
//! @throw If sketch size < 0.0625KB or 64B or standard deviation > 0.2765. Throws if called from
//! host; __trap() if called from device.
//! @throw If sketch size implies precision outside [4, 18]. Throws if called from host; __trap() if
//! called from device.
//! @throw If sketch storage has insufficient alignment. Throws if called from host; __trap() if called from device.
//!
//! @param __sketch_span Reference to sketch storage
//! @param __policy The policy used to hash items and finalize the estimate
_CCCL_HOST_DEVICE_API constexpr __hyperloglog_impl(::cuda::std::span<::cuda::std::byte> __sketch_span,
const _Policy& __policy)
: __policy{__policy}
, __precision{::cuda::std::countr_zero(
__sketch_bytes(static_cast<::cuda::experimental::cuco::__sketch_size_kb_t>(__sketch_span.size() / 1024.0))
/ sizeof(__register_type))}
, __sketch{reinterpret_cast<int*>(__sketch_span.data()), __sketch_bytes() / sizeof(__register_type)}
// MSVC fails with __register_type*, use int* instead
{
constexpr ::cuda::std::size_t __minimum_sketch_bytes = sizeof(__register_type) * (1ull << 4);
if (__sketch_span.size() < __minimum_sketch_bytes)
{
_CCCL_THROW(::std::invalid_argument, "Minimum required sketch size is 0.0625KB or 64B");
}
if (!::cuda::is_aligned(__sketch_span.data(), __sketch_alignment()))
{
_CCCL_THROW(::std::invalid_argument, "Sketch storage has insufficient alignment");
}
if (!::cuda::in_range(__precision, 4, 18))
{
_CCCL_THROW(::std::invalid_argument, "Minimum required sketch size is 0.0625KB or 64B");
}
}
//! @brief Resets the estimator, i.e., clears the current count estimate.
//!
//! @tparam _CG CUDA Cooperative Group type
//!
//! @param __group CUDA Cooperative group this operation is executed in
template <class _CG>
_CCCL_DEVICE_API constexpr void __clear(_CG __group) noexcept
{
for (int __i = __group.thread_rank(); __i < __sketch.size(); __i += __group.size())
{
__sketch[__i] = 0;
}
}
//! @brief Resets the estimator, i.e., clears the current count estimate.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `__clear_async`.
//!
//! @param __stream CUDA stream this operation is executed in
_CCCL_HOST_API constexpr void __clear(::cuda::stream_ref __stream)
{
__clear_async(__stream);
__stream.sync();
}
//! @brief Asynchronously resets the estimator, i.e., clears the current count estimate.
//!
//! @param __stream CUDA stream this operation is executed in
_CCCL_HOST_API constexpr void __clear_async(::cuda::stream_ref __stream)
{
constexpr auto __block_size = 1024;
::cuda::experimental::cuco::__hyperloglog_ns::__clear<<<1, __block_size, 0, __stream.get()>>>(*this);
}
//! @brief Adds an item to the estimator.
//!
//! @note Hash, register index, and rho are determined by the active policy.
//!
//! @param __item The item to be counted
_CCCL_DEVICE_API constexpr void __add(const _Tp& __item) noexcept
{
const auto __h = __policy.hash(__item);
__update_max(__policy.register_index(__h, __precision), __policy.register_value(__h, __precision));
}
//! @brief Asynchronously adds to be counted items to the estimator.
//!
//! @tparam _InputIt Device accessible random access input iterator where
//! <tt>std::is_convertible<std::iterator_traits<_InputIt>::value_type,
//! _Tp></tt> is `true`
//!
//! @param __first Beginning of the sequence of items
//! @param __last End of the sequence of items
//! @param __stream CUDA stream this operation is executed in
template <class _InputIt>
_CCCL_HOST_API constexpr void __add_async(_InputIt __first, _InputIt __last, ::cuda::stream_ref __stream)
{
const auto __num_items = ::cuda::std::distance(__first, __last);
if (__num_items == 0)
{
return;
}
int __grid_size = 0;
int __block_size = 0;
const int __shmem_bytes = __sketch_bytes();
const void* __kernel = nullptr;
// In case the input iterator represents a contiguous memory segment we can employ efficient
// vectorized loads
if constexpr (::cuda::std::contiguous_iterator<_InputIt>)
{
const auto __ptr = ::cuda::std::to_address(__first);
constexpr auto __max_vector_bytes = 32;
const auto __alignment =
1u << ::cuda::std::countr_zero(reinterpret_cast<::cuda::std::uintptr_t>(__ptr) | __max_vector_bytes);
const auto __vector_size = __alignment / sizeof(__value_type);
switch (__vector_size)
{
using ::cuda::experimental::cuco::__hyperloglog_ns::__add_shmem_vectorized;
case 2:
__kernel = reinterpret_cast<const void*>(__add_shmem_vectorized<2, __hyperloglog_impl>);
break;
case 4:
__kernel = reinterpret_cast<const void*>(__add_shmem_vectorized<4, __hyperloglog_impl>);
break;
case 8:
__kernel = reinterpret_cast<const void*>(__add_shmem_vectorized<8, __hyperloglog_impl>);
break;
case 16:
__kernel = reinterpret_cast<const void*>(__add_shmem_vectorized<16, __hyperloglog_impl>);
break;
};
}
if (__kernel != nullptr && __try_reserve_shmem(__kernel, __shmem_bytes))
{
if constexpr (::cuda::std::contiguous_iterator<_InputIt>)
{
// We make use of the occupancy calculator to get the minimum number of blocks which still
// saturates the GPU. This reduces the shmem initialization overhead and atomic contention
// on the final register array during the merge phase.
_CCCL_TRY_CUDA_API(
::cudaOccupancyMaxPotentialBlockSize,
"cudaOccupancyMaxPotentialBlockSize failed",
&__grid_size,
&__block_size,
__kernel,
__shmem_bytes);
const auto __ptr = ::cuda::std::to_address(__first);
void* __kernel_args[] = {const_cast<void*>(reinterpret_cast<const void*>(&__ptr)),
const_cast<void*>(reinterpret_cast<const void*>(&__num_items)),
reinterpret_cast<void*>(this)};
_CCCL_TRY_CUDA_API(
::cudaLaunchKernel,
"cudaLaunchKernel failed",
__kernel,
__grid_size,
__block_size,
__kernel_args,
__shmem_bytes,
__stream.get());
}
}
else
{
__kernel = reinterpret_cast<const void*>(
::cuda::experimental::cuco::__hyperloglog_ns::__add_shmem<_InputIt, __hyperloglog_impl>);
void* __kernel_args[] = {const_cast<void*>(reinterpret_cast<const void*>(&__first)),
const_cast<void*>(reinterpret_cast<const void*>(&__num_items)),
reinterpret_cast<void*>(this)};
if (__try_reserve_shmem(__kernel, __shmem_bytes))
{
_CCCL_TRY_CUDA_API(
::cudaOccupancyMaxPotentialBlockSize,
"cudaOccupancyMaxPotentialBlockSize failed",
&__grid_size,
&__block_size,
__kernel,
__shmem_bytes);
_CCCL_TRY_CUDA_API(
::cudaLaunchKernel,
"cudaLaunchKernel failed",
__kernel,
__grid_size,
__block_size,
__kernel_args,
__shmem_bytes,
__stream.get());
}
else
{
// Computes sketch directly in global memory. (Fallback path in case there is not enough
// shared memory available)
__kernel = reinterpret_cast<const void*>(
::cuda::experimental::cuco::__hyperloglog_ns::__add_gmem<_InputIt, __hyperloglog_impl>);
_CCCL_TRY_CUDA_API(
::cudaOccupancyMaxPotentialBlockSize,
"cudaOccupancyMaxPotentialBlockSize failed",
&__grid_size,
&__block_size,
__kernel,
0);
_CCCL_TRY_CUDA_API(
::cudaLaunchKernel,
"cudaLaunchKernel failed",
__kernel,
__grid_size,
__block_size,
__kernel_args,
0,
__stream.get());
}
}
}
//! @brief Adds to be counted items to the estimator.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `__add_async`.
//!
//! @tparam _InputIt Device accessible random access input iterator where
//! <tt>std::is_convertible<std::iterator_traits<_InputIt>::value_type,
//! _Tp></tt> is `true`
//!
//! @param __first Beginning of the sequence of items
//! @param __last End of the sequence of items
//! @param __stream CUDA stream this operation is executed in
template <class _InputIt>
_CCCL_HOST_API constexpr void __add(_InputIt __first, _InputIt __last, ::cuda::stream_ref __stream)
{
__add_async(__first, __last, __stream);
__stream.sync();
}
//! @brief Merges the result of `other` estimator reference into `*this` estimator reference.
//!
//! @throw If __sketch_bytes() != other.__sketch_bytes(), then terminates execution with a device __trap()
//!
//! @tparam _CG CUDA Cooperative Group type
//! @tparam _OtherScope Thread scope of `other` estimator
//!
//! @param __group CUDA Cooperative group this operation is executed in
//! @param __other Other estimator reference to be merged into `*this`
template <class _CG, ::cuda::thread_scope _OtherScope>
_CCCL_DEVICE_API constexpr void __merge(_CG __group, const __hyperloglog_impl<_Tp, _OtherScope, _Policy>& __other)
{
if (__other.__precision != __precision)
{
_CCCL_THROW(::std::invalid_argument, "Cannot merge estimators with different sketch sizes");
}
for (int __i = __group.thread_rank(); __i < __sketch.size(); __i += __group.size())
{
__update_max(__i, __other.__sketch[__i]);
}
}
//! @brief Asynchronously merges the result of `other` estimator reference into `*this`
//! estimator.
//!
//! @throw If __sketch_bytes() != __other.__sketch_bytes(), then terminates execution with a device __trap()
//!
//! @tparam _OtherScope Thread scope of `other` estimator
//!
//! @param __other Other estimator reference to be merged into `*this`
//! @param __stream CUDA stream this operation is executed in
template <::cuda::thread_scope _OtherScope>
_CCCL_HOST_API constexpr void
__merge_async(const __hyperloglog_impl<_Tp, _OtherScope, _Policy>& __other, ::cuda::stream_ref __stream)
{
if (__other.__precision != __precision)
{
_CCCL_THROW(::std::invalid_argument, "Cannot merge estimators with different sketch sizes");
}
constexpr auto __block_size = 1024;
::cuda::experimental::cuco::__hyperloglog_ns::__merge<<<1, __block_size, 0, __stream.get()>>>(__other, *this);
}
//! @brief Merges the result of `other` estimator reference into `*this` estimator.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `__merge_async`.
//!
//! @throw If __sketch_bytes() != __other.__sketch_bytes(), then terminates execution with a device __trap()
//!
//! @tparam _OtherScope Thread scope of `other` estimator
//!
//! @param __other Other estimator reference to be merged into `*this`
//! @param __stream CUDA stream this operation is executed in
template <::cuda::thread_scope _OtherScope>
_CCCL_HOST_API constexpr void
__merge(const __hyperloglog_impl<_Tp, _OtherScope, _Policy>& __other, ::cuda::stream_ref __stream)
{
__merge_async(__other, __stream);
__stream.sync();
}
//! @brief Compute the estimated distinct items count.
//!
//! @param __group CUDA thread block group this operation is executed in
//!
//! @return Approximate distinct items count
[[nodiscard]] _CCCL_DEVICE_API double __estimate(const ::cooperative_groups::thread_block& __group) const noexcept
{
__shared__ ::cuda::atomic<__fp_type, ::cuda::std::thread_scope_block> __block_sum;
__shared__ ::cuda::atomic<::cuda::std::int32_t, ::cuda::std::thread_scope_block> __block_zeroes;
__shared__ __fp_type __estimate;
if (__group.thread_rank() == 0)
{
__block_sum.store(0);
__block_zeroes.store(0);
}
__group.sync();
__fp_type __thread_sum = 0;
::cuda::std::int32_t __thread_zeroes = 0;
for (int __i = __group.thread_rank(); __i < __sketch.size(); __i += __group.size())
{
const auto __reg = __sketch[__i];
__thread_sum += __fp_type{1} / static_cast<__fp_type>(1ull << __reg);
__thread_zeroes += __reg == 0;
}
// warp reduce Z and V
const auto __warp = ::cooperative_groups::tiled_partition<32, ::cooperative_groups::thread_block>(__group);
::cooperative_groups::reduce_update_async(
__warp, __block_sum, __thread_sum, ::cooperative_groups::plus<__fp_type>());
::cooperative_groups::reduce_update_async(
__warp, __block_zeroes, __thread_zeroes, ::cooperative_groups::plus<::cuda::std::int32_t>());
__group.sync();
if (__group.thread_rank() == 0)
{
const auto __z = __block_sum.load(::cuda::std::memory_order_relaxed);
const auto __v = __block_zeroes.load(::cuda::std::memory_order_relaxed);
__estimate = _Policy::finalize(__z, __v, __precision);
}
__group.sync();
return __estimate;
}
//! @brief Compute the estimated distinct items count.
//!
//! @note This function synchronizes the given stream.
//!
//! @tparam _HostMemoryResource Host memory resource used for allocating the host buffer required to
//! compute the final estimate by copying the sketch from device to host
//!
//! @param __host_mr Host memory resource used for copying the sketch
//! @param __stream CUDA stream this operation is executed in
//!
//! @return Approximate distinct items count
template <typename _HostMemoryResource>
[[nodiscard]] _CCCL_HOST_API double __estimate(_HostMemoryResource __host_mr, ::cuda::stream_ref __stream) const
{
const auto __num_regs = __sketch.size();
::cuda::host_buffer<__register_type> __host_sketch_buf{__stream, __host_mr, __sketch.size(), ::cuda::no_init};
::cuda::__driver::__memcpyAsync(
__host_sketch_buf.data(), __sketch.data(), sizeof(__register_type) * __num_regs, __stream.get());
__stream.sync();
__fp_type __sum = 0;
::cuda::std::int32_t __zeroes = 0;
// geometric mean computation + count registers with 0s
for (const auto __reg : __host_sketch_buf)
{
__sum += __fp_type{1} / static_cast<__fp_type>(1ull << __reg);
__zeroes += __reg == 0;
}
// dispatch to the policy's finalizer for bias correction, etc.
return _Policy::finalize(__sum, __zeroes, __precision);
}
// #endif
//! @brief Gets the hash function.
//!
//! @return The hash function, as exposed by the policy via `hash_function()`.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto __hash_function() const noexcept
{
return __policy.hash_function();
}
//! @brief Gets the policy.
//!
//! @return The policy
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr const _Policy& __policy_() const noexcept
{
return __policy;
}
//! @brief Gets the span of the sketch.
//!
//! @return The ::cuda::std::span of the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::span<::cuda::std::byte> __sketch_span() const noexcept
{
return ::cuda::std::span<::cuda::std::byte>(reinterpret_cast<::cuda::std::byte*>(__sketch.data()), __sketch_bytes());
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::size_t __sketch_bytes() const noexcept
{
return (1ull << __precision) * sizeof(__register_type);
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param sketch_size_kb Upper bound sketch size in KB
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t
__sketch_bytes(::cuda::experimental::cuco::__sketch_size_kb_t __sketch_size_kb) noexcept
{
// minimum precision is 4 or 64 bytes
return ::cuda::std::max(static_cast<::cuda::std::size_t>(sizeof(__register_type) * (1ull << 4)),
::cuda::std::bit_floor(static_cast<::cuda::std::size_t>(__sketch_size_kb * 1024)));
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param __standard_deviation Upper bound standard deviation for approximation error
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t
sketch_bytes(::cuda::experimental::cuco::__standard_deviation_t __standard_deviation) noexcept
{
// implementation taken from
// https://github.com/apache/spark/blob/6a27789ad7d59cd133653a49be0bb49729542abe/sql/catalyst/src/main/scala/org/apache/spark/sql/catalyst/util/HyperLogLogPlusPlusHelper.scala#L43
const auto __precision_from_sd =
static_cast<::cuda::std::int32_t>(::cuda::std::ceil(2.0 * ::cuda::std::log2(1.106 / __standard_deviation)));
// minimum precision is 4 or 64 bytes
const auto __precision_ = ::cuda::std::max(::cuda::std::int32_t{4}, __precision_from_sd);
// inverse of this function (omitting the minimum precision constraint) is
// standard_deviation = 1.106 / exp((__precision_ * log(2.0)) / 2.0)
return sizeof(__register_type) * (1ull << __precision_);
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param __precision HyperLogLog precision parameter
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t
sketch_bytes(::cuda::experimental::cuco::__precision_t __precision) noexcept
{
const auto __precision_value = static_cast<::cuda::std::int32_t>(__precision);
return sizeof(__register_type) * (1ull << __precision_value);
}
//! @brief Gets the alignment required for the sketch storage.
//!
//! @return The required alignment
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t __sketch_alignment() noexcept
{
return alignof(__register_type);
}
private:
//! @brief Atomically updates the register at position `i` with `max(reg[i], value)`.
//!
//! @param __i Register index
//! @param __value New value
_CCCL_DEVICE_API constexpr void __update_max(int __i, __register_type __value) noexcept
{
::cuda::atomic_ref<__register_type, _Scope> __register_ref(__sketch[__i]);
__register_ref.fetch_max(__value, ::cuda::memory_order_relaxed);
}
//! @brief Try expanding the shmem partition for a given kernel beyond 48KB if necessary.
//!
//! @tparam _Kernel Type of kernel function
//!
//! @param __kernel The kernel function
//! @param __shmem_bytes Number of requested dynamic shared memory bytes
//!
//! @returns True iff kernel configuration is successful
template <typename _Kernel>
[[nodiscard]] _CCCL_HOST_API constexpr bool __try_reserve_shmem(_Kernel __kernel, int __shmem_bytes) const
{
int __device = -1;
_CCCL_TRY_CUDA_API(::cudaGetDevice, "cudaGetDevice failed", &__device);
int __max_shmem_bytes = 0;
_CCCL_TRY_CUDA_API(
::cudaDeviceGetAttribute,
"cudaDeviceGetAttribute failed",
&__max_shmem_bytes,
::cudaDevAttrMaxSharedMemoryPerBlockOptin,
__device);
if (__shmem_bytes <= __max_shmem_bytes)
{
_CCCL_TRY_CUDA_API(
::cudaFuncSetAttribute,
"cudaFuncSetAttribute failed",
reinterpret_cast<const void*>(__kernel),
cudaFuncAttributeMaxDynamicSharedMemorySize,
__shmem_bytes);
return true;
}
else
{
return false;
}
}
};
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_HYPERLOGLOG_IMPL_CUH

View File

@@ -0,0 +1,182 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_HYPERLOGLOG_KERNELS_CUH
#define _CUDAX___CUCO_DETAIL_HYPERLOGLOG_KERNELS_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__memory/assume_aligned.h>
#include <cuda/std/array>
#include <cuda/std/cstdint>
#include <cuda/std/span>
#include <cooperative_groups.h>
#include <cuda/std/__cccl/prologue.h>
#if _CCCL_CUDA_COMPILATION()
_CCCL_DIAG_PUSH
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
namespace cuda::experimental::cuco::__hyperloglog_ns
{
//! @brief Returns the global thread ID in a 1D grid
//!
//! @return The global thread ID
[[nodiscard]] _CCCL_DEVICE_API inline ::cuda::std::int64_t __global_thread_id() noexcept
{
return static_cast<::cuda::std::int64_t>(blockDim.x) * blockIdx.x + threadIdx.x;
}
//! @brief Returns the grid stride of a 1D grid
//!
//! @return The grid stride
[[nodiscard]] _CCCL_DEVICE_API inline ::cuda::std::int64_t __grid_stride() noexcept
{
return static_cast<::cuda::std::int64_t>(gridDim.x) * blockDim.x;
}
template <class _RefType>
_CCCL_KERNEL_ATTRIBUTES void __clear(_RefType __ref)
{
const auto __block = ::cooperative_groups::this_thread_block();
if (__block.group_index().x == 0)
{
__ref.__clear(__block);
}
}
template <int _VectorSize, class _RefType>
_CCCL_KERNEL_ATTRIBUTES void
__add_shmem_vectorized(const typename _RefType::__value_type* __first, ::cuda::std::int64_t __n, _RefType __ref)
{
using __value_type = typename _RefType::__value_type;
// TODO: replace with ::cuda::__vector_type
using __vector_type = ::cuda::std::array<__value_type, _VectorSize>;
using __local_ref_type = typename _RefType::template __rebind_scope<::cuda::std::thread_scope_block>;
// Base address of dynamic shared memory is guaranteed to be aligned to at least 16 bytes which is
// sufficient for this purpose
extern __shared__ ::cuda::std::byte __local_sketch[];
const auto __loop_stride = __grid_stride();
auto __idx = __global_thread_id();
const auto __grid = ::cooperative_groups::this_grid();
const auto __block = ::cooperative_groups::this_thread_block();
__local_ref_type __local_ref(::cuda::std::span{__local_sketch, __ref.__sketch_bytes()}, {});
__local_ref.__clear(__block);
__block.sync();
// each thread processes VectorSize-many items per iteration
__vector_type __vec;
while (__idx < __n / _VectorSize)
{
__vec = *reinterpret_cast<const __vector_type*>(
::cuda::std::assume_aligned<sizeof(__vector_type)>(__first + __idx * _VectorSize));
for (int i = 0; i < _VectorSize; ++i)
{
__local_ref.__add(__vec[i]);
}
__idx += __loop_stride;
}
// a single thread processes the remaining items
# if _CCCL_CTK_AT_LEAST(12, 1)
::cooperative_groups::invoke_one(__grid, [&]() {
const auto __remainder = __n % _VectorSize;
for (int __i = 0; __i < __remainder; ++__i)
{
__local_ref.__add(*(__first + __n - __i - 1));
}
});
# else // ^^^ _CCCL_CTK_AT_LEAST(12, 1) ^^^ / vvv _CCCL_CTK_BELOW(12, 1) vvv
if (__grid.thread_rank() == 0)
{
const auto __remainder = __n % _VectorSize;
for (int __i = 0; __i < __remainder; ++__i)
{
__local_ref.__add(*(__first + __n - __i - 1));
}
}
# endif // ^^^ _CCCL_CTK_BELOW(12, 1) ^^^
__block.sync();
__ref.__merge(__block, __local_ref);
}
template <class _InputIt, class _RefType>
_CCCL_KERNEL_ATTRIBUTES void __add_shmem(_InputIt __first, ::cuda::std::int64_t __n, _RefType __ref)
{
using __local_ref_type = typename _RefType::template __rebind_scope<::cuda::std::thread_scope_block>;
// TODO assert alignment
extern __shared__ ::cuda::std::byte __local_sketch[];
const auto __loop_stride = __grid_stride();
auto __idx = __global_thread_id();
const auto __block = ::cooperative_groups::this_thread_block();
__local_ref_type __local_ref(::cuda::std::span{__local_sketch, __ref.__sketch_bytes()}, {});
__local_ref.__clear(__block);
__block.sync();
while (__idx < __n)
{
__local_ref.__add(*(__first + __idx));
__idx += __loop_stride;
}
__block.sync();
__ref.__merge(__block, __local_ref);
}
template <class _InputIt, class _RefType>
_CCCL_KERNEL_ATTRIBUTES void __add_gmem(_InputIt __first, ::cuda::std::int64_t __n, _RefType __ref)
{
const auto __loop_stride = __grid_stride();
auto __idx = __global_thread_id();
while (__idx < __n)
{
__ref.__add(*(__first + __idx));
__idx += __loop_stride;
}
}
template <class _OtherRefType, class _RefType>
_CCCL_KERNEL_ATTRIBUTES void __merge(_OtherRefType __other_ref, _RefType __ref)
{
const auto __block = ::cooperative_groups::this_thread_block();
if (__block.group_index().x == 0)
{
__ref.__merge(__block, __other_ref);
}
}
} // namespace cuda::experimental::cuco::__hyperloglog_ns
_CCCL_DIAG_POP
#endif // _CCCL_CUDA_COMPILATION()
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_HYPERLOGLOG_KERNELS_CUH

View File

@@ -0,0 +1,166 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_HYPERLOGLOG_TUNING_CUH
#define _CUDAX___CUCO_DETAIL_HYPERLOGLOG_TUNING_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::__hyperloglog_ns
{
#ifndef _CUDAX_CUCO_HLL_TUNING_ARR_DECL
# if _CCCL_OS(WINDOWS)
# define _CUDAX_CUCO_HLL_TUNING_ARR_DECL _CCCL_GLOBAL_CONSTANT double
# else
# define _CUDAX_CUCO_HLL_TUNING_ARR_DECL _CCCL_DEVICE inline constexpr double
# endif
#endif
// clang-format off
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __threshold_data[] = {10.0, 20.0, 40.0, 80.0, 220.0, 400.0, 900.0, 1800.0, 3100.0, 6500.0, 15500.0, 20000.0, 50000.0, 120000.0, 350000.0};
//! @brief Get threshold value for a given precision
//!
//! @param __precision The precision value (4-18)
//! @return The threshold value for the given precision
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto __threshold(::cuda::std::int32_t __precision) noexcept {
return __threshold_data[__precision - 4];
}
// HLL++ uses an interpolation method over the raw estimated cardinality to select the optimal bias.
// Parameters/interpolation points taken from
// https://docs.google.com/document/d/1gyjfMHy43U9OWBXxfaeG-3MjGzejW1dlpyMwEYAAWEI/mobilebasic
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p4[] = {11.0, 11.717, 12.207, 12.7896, 13.2882, 13.8204, 14.3772, 14.9342, 15.5202, 16.161, 16.7722, 17.4636, 18.0396, 18.6766, 19.3566, 20.0454, 20.7936, 21.4856, 22.2666, 22.9946, 23.766, 24.4692, 25.3638, 26.0764, 26.7864, 27.7602, 28.4814, 29.433, 30.2926, 31.0664, 31.9996, 32.7956, 33.5366, 34.5894, 35.5738, 36.2698, 37.3682, 38.0544, 39.2342, 40.0108, 40.7966, 41.9298, 42.8704, 43.6358, 44.5194, 45.773, 46.6772, 47.6174, 48.4888, 49.3304, 50.2506, 51.4996, 52.3824, 53.3078, 54.3984, 55.5838, 56.6618, 57.2174, 58.3514, 59.0802, 60.1482, 61.0376, 62.3598, 62.8078, 63.9744, 64.914, 65.781, 67.1806, 68.0594, 68.8446, 69.7928, 70.8248, 71.8324, 72.8598, 73.6246, 74.7014, 75.393, 76.6708, 77.2394};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p5[] = {23.0, 23.1194, 23.8208, 24.2318, 24.77, 25.2436, 25.7774, 26.2848, 26.8224, 27.3742, 27.9336, 28.503, 29.0494, 29.6292, 30.2124, 30.798, 31.367, 31.9728, 32.5944, 33.217, 33.8438, 34.3696, 35.0956, 35.7044, 36.324, 37.0668, 37.6698, 38.3644, 39.049, 39.6918, 40.4146, 41.082, 41.687, 42.5398, 43.2462, 43.857, 44.6606, 45.4168, 46.1248, 46.9222, 47.6804, 48.447, 49.3454, 49.9594, 50.7636, 51.5776, 52.331, 53.19, 53.9676, 54.7564, 55.5314, 56.4442, 57.3708, 57.9774, 58.9624, 59.8796, 60.755, 61.472, 62.2076, 63.1024, 63.8908, 64.7338, 65.7728, 66.629, 67.413, 68.3266, 69.1524, 70.2642, 71.1806, 72.0566, 72.9192, 73.7598, 74.3516, 75.5802, 76.4386, 77.4916, 78.1524, 79.1892, 79.8414, 80.8798, 81.8376, 82.4698, 83.7656, 84.331, 85.5914, 86.6012, 87.7016, 88.5582, 89.3394, 90.3544, 91.4912, 92.308, 93.3552, 93.9746, 95.2052, 95.727, 97.1322, 98.3944, 98.7588, 100.242, 101.1914, 102.2538, 102.8776, 103.6292, 105.1932, 105.9152, 107.0868, 107.6728, 108.7144, 110.3114, 110.8716, 111.245, 112.7908, 113.7064, 114.636, 115.7464, 116.1788, 117.7464, 118.4896, 119.6166, 120.5082, 121.7798, 122.9028, 123.4426, 124.8854, 125.705, 126.4652, 128.3464, 128.3462, 130.0398, 131.0342, 131.0042, 132.4766, 133.511, 134.7252, 135.425, 136.5172, 138.0572, 138.6694, 139.3712, 140.8598, 141.4594, 142.554, 143.4006, 144.7374, 146.1634, 146.8994, 147.605, 147.9304, 149.1636, 150.2468, 151.5876, 152.2096, 153.7032, 154.7146, 155.807, 156.9228, 157.0372, 158.5852};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p6[] = {46.0, 46.1902, 47.271, 47.8358, 48.8142, 49.2854, 50.317, 51.354, 51.8924, 52.9436, 53.4596, 54.5262, 55.6248, 56.1574, 57.2822, 57.837, 58.9636, 60.074, 60.7042, 61.7976, 62.4772, 63.6564, 64.7942, 65.5004, 66.686, 67.291, 68.5672, 69.8556, 70.4982, 71.8204, 72.4252, 73.7744, 75.0786, 75.8344, 77.0294, 77.8098, 79.0794, 80.5732, 81.1878, 82.5648, 83.2902, 84.6784, 85.3352, 86.8946, 88.3712, 89.0852, 90.499, 91.2686, 92.6844, 94.2234, 94.9732, 96.3356, 97.2286, 98.7262, 100.3284, 101.1048, 102.5962, 103.3562, 105.1272, 106.4184, 107.4974, 109.0822, 109.856, 111.48, 113.2834, 114.0208, 115.637, 116.5174, 118.0576, 119.7476, 120.427, 122.1326, 123.2372, 125.2788, 126.6776, 127.7926, 129.1952, 129.9564, 131.6454, 133.87, 134.5428, 136.2, 137.0294, 138.6278, 139.6782, 141.792, 143.3516, 144.2832, 146.0394, 147.0748, 148.4912, 150.849, 151.696, 153.5404, 154.073, 156.3714, 157.7216, 158.7328, 160.4208, 161.4184, 163.9424, 165.2772, 166.411, 168.1308, 168.769, 170.9258, 172.6828, 173.7502, 175.706, 176.3886, 179.0186, 180.4518, 181.927, 183.4172, 184.4114, 186.033, 188.5124, 189.5564, 191.6008, 192.4172, 193.8044, 194.997, 197.4548, 198.8948, 200.2346, 202.3086, 203.1548, 204.8842, 206.6508, 206.6772, 209.7254, 210.4752, 212.7228, 214.6614, 215.1676, 217.793, 218.0006, 219.9052, 221.66, 223.5588, 225.1636, 225.6882, 227.7126, 229.4502, 231.1978, 232.9756, 233.1654, 236.727, 238.1974, 237.7474, 241.1346, 242.3048, 244.1948, 245.3134, 246.879, 249.1204, 249.853, 252.6792, 253.857, 254.4486, 257.2362, 257.9534, 260.0286, 260.5632, 262.663, 264.723, 265.7566, 267.2566, 267.1624, 270.62, 272.8216, 273.2166, 275.2056, 276.2202, 278.3726, 280.3344, 281.9284, 283.9728, 284.1924, 286.4872, 287.587, 289.807, 291.1206, 292.769, 294.8708, 296.665, 297.1182, 299.4012, 300.6352, 302.1354, 304.1756, 306.1606, 307.3462, 308.5214, 309.4134, 310.8352, 313.9684, 315.837, 316.7796, 318.9858};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p7[] = {92.0, 93.4934, 94.9758, 96.4574, 97.9718, 99.4954, 101.5302, 103.0756, 104.6374, 106.1782, 107.7888, 109.9522, 111.592, 113.2532, 114.9086, 116.5938, 118.9474, 120.6796, 122.4394, 124.2176, 125.9768, 128.4214, 130.2528, 132.0102, 133.8658, 135.7278, 138.3044, 140.1316, 142.093, 144.0032, 145.9092, 148.6306, 150.5294, 152.5756, 154.6508, 156.662, 159.552, 161.3724, 163.617, 165.5754, 167.7872, 169.8444, 172.7988, 174.8606, 177.2118, 179.3566, 181.4476, 184.5882, 186.6816, 189.0824, 191.0258, 193.6048, 196.4436, 198.7274, 200.957, 203.147, 205.4364, 208.7592, 211.3386, 213.781, 215.8028, 218.656, 221.6544, 223.996, 226.4718, 229.1544, 231.6098, 234.5956, 237.0616, 239.5758, 242.4878, 244.5244, 248.2146, 250.724, 252.8722, 255.5198, 258.0414, 261.941, 264.9048, 266.87, 269.4304, 272.028, 274.4708, 278.37, 281.0624, 283.4668, 286.5532, 289.4352, 293.2564, 295.2744, 298.2118, 300.7472, 304.1456, 307.2928, 309.7504, 312.5528, 315.979, 318.2102, 322.1834, 324.3494, 327.325, 330.6614, 332.903, 337.2544, 339.9042, 343.215, 345.2864, 348.0814, 352.6764, 355.301, 357.139, 360.658, 363.1732, 366.5902, 369.9538, 373.0828, 375.922, 378.9902, 382.7328, 386.4538, 388.1136, 391.2234, 394.0878, 396.708, 401.1556, 404.1852, 406.6372, 409.6822, 412.7796, 416.6078, 418.4916, 422.131, 424.5376, 428.1988, 432.211, 434.4502, 438.5282, 440.912, 444.0448, 447.7432, 450.8524, 453.7988, 456.7858, 458.8868, 463.9886, 466.5064, 468.9124, 472.6616, 475.4682, 478.582, 481.304, 485.2738, 488.6894, 490.329, 496.106, 497.6908, 501.1374, 504.5322, 506.8848, 510.3324, 513.4512, 516.179, 520.4412, 522.6066, 526.167, 528.7794, 533.379, 536.067, 538.46, 542.9116, 545.692, 547.9546, 552.493, 555.2722, 557.335, 562.449, 564.2014, 569.0738, 571.0974, 574.8564, 578.2996, 581.409, 583.9704, 585.8098, 589.6528, 594.5998, 595.958, 600.068, 603.3278, 608.2016, 609.9632, 612.864, 615.43, 620.7794, 621.272, 625.8644, 629.206, 633.219, 634.5154, 638.6102};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p8[] = {184.2152, 187.2454, 190.2096, 193.6652, 196.6312, 199.6822, 203.249, 206.3296, 210.0038, 213.2074, 216.4612, 220.27, 223.5178, 227.4412, 230.8032, 234.1634, 238.1688, 241.6074, 245.6946, 249.2664, 252.8228, 257.0432, 260.6824, 264.9464, 268.6268, 272.2626, 276.8376, 280.4034, 284.8956, 288.8522, 292.7638, 297.3552, 301.3556, 305.7526, 309.9292, 313.8954, 318.8198, 322.7668, 327.298, 331.6688, 335.9466, 340.9746, 345.1672, 349.3474, 354.3028, 358.8912, 364.114, 368.4646, 372.9744, 378.4092, 382.6022, 387.843, 392.5684, 397.1652, 402.5426, 407.4152, 412.5388, 417.3592, 422.1366, 427.486, 432.3918, 437.5076, 442.509, 447.3834, 453.3498, 458.0668, 463.7346, 469.1228, 473.4528, 479.7, 484.644, 491.0518, 495.5774, 500.9068, 506.432, 512.1666, 517.434, 522.6644, 527.4894, 533.6312, 538.3804, 544.292, 550.5496, 556.0234, 562.8206, 566.6146, 572.4188, 579.117, 583.6762, 590.6576, 595.7864, 601.509, 607.5334, 612.9204, 619.772, 624.2924, 630.8654, 636.1836, 642.745, 649.1316, 655.0386, 660.0136, 666.6342, 671.6196, 678.1866, 684.4282, 689.3324, 695.4794, 702.5038, 708.129, 713.528, 720.3204, 726.463, 732.7928, 739.123, 744.7418, 751.2192, 756.5102, 762.6066, 769.0184, 775.2224, 781.4014, 787.7618, 794.1436, 798.6506, 805.6378, 811.766, 819.7514, 824.5776, 828.7322, 837.8048, 843.6302, 849.9336, 854.4798, 861.3388, 867.9894, 873.8196, 880.3136, 886.2308, 892.4588, 899.0816, 905.4076, 912.0064, 917.3878, 923.619, 929.998, 937.3482, 943.9506, 947.991, 955.1144, 962.203, 968.8222, 975.7324, 981.7826, 988.7666, 994.2648, 1000.3128, 1007.4082, 1013.7536, 1020.3376, 1026.7156, 1031.7478, 1037.4292, 1045.393, 1051.2278, 1058.3434, 1062.8726, 1071.884, 1076.806, 1082.9176, 1089.1678, 1095.5032, 1102.525, 1107.2264, 1115.315, 1120.93, 1127.252, 1134.1496, 1139.0408, 1147.5448, 1153.3296, 1158.1974, 1166.5262, 1174.3328, 1175.657, 1184.4222, 1190.9172, 1197.1292, 1204.4606, 1210.4578, 1218.8728, 1225.3336, 1226.6592, 1236.5768, 1241.363, 1249.4074, 1254.6566, 1260.8014, 1266.5454, 1274.5192};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p9[] = {369.0, 374.8294, 381.2452, 387.6698, 394.1464, 400.2024, 406.8782, 413.6598, 420.462, 427.2826, 433.7102, 440.7416, 447.9366, 455.1046, 462.285, 469.0668, 476.306, 483.8448, 491.301, 498.9886, 506.2422, 513.8138, 521.7074, 529.7428, 537.8402, 545.1664, 553.3534, 561.594, 569.6886, 577.7876, 585.65, 594.228, 602.8036, 611.1666, 620.0818, 628.0824, 637.2574, 646.302, 655.1644, 664.0056, 672.3802, 681.7192, 690.5234, 700.2084, 708.831, 718.485, 728.1112, 737.4764, 746.76, 756.3368, 766.5538, 775.5058, 785.2646, 795.5902, 804.3818, 814.8998, 824.9532, 835.2062, 845.2798, 854.4728, 864.9582, 875.3292, 886.171, 896.781, 906.5716, 916.7048, 927.5322, 937.875, 949.3972, 958.3464, 969.7274, 980.2834, 992.1444, 1003.4264, 1013.0166, 1024.018, 1035.0438, 1046.34, 1057.6856, 1068.9836, 1079.0312, 1091.677, 1102.3188, 1113.4846, 1124.4424, 1135.739, 1147.1488, 1158.9202, 1169.406, 1181.5342, 1193.2834, 1203.8954, 1216.3286, 1226.2146, 1239.6684, 1251.9946, 1262.123, 1275.4338, 1285.7378, 1296.076, 1308.9692, 1320.4964, 1333.0998, 1343.9864, 1357.7754, 1368.3208, 1380.4838, 1392.7388, 1406.0758, 1416.9098, 1428.9728, 1440.9228, 1453.9292, 1462.617, 1476.05, 1490.2996, 1500.6128, 1513.7392, 1524.5174, 1536.6322, 1548.2584, 1562.3766, 1572.423, 1587.1232, 1596.5164, 1610.5938, 1622.5972, 1633.1222, 1647.7674, 1658.5044, 1671.57, 1683.7044, 1695.4142, 1708.7102, 1720.6094, 1732.6522, 1747.841, 1756.4072, 1769.9786, 1782.3276, 1797.5216, 1808.3186, 1819.0694, 1834.354, 1844.575, 1856.2808, 1871.1288, 1880.7852, 1893.9622, 1906.3418, 1920.6548, 1932.9302, 1945.8584, 1955.473, 1968.8248, 1980.6446, 1995.9598, 2008.349, 2019.8556, 2033.0334, 2044.0206, 2059.3956, 2069.9174, 2082.6084, 2093.7036, 2106.6108, 2118.9124, 2132.301, 2144.7628, 2159.8422, 2171.0212, 2183.101, 2193.5112, 2208.052, 2221.3194, 2233.3282, 2247.295, 2257.7222, 2273.342, 2286.5638, 2299.6786, 2310.8114, 2322.3312, 2335.516, 2349.874, 2363.5968, 2373.865, 2387.1918, 2401.8328, 2414.8496, 2424.544, 2436.7592, 2447.1682, 2464.1958, 2474.3438, 2489.0006, 2497.4526, 2513.6586, 2527.19, 2540.7028, 2553.768};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p10[] = {738.1256, 750.4234, 763.1064, 775.4732, 788.4636, 801.0644, 814.488, 827.9654, 841.0832, 854.7864, 868.1992, 882.2176, 896.5228, 910.1716, 924.7752, 938.899, 953.6126, 968.6492, 982.9474, 998.5214, 1013.1064, 1028.6364, 1044.2468, 1059.4588, 1075.3832, 1091.0584, 1106.8606, 1123.3868, 1139.5062, 1156.1862, 1172.463, 1189.339, 1206.1936, 1223.1292, 1240.1854, 1257.2908, 1275.3324, 1292.8518, 1310.5204, 1328.4854, 1345.9318, 1364.552, 1381.4658, 1400.4256, 1419.849, 1438.152, 1456.8956, 1474.8792, 1494.118, 1513.62, 1532.5132, 1551.9322, 1570.7726, 1590.6086, 1610.5332, 1630.5918, 1650.4294, 1669.7662, 1690.4106, 1710.7338, 1730.9012, 1750.4486, 1770.1556, 1791.6338, 1812.7312, 1833.6264, 1853.9526, 1874.8742, 1896.8326, 1918.1966, 1939.5594, 1961.07, 1983.037, 2003.1804, 2026.071, 2047.4884, 2070.0848, 2091.2944, 2114.333, 2135.9626, 2158.2902, 2181.0814, 2202.0334, 2224.4832, 2246.39, 2269.7202, 2292.1714, 2314.2358, 2338.9346, 2360.891, 2384.0264, 2408.3834, 2430.1544, 2454.8684, 2476.9896, 2501.4368, 2522.8702, 2548.0408, 2570.6738, 2593.5208, 2617.0158, 2640.2302, 2664.0962, 2687.4986, 2714.2588, 2735.3914, 2759.6244, 2781.8378, 2808.0072, 2830.6516, 2856.2454, 2877.2136, 2903.4546, 2926.785, 2951.2294, 2976.468, 3000.867, 3023.6508, 3049.91, 3073.5984, 3098.162, 3121.5564, 3146.2328, 3170.9484, 3195.5902, 3221.3346, 3242.7032, 3271.6112, 3296.5546, 3317.7376, 3345.072, 3369.9518, 3394.326, 3418.1818, 3444.6926, 3469.086, 3494.2754, 3517.8698, 3544.248, 3565.3768, 3588.7234, 3616.979, 3643.7504, 3668.6812, 3695.72, 3719.7392, 3742.6224, 3770.4456, 3795.6602, 3819.9058, 3844.002, 3869.517, 3895.6824, 3920.8622, 3947.1364, 3973.985, 3995.4772, 4021.62, 4046.628, 4074.65, 4096.2256, 4121.831, 4146.6406, 4173.276, 4195.0744, 4223.9696, 4251.3708, 4272.9966, 4300.8046, 4326.302, 4353.1248, 4374.312, 4403.0322, 4426.819, 4450.0598, 4478.5206, 4504.8116, 4528.8928, 4553.9584, 4578.8712, 4603.8384, 4632.3872, 4655.5128, 4675.821, 4704.6222, 4731.9862, 4755.4174, 4781.2628, 4804.332, 4832.3048, 4862.8752, 4883.4148, 4906.9544, 4935.3516, 4954.3532, 4984.0248, 5011.217, 5035.3258, 5057.3672, 5084.1828};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p11[] = {1477.0, 1501.6014, 1526.5802, 1551.7942, 1577.3042, 1603.2062, 1629.8402, 1656.2292, 1682.9462, 1709.9926, 1737.3026, 1765.4252, 1793.0578, 1821.6092, 1849.626, 1878.5568, 1908.527, 1937.5154, 1967.1874, 1997.3878, 2027.37, 2058.1972, 2089.5728, 2120.1012, 2151.9668, 2183.292, 2216.0772, 2247.8578, 2280.6562, 2313.041, 2345.714, 2380.3112, 2414.1806, 2447.9854, 2481.656, 2516.346, 2551.5154, 2586.8378, 2621.7448, 2656.6722, 2693.5722, 2729.1462, 2765.4124, 2802.8728, 2838.898, 2876.408, 2913.4926, 2951.4938, 2989.6776, 3026.282, 3065.7704, 3104.1012, 3143.7388, 3181.6876, 3221.1872, 3261.5048, 3300.0214, 3339.806, 3381.409, 3421.4144, 3461.4294, 3502.2286, 3544.651, 3586.6156, 3627.337, 3670.083, 3711.1538, 3753.5094, 3797.01, 3838.6686, 3882.1678, 3922.8116, 3967.9978, 4009.9204, 4054.3286, 4097.5706, 4140.6014, 4185.544, 4229.5976, 4274.583, 4316.9438, 4361.672, 4406.2786, 4451.8628, 4496.1834, 4543.505, 4589.1816, 4632.5188, 4678.2294, 4724.8908, 4769.0194, 4817.052, 4861.4588, 4910.1596, 4956.4344, 5002.5238, 5048.13, 5093.6374, 5142.8162, 5187.7894, 5237.3984, 5285.6078, 5331.0858, 5379.1036, 5428.6258, 5474.6018, 5522.7618, 5571.5822, 5618.59, 5667.9992, 5714.88, 5763.454, 5808.6982, 5860.3644, 5910.2914, 5953.571, 6005.9232, 6055.1914, 6104.5882, 6154.5702, 6199.7036, 6251.1764, 6298.7596, 6350.0302, 6398.061, 6448.4694, 6495.933, 6548.0474, 6597.7166, 6646.9416, 6695.9208, 6742.6328, 6793.5276, 6842.1934, 6894.2372, 6945.3864, 6996.9228, 7044.2372, 7094.1374, 7142.2272, 7192.2942, 7238.8338, 7288.9006, 7344.0908, 7394.8544, 7443.5176, 7490.4148, 7542.9314, 7595.6738, 7641.9878, 7694.3688, 7743.0448, 7797.522, 7845.53, 7899.594, 7950.3132, 7996.455, 8050.9442, 8092.9114, 8153.1374, 8197.4472, 8252.8278, 8301.8728, 8348.6776, 8401.4698, 8453.551, 8504.6598, 8553.8944, 8604.1276, 8657.6514, 8710.3062, 8758.908, 8807.8706, 8862.1702, 8910.4668, 8960.77, 9007.2766, 9063.164, 9121.0534, 9164.1354, 9218.1594, 9267.767, 9319.0594, 9372.155, 9419.7126, 9474.3722, 9520.1338, 9572.368, 9622.7702, 9675.8448, 9726.5396, 9778.7378, 9827.6554, 9878.1922, 9928.7782, 9978.3984, 10026.578, 10076.5626, 10137.1618, 10177.5244, 10229.9176};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p12[] = {2954.0, 3003.4782, 3053.3568, 3104.3666, 3155.324, 3206.9598, 3259.648, 3312.539, 3366.1474, 3420.2576, 3474.8376, 3530.6076, 3586.451, 3643.38, 3700.4104, 3757.5638, 3815.9676, 3875.193, 3934.838, 3994.8548, 4055.018, 4117.1742, 4178.4482, 4241.1294, 4304.4776, 4367.4044, 4431.8724, 4496.3732, 4561.4304, 4627.5326, 4693.949, 4761.5532, 4828.7256, 4897.6182, 4965.5186, 5034.4528, 5104.865, 5174.7164, 5244.6828, 5316.6708, 5387.8312, 5459.9036, 5532.476, 5604.8652, 5679.6718, 5753.757, 5830.2072, 5905.2828, 5980.0434, 6056.6264, 6134.3192, 6211.5746, 6290.0816, 6367.1176, 6447.9796, 6526.5576, 6606.1858, 6686.9144, 6766.1142, 6847.0818, 6927.9664, 7010.9096, 7091.0816, 7175.3962, 7260.3454, 7344.018, 7426.4214, 7511.3106, 7596.0686, 7679.8094, 7765.818, 7852.4248, 7936.834, 8022.363, 8109.5066, 8200.4554, 8288.5832, 8373.366, 8463.4808, 8549.7682, 8642.0522, 8728.3288, 8820.9528, 8907.727, 9001.0794, 9091.2522, 9179.988, 9269.852, 9362.6394, 9453.642, 9546.9024, 9640.6616, 9732.6622, 9824.3254, 9917.7484, 10007.9392, 10106.7508, 10196.2152, 10289.8114, 10383.5494, 10482.3064, 10576.8734, 10668.7872, 10764.7156, 10862.0196, 10952.793, 11049.9748, 11146.0702, 11241.4492, 11339.2772, 11434.2336, 11530.741, 11627.6136, 11726.311, 11821.5964, 11918.837, 12015.3724, 12113.0162, 12213.0424, 12306.9804, 12408.4518, 12504.8968, 12604.586, 12700.9332, 12798.705, 12898.5142, 12997.0488, 13094.788, 13198.475, 13292.7764, 13392.9698, 13486.8574, 13590.1616, 13686.5838, 13783.6264, 13887.2638, 13992.0978, 14081.0844, 14189.9956, 14280.0912, 14382.4956, 14486.4384, 14588.1082, 14686.2392, 14782.276, 14888.0284, 14985.1864, 15088.8596, 15187.0998, 15285.027, 15383.6694, 15495.8266, 15591.3736, 15694.2008, 15790.3246, 15898.4116, 15997.4522, 16095.5014, 16198.8514, 16291.7492, 16402.6424, 16499.1266, 16606.2436, 16697.7186, 16796.3946, 16902.3376, 17005.7672, 17100.814, 17206.8282, 17305.8262, 17416.0744, 17508.4092, 17617.0178, 17715.4554, 17816.758, 17920.1748, 18012.9236, 18119.7984, 18223.2248, 18324.2482, 18426.6276, 18525.0932, 18629.8976, 18733.2588, 18831.0466, 18940.1366, 19032.2696, 19131.729, 19243.4864, 19349.6932, 19442.866, 19547.9448, 19653.2798, 19754.4034, 19854.0692, 19965.1224, 20065.1774, 20158.2212, 20253.353, 20366.3264, 20463.22};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p13[] = {5908.5052, 6007.2672, 6107.347, 6208.5794, 6311.2622, 6414.5514, 6519.3376, 6625.6952, 6732.5988, 6841.3552, 6950.5972, 7061.3082, 7173.5646, 7287.109, 7401.8216, 7516.4344, 7633.3802, 7751.2962, 7870.3784, 7990.292, 8110.79, 8233.4574, 8356.6036, 8482.2712, 8607.7708, 8735.099, 8863.1858, 8993.4746, 9123.8496, 9255.6794, 9388.5448, 9522.7516, 9657.3106, 9792.6094, 9930.5642, 10068.794, 10206.7256, 10347.81, 10490.3196, 10632.0778, 10775.9916, 10920.4662, 11066.124, 11213.073, 11358.0362, 11508.1006, 11659.1716, 11808.7514, 11959.4884, 12112.1314, 12265.037, 12420.3756, 12578.933, 12734.311, 12890.0006, 13047.2144, 13207.3096, 13368.5144, 13528.024, 13689.847, 13852.7528, 14018.3168, 14180.5372, 14346.9668, 14513.5074, 14677.867, 14846.2186, 15017.4186, 15184.9716, 15356.339, 15529.2972, 15697.3578, 15871.8686, 16042.187, 16216.4094, 16389.4188, 16565.9126, 16742.3272, 16919.0042, 17094.7592, 17273.965, 17451.8342, 17634.4254, 17810.5984, 17988.9242, 18171.051, 18354.7938, 18539.466, 18721.0408, 18904.9972, 19081.867, 19271.9118, 19451.8694, 19637.9816, 19821.2922, 20013.1292, 20199.3858, 20387.8726, 20572.9514, 20770.7764, 20955.1714, 21144.751, 21329.9952, 21520.709, 21712.7016, 21906.3868, 22096.2626, 22286.0524, 22475.051, 22665.5098, 22862.8492, 23055.5294, 23249.6138, 23437.848, 23636.273, 23826.093, 24020.3296, 24213.3896, 24411.7392, 24602.9614, 24805.7952, 24998.1552, 25193.9588, 25389.0166, 25585.8392, 25780.6976, 25981.2728, 26175.977, 26376.5252, 26570.1964, 26773.387, 26962.9812, 27163.0586, 27368.164, 27565.0534, 27758.7428, 27961.1276, 28163.2324, 28362.3816, 28565.7668, 28758.644, 28956.9768, 29163.4722, 29354.7026, 29561.1186, 29767.9948, 29959.9986, 30164.0492, 30366.9818, 30562.5338, 30762.9928, 30976.1592, 31166.274, 31376.722, 31570.3734, 31770.809, 31974.8934, 32179.5286, 32387.5442, 32582.3504, 32794.076, 32989.9528, 33191.842, 33392.4684, 33595.659, 33801.8672, 34000.3414, 34200.0922, 34402.6792, 34610.0638, 34804.0084, 35011.13, 35218.669, 35418.6634, 35619.0792, 35830.6534, 36028.4966, 36229.7902, 36438.6422, 36630.7764, 36833.3102, 37048.6728, 37247.3916, 37453.5904, 37669.3614, 37854.5526, 38059.305, 38268.0936, 38470.2516, 38674.7064, 38876.167, 39068.3794, 39281.9144, 39492.8566, 39684.8628, 39898.4108, 40093.1836, 40297.6858, 40489.7086, 40717.2424};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p14[] = {11817.475, 12015.0046, 12215.3792, 12417.7504, 12623.1814, 12830.0086, 13040.0072, 13252.503, 13466.178, 13683.2738, 13902.0344, 14123.9798, 14347.394, 14573.7784, 14802.6894, 15033.6824, 15266.9134, 15502.8624, 15741.4944, 15980.7956, 16223.8916, 16468.6316, 16715.733, 16965.5726, 17217.204, 17470.666, 17727.8516, 17986.7886, 18247.6902, 18510.9632, 18775.304, 19044.7486, 19314.4408, 19587.202, 19862.2576, 20135.924, 20417.0324, 20697.9788, 20979.6112, 21265.0274, 21550.723, 21841.6906, 22132.162, 22428.1406, 22722.127, 23020.5606, 23319.7394, 23620.4014, 23925.2728, 24226.9224, 24535.581, 24845.505, 25155.9618, 25470.3828, 25785.9702, 26103.7764, 26420.4132, 26742.0186, 27062.8852, 27388.415, 27714.6024, 28042.296, 28365.4494, 28701.1526, 29031.8008, 29364.2156, 29704.497, 30037.1458, 30380.111, 30723.8168, 31059.5114, 31404.9498, 31751.6752, 32095.2686, 32444.7792, 32794.767, 33145.204, 33498.4226, 33847.6502, 34209.006, 34560.849, 34919.4838, 35274.9778, 35635.1322, 35996.3266, 36359.1394, 36722.8266, 37082.8516, 37447.7354, 37815.9606, 38191.0692, 38559.4106, 38924.8112, 39294.6726, 39663.973, 40042.261, 40416.2036, 40779.2036, 41161.6436, 41540.9014, 41921.1998, 42294.7698, 42678.5264, 43061.3464, 43432.375, 43818.432, 44198.6598, 44583.0138, 44970.4794, 45353.924, 45729.858, 46118.2224, 46511.5724, 46900.7386, 47280.6964, 47668.1472, 48055.6796, 48446.9436, 48838.7146, 49217.7296, 49613.7796, 50010.7508, 50410.0208, 50793.7886, 51190.2456, 51583.1882, 51971.0796, 52376.5338, 52763.319, 53165.5534, 53556.5594, 53948.2702, 54346.352, 54748.7914, 55138.577, 55543.4824, 55941.1748, 56333.7746, 56745.1552, 57142.7944, 57545.2236, 57935.9956, 58348.5268, 58737.5474, 59158.5962, 59542.6896, 59958.8004, 60349.3788, 60755.0212, 61147.6144, 61548.194, 61946.0696, 62348.6042, 62763.603, 63162.781, 63560.635, 63974.3482, 64366.4908, 64771.5876, 65176.7346, 65597.3916, 65995.915, 66394.0384, 66822.9396, 67203.6336, 67612.2032, 68019.0078, 68420.0388, 68821.22, 69235.8388, 69640.0724, 70055.155, 70466.357, 70863.4266, 71276.2482, 71677.0306, 72080.2006, 72493.0214, 72893.5952, 73314.5856, 73714.9852, 74125.3022, 74521.2122, 74933.6814, 75341.5904, 75743.0244, 76166.0278, 76572.1322, 76973.1028, 77381.6284, 77800.6092, 78189.328, 78607.0962, 79012.2508, 79407.8358, 79825.725, 80238.701, 80646.891, 81035.6436, 81460.0448, 81876.3884};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p15[] = {23635.0036, 24030.8034, 24431.4744, 24837.1524, 25246.7928, 25661.326, 26081.3532, 26505.2806, 26933.9892, 27367.7098, 27805.318, 28248.799, 28696.4382, 29148.8244, 29605.5138, 30066.8668, 30534.2344, 31006.32, 31480.778, 31962.2418, 32447.3324, 32938.0232, 33432.731, 33930.728, 34433.9896, 34944.1402, 35457.5588, 35974.5958, 36497.3296, 37021.9096, 37554.326, 38088.0826, 38628.8816, 39171.3192, 39723.2326, 40274.5554, 40832.3142, 41390.613, 41959.5908, 42532.5466, 43102.0344, 43683.5072, 44266.694, 44851.2822, 45440.7862, 46038.0586, 46640.3164, 47241.064, 47846.155, 48454.7396, 49076.9168, 49692.542, 50317.4778, 50939.65, 51572.5596, 52210.2906, 52843.7396, 53481.3996, 54127.236, 54770.406, 55422.6598, 56078.7958, 56736.7174, 57397.6784, 58064.5784, 58730.308, 59404.9784, 60077.0864, 60751.9158, 61444.1386, 62115.817, 62808.7742, 63501.4774, 64187.5454, 64883.6622, 65582.7468, 66274.5318, 66976.9276, 67688.7764, 68402.138, 69109.6274, 69822.9706, 70543.6108, 71265.5202, 71983.3848, 72708.4656, 73433.384, 74158.4664, 74896.4868, 75620.9564, 76362.1434, 77098.3204, 77835.7662, 78582.6114, 79323.9902, 80067.8658, 80814.9246, 81567.0136, 82310.8536, 83061.9952, 83821.4096, 84580.8608, 85335.547, 86092.5802, 86851.6506, 87612.311, 88381.2016, 89146.3296, 89907.8974, 90676.846, 91451.4152, 92224.5518, 92995.8686, 93763.5066, 94551.2796, 95315.1944, 96096.1806, 96881.0918, 97665.679, 98442.68, 99229.3002, 100011.0994, 100790.6386, 101580.1564, 102377.7484, 103152.1392, 103944.2712, 104730.216, 105528.6336, 106324.9398, 107117.6706, 107890.3988, 108695.2266, 109485.238, 110294.7876, 111075.0958, 111878.0496, 112695.2864, 113464.5486, 114270.0474, 115068.608, 115884.3626, 116673.2588, 117483.3716, 118275.097, 119085.4092, 119879.2808, 120687.5868, 121499.9944, 122284.916, 123095.9254, 123912.5038, 124709.0454, 125503.7182, 126323.259, 127138.9412, 127943.8294, 128755.646, 129556.5354, 130375.3298, 131161.4734, 131971.1962, 132787.5458, 133588.1056, 134431.351, 135220.2906, 136023.398, 136846.6558, 137667.0004, 138463.663, 139283.7154, 140074.6146, 140901.3072, 141721.8548, 142543.2322, 143356.1096, 144173.7412, 144973.0948, 145794.3162, 146609.5714, 147420.003, 148237.9784, 149050.5696, 149854.761, 150663.1966, 151494.0754, 152313.1416, 153112.6902, 153935.7206, 154746.9262, 155559.547, 156401.9746, 157228.7036, 158008.7254, 158820.75, 159646.9184, 160470.4458, 161279.5348, 162093.3114, 162918.542, 163729.2842};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p16[] = {47271.0, 48062.3584, 48862.7074, 49673.152, 50492.8416, 51322.9514, 52161.03, 53009.407, 53867.6348, 54734.206, 55610.5144, 56496.2096, 57390.795, 58297.268, 59210.6448, 60134.665, 61068.0248, 62010.4472, 62962.5204, 63923.5742, 64895.0194, 65876.4182, 66862.6136, 67862.6968, 68868.8908, 69882.8544, 70911.271, 71944.0924, 72990.0326, 74040.692, 75100.6336, 76174.7826, 77252.5998, 78340.2974, 79438.2572, 80545.4976, 81657.2796, 82784.6336, 83915.515, 85059.7362, 86205.9368, 87364.4424, 88530.3358, 89707.3744, 90885.9638, 92080.197, 93275.5738, 94479.391, 95695.918, 96919.2236, 98148.4602, 99382.3474, 100625.6974, 101878.0284, 103141.6278, 104409.4588, 105686.2882, 106967.5402, 108261.6032, 109548.1578, 110852.0728, 112162.231, 113479.0072, 114806.2626, 116137.9072, 117469.5048, 118813.5186, 120165.4876, 121516.2556, 122875.766, 124250.5444, 125621.2222, 127003.2352, 128387.848, 129775.2644, 131181.7776, 132577.3086, 133979.9458, 135394.1132, 136800.9078, 138233.217, 139668.5308, 141085.212, 142535.2122, 143969.0684, 145420.2872, 146878.1542, 148332.7572, 149800.3202, 151269.66, 152743.6104, 154213.0948, 155690.288, 157169.4246, 158672.1756, 160160.059, 161650.6854, 163145.7772, 164645.6726, 166159.1952, 167682.1578, 169177.3328, 170700.0118, 172228.8964, 173732.6664, 175265.5556, 176787.799, 178317.111, 179856.6914, 181400.865, 182943.4612, 184486.742, 186033.4698, 187583.7886, 189148.1868, 190688.4526, 192250.1926, 193810.9042, 195354.2972, 196938.7682, 198493.5898, 200079.2824, 201618.912, 203205.5492, 204765.5798, 206356.1124, 207929.3064, 209498.7196, 211086.229, 212675.1324, 214256.7892, 215826.2392, 217412.8474, 218995.6724, 220618.6038, 222207.1166, 223781.0364, 225387.4332, 227005.7928, 228590.4336, 230217.8738, 231805.1054, 233408.9, 234995.3432, 236601.4956, 238190.7904, 239817.2548, 241411.2832, 243002.4066, 244640.1884, 246255.3128, 247849.3508, 249479.9734, 251106.8822, 252705.027, 254332.9242, 255935.129, 257526.9014, 259154.772, 260777.625, 262390.253, 264004.4906, 265643.59, 267255.4076, 268873.426, 270470.7252, 272106.4804, 273722.4456, 275337.794, 276945.7038, 278592.9154, 280204.3726, 281841.1606, 283489.171, 285130.1716, 286735.3362, 288364.7164, 289961.1814, 291595.5524, 293285.683, 294899.6668, 296499.3434, 298128.0462, 299761.8946, 301394.2424, 302997.6748, 304615.1478, 306269.7724, 307886.114, 309543.1028, 311153.2862, 312782.8546, 314421.2008, 316033.2438, 317692.9636, 319305.2648, 320948.7406, 322566.3364, 324228.4224, 325847.1542};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p17[] = {94542.0, 96125.811, 97728.019, 99348.558, 100987.9705, 102646.7565, 104324.5125, 106021.7435, 107736.7865, 109469.272, 111223.9465, 112995.219, 114787.432, 116593.152, 118422.71, 120267.2345, 122134.6765, 124020.937, 125927.2705, 127851.255, 129788.9485, 131751.016, 133726.8225, 135722.592, 137736.789, 139770.568, 141821.518, 143891.343, 145982.1415, 148095.387, 150207.526, 152355.649, 154515.6415, 156696.05, 158887.7575, 161098.159, 163329.852, 165569.053, 167837.4005, 170121.6165, 172420.4595, 174732.6265, 177062.77, 179412.502, 181774.035, 184151.939, 186551.6895, 188965.691, 191402.8095, 193857.949, 196305.0775, 198774.6715, 201271.2585, 203764.78, 206299.3695, 208818.1365, 211373.115, 213946.7465, 216532.076, 219105.541, 221714.5375, 224337.5135, 226977.5125, 229613.0655, 232270.2685, 234952.2065, 237645.3555, 240331.1925, 243034.517, 245756.0725, 248517.6865, 251232.737, 254011.3955, 256785.995, 259556.44, 262368.335, 265156.911, 267965.266, 270785.583, 273616.0495, 276487.4835, 279346.639, 282202.509, 285074.3885, 287942.2855, 290856.018, 293774.0345, 296678.5145, 299603.6355, 302552.6575, 305492.9785, 308466.8605, 311392.581, 314347.538, 317319.4295, 320285.9785, 323301.7325, 326298.3235, 329301.3105, 332301.987, 335309.791, 338370.762, 341382.923, 344431.1265, 347464.1545, 350507.28, 353619.2345, 356631.2005, 359685.203, 362776.7845, 365886.488, 368958.2255, 372060.6825, 375165.4335, 378237.935, 381328.311, 384430.5225, 387576.425, 390683.242, 393839.648, 396977.8425, 400101.9805, 403271.296, 406409.8425, 409529.5485, 412678.7, 415847.423, 419020.8035, 422157.081, 425337.749, 428479.6165, 431700.902, 434893.1915, 438049.582, 441210.5415, 444379.2545, 447577.356, 450741.931, 453959.548, 457137.0935, 460329.846, 463537.4815, 466732.3345, 469960.5615, 473164.681, 476347.6345, 479496.173, 482813.1645, 486025.6995, 489249.4885, 492460.1945, 495675.8805, 498908.0075, 502131.802, 505374.3855, 508550.9915, 511806.7305, 515026.776, 518217.0005, 521523.9855, 524705.9855, 527950.997, 531210.0265, 534472.497, 537750.7315, 540926.922, 544207.094, 547429.4345, 550666.3745, 553975.3475, 557150.7185, 560399.6165, 563662.697, 566916.7395, 570146.1215, 573447.425, 576689.6245, 579874.5745, 583202.337, 586503.0255, 589715.635, 592910.161, 596214.3885, 599488.035, 602740.92, 605983.0685, 609248.67, 612491.3605, 615787.912, 619107.5245, 622307.9555, 625577.333, 628840.4385, 632085.2155, 635317.6135, 638691.7195, 641887.467, 645139.9405, 648441.546, 651666.252, 654941.845};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __raw_estimate_data_p18[] = {189084.0, 192250.913, 195456.774, 198696.946, 201977.762, 205294.444, 208651.754, 212042.099, 215472.269, 218941.91, 222443.912, 225996.845, 229568.199, 233193.568, 236844.457, 240543.233, 244279.475, 248044.27, 251854.588, 255693.2, 259583.619, 263494.621, 267445.385, 271454.061, 275468.769, 279549.456, 283646.446, 287788.198, 291966.099, 296181.164, 300431.469, 304718.618, 309024.004, 313393.508, 317760.803, 322209.731, 326675.061, 331160.627, 335654.47, 340241.442, 344841.833, 349467.132, 354130.629, 358819.432, 363574.626, 368296.587, 373118.482, 377914.93, 382782.301, 387680.669, 392601.981, 397544.323, 402529.115, 407546.018, 412593.658, 417638.657, 422762.865, 427886.169, 433017.167, 438213.273, 443441.254, 448692.421, 453937.533, 459239.049, 464529.569, 469910.083, 475274.03, 480684.473, 486070.26, 491515.237, 496995.651, 502476.617, 507973.609, 513497.19, 519083.233, 524726.509, 530305.505, 535945.728, 541584.404, 547274.055, 552967.236, 558667.862, 564360.216, 570128.148, 575965.08, 581701.952, 587532.523, 593361.144, 599246.128, 605033.418, 610958.779, 616837.117, 622772.818, 628672.04, 634675.369, 640574.831, 646585.739, 652574.547, 658611.217, 664642.684, 670713.914, 676737.681, 682797.313, 688837.897, 694917.874, 701009.882, 707173.648, 713257.254, 719415.392, 725636.761, 731710.697, 737906.209, 744103.074, 750313.39, 756504.185, 762712.579, 768876.985, 775167.859, 781359.0, 787615.959, 793863.597, 800245.477, 806464.582, 812785.294, 819005.925, 825403.057, 831676.197, 837936.284, 844266.968, 850642.711, 856959.756, 863322.774, 869699.931, 876102.478, 882355.787, 888694.463, 895159.952, 901536.143, 907872.631, 914293.672, 920615.14, 927130.974, 933409.404, 939922.178, 946331.47, 952745.93, 959209.264, 965590.224, 972077.284, 978501.961, 984953.19, 991413.271, 997817.479, 1004222.658, 1010725.676, 1017177.138, 1023612.529, 1030098.236, 1036493.719, 1043112.207, 1049537.036, 1056008.096, 1062476.184, 1068942.337, 1075524.95, 1081932.864, 1088426.025, 1094776.005, 1101327.448, 1107901.673, 1114423.639, 1120884.602, 1127324.923, 1133794.24, 1140328.886, 1146849.376, 1153346.682, 1159836.502, 1166478.703, 1172953.304, 1179391.502, 1185950.982, 1192544.052, 1198913.41, 1205430.994, 1212015.525, 1218674.042, 1225121.683, 1231551.101, 1238126.379, 1244673.795, 1251260.649, 1257697.86, 1264320.983, 1270736.319, 1277274.694, 1283804.95, 1290211.514, 1296858.568, 1303455.691};
//! @brief Get raw estimate data array for a given precision
//!
//! @param __precision The precision value (4-18)
//! @return Pointer to the raw estimate data array for the given precision
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr const double* __raw_estimate_data(::cuda::std::int32_t __precision) noexcept {
switch (__precision) {
case 4: return __raw_estimate_data_p4;
case 5: return __raw_estimate_data_p5;
case 6: return __raw_estimate_data_p6;
case 7: return __raw_estimate_data_p7;
case 8: return __raw_estimate_data_p8;
case 9: return __raw_estimate_data_p9;
case 10: return __raw_estimate_data_p10;
case 11: return __raw_estimate_data_p11;
case 12: return __raw_estimate_data_p12;
case 13: return __raw_estimate_data_p13;
case 14: return __raw_estimate_data_p14;
case 15: return __raw_estimate_data_p15;
case 16: return __raw_estimate_data_p16;
case 17: return __raw_estimate_data_p17;
case 18: return __raw_estimate_data_p18;
default: return nullptr;
}
}
//! @brief Get size of raw estimate data array for a given precision
//!
//! @param __precision The precision value (4-18)
//! @return Size of the raw estimate data array for the given precision
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::size_t __raw_estimate_data_size(::cuda::std::int32_t __precision) noexcept {
constexpr auto __size_of_double = sizeof(double);
switch (__precision) {
case 4: return sizeof(__raw_estimate_data_p4) / __size_of_double;
case 5: return sizeof(__raw_estimate_data_p5) / __size_of_double;
case 6: return sizeof(__raw_estimate_data_p6) / __size_of_double;
case 7: return sizeof(__raw_estimate_data_p7) / __size_of_double;
case 8: return sizeof(__raw_estimate_data_p8) / __size_of_double;
case 9: return sizeof(__raw_estimate_data_p9) / __size_of_double;
case 10: return sizeof(__raw_estimate_data_p10) / __size_of_double;
case 11: return sizeof(__raw_estimate_data_p11) / __size_of_double;
case 12: return sizeof(__raw_estimate_data_p12) / __size_of_double;
case 13: return sizeof(__raw_estimate_data_p13) / __size_of_double;
case 14: return sizeof(__raw_estimate_data_p14) / __size_of_double;
case 15: return sizeof(__raw_estimate_data_p15) / __size_of_double;
case 16: return sizeof(__raw_estimate_data_p16) / __size_of_double;
case 17: return sizeof(__raw_estimate_data_p17) / __size_of_double;
case 18: return sizeof(__raw_estimate_data_p18) / __size_of_double;
default: return 0;
}
}
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p4[] = {10.0, 9.717, 9.207, 8.7896, 8.2882, 7.8204, 7.3772, 6.9342, 6.5202, 6.161, 5.7722, 5.4636, 5.0396, 4.6766, 4.3566, 4.0454, 3.7936, 3.4856, 3.2666, 2.9946, 2.766, 2.4692, 2.3638, 2.0764, 1.7864, 1.7602, 1.4814, 1.433, 1.2926, 1.0664, 0.999600000000001, 0.7956, 0.5366, 0.589399999999998, 0.573799999999999, 0.269799999999996, 0.368200000000002, 0.0544000000000011, 0.234200000000001, 0.0108000000000033, -0.203400000000002, -0.0701999999999998, -0.129600000000003, -0.364199999999997, -0.480600000000003, -0.226999999999997, -0.322800000000001, -0.382599999999996, -0.511200000000002, -0.669600000000003, -0.749400000000001, -0.500399999999999, -0.617600000000003, -0.6922, -0.601599999999998, -0.416200000000003, -0.338200000000001, -0.782600000000002, -0.648600000000002, -0.919800000000002, -0.851799999999997, -0.962400000000002, -0.6402, -1.1922, -1.0256, -1.086, -1.21899999999999, -0.819400000000002, -0.940600000000003, -1.1554, -1.2072, -1.1752, -1.16759999999999, -1.14019999999999, -1.3754, -1.29859999999999, -1.607, -1.3292, -1.7606};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p5[] = {22.0, 21.1194, 20.8208, 20.2318, 19.77, 19.2436, 18.7774, 18.2848, 17.8224, 17.3742, 16.9336, 16.503, 16.0494, 15.6292, 15.2124, 14.798, 14.367, 13.9728, 13.5944, 13.217, 12.8438, 12.3696, 12.0956, 11.7044, 11.324, 11.0668, 10.6698, 10.3644, 10.049, 9.6918, 9.4146, 9.082, 8.687, 8.5398, 8.2462, 7.857, 7.6606, 7.4168, 7.1248, 6.9222, 6.6804, 6.447, 6.3454, 5.9594, 5.7636, 5.5776, 5.331, 5.19, 4.9676, 4.7564, 4.5314, 4.4442, 4.3708, 3.9774, 3.9624, 3.8796, 3.755, 3.472, 3.2076, 3.1024, 2.8908, 2.7338, 2.7728, 2.629, 2.413, 2.3266, 2.1524, 2.2642, 2.1806, 2.0566, 1.9192, 1.7598, 1.3516, 1.5802, 1.43859999999999, 1.49160000000001, 1.1524, 1.1892, 0.841399999999993, 0.879800000000003, 0.837599999999995, 0.469800000000006, 0.765600000000006, 0.331000000000003, 0.591399999999993, 0.601200000000006, 0.701599999999999, 0.558199999999999, 0.339399999999998, 0.354399999999998, 0.491200000000006, 0.308000000000007, 0.355199999999996, -0.0254000000000048, 0.205200000000005, -0.272999999999996, 0.132199999999997, 0.394400000000005, -0.241200000000006, 0.242000000000004, 0.191400000000002, 0.253799999999998, -0.122399999999999, -0.370800000000003, 0.193200000000004, -0.0848000000000013, 0.0867999999999967, -0.327200000000005, -0.285600000000002, 0.311400000000006, -0.128399999999999, -0.754999999999995, -0.209199999999996, -0.293599999999998, -0.364000000000004, -0.253600000000006, -0.821200000000005, -0.253600000000006, -0.510400000000004, -0.383399999999995, -0.491799999999998, -0.220200000000006, -0.0972000000000008, -0.557400000000001, -0.114599999999996, -0.295000000000002, -0.534800000000004, 0.346399999999988, -0.65379999999999, 0.0398000000000138, 0.0341999999999985, -0.995800000000003, -0.523400000000009, -0.489000000000004, -0.274799999999999, -0.574999999999989, -0.482799999999997, 0.0571999999999946, -0.330600000000004, -0.628800000000012, -0.140199999999993, -0.540600000000012, -0.445999999999998, -0.599400000000003, -0.262599999999992, 0.163399999999996, -0.100599999999986, -0.39500000000001, -1.06960000000001, -0.836399999999998, -0.753199999999993, -0.412399999999991, -0.790400000000005, -0.29679999999999, -0.28540000000001, -0.193000000000012, -0.0772000000000048, -0.962799999999987, -0.414800000000014};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p6[] = {45.0, 44.1902, 43.271, 42.8358, 41.8142, 41.2854, 40.317, 39.354, 38.8924, 37.9436, 37.4596, 36.5262, 35.6248, 35.1574, 34.2822, 33.837, 32.9636, 32.074, 31.7042, 30.7976, 30.4772, 29.6564, 28.7942, 28.5004, 27.686, 27.291, 26.5672, 25.8556, 25.4982, 24.8204, 24.4252, 23.7744, 23.0786, 22.8344, 22.0294, 21.8098, 21.0794, 20.5732, 20.1878, 19.5648, 19.2902, 18.6784, 18.3352, 17.8946, 17.3712, 17.0852, 16.499, 16.2686, 15.6844, 15.2234, 14.9732, 14.3356, 14.2286, 13.7262, 13.3284, 13.1048, 12.5962, 12.3562, 12.1272, 11.4184, 11.4974, 11.0822, 10.856, 10.48, 10.2834, 10.0208, 9.637, 9.51739999999999, 9.05759999999999, 8.74760000000001, 8.42700000000001, 8.1326, 8.2372, 8.2788, 7.6776, 7.79259999999999, 7.1952, 6.9564, 6.6454, 6.87, 6.5428, 6.19999999999999, 6.02940000000001, 5.62780000000001, 5.6782, 5.792, 5.35159999999999, 5.28319999999999, 5.0394, 5.07480000000001, 4.49119999999999, 4.84899999999999, 4.696, 4.54040000000001, 4.07300000000001, 4.37139999999999, 3.7216, 3.7328, 3.42080000000001, 3.41839999999999, 3.94239999999999, 3.27719999999999, 3.411, 3.13079999999999, 2.76900000000001, 2.92580000000001, 2.68279999999999, 2.75020000000001, 2.70599999999999, 2.3886, 3.01859999999999, 2.45179999999999, 2.92699999999999, 2.41720000000001, 2.41139999999999, 2.03299999999999, 2.51240000000001, 2.5564, 2.60079999999999, 2.41720000000001, 1.80439999999999, 1.99700000000001, 2.45480000000001, 1.8948, 2.2346, 2.30860000000001, 2.15479999999999, 1.88419999999999, 1.6508, 0.677199999999999, 1.72540000000001, 1.4752, 1.72280000000001, 1.66139999999999, 1.16759999999999, 1.79300000000001, 1.00059999999999, 0.905200000000008, 0.659999999999997, 1.55879999999999, 1.1636, 0.688199999999995, 0.712600000000009, 0.450199999999995, 1.1978, 0.975599999999986, 0.165400000000005, 1.727, 1.19739999999999, -0.252600000000001, 1.13460000000001, 1.3048, 1.19479999999999, 0.313400000000001, 0.878999999999991, 1.12039999999999, 0.853000000000009, 1.67920000000001, 0.856999999999999, 0.448599999999999, 1.2362, 0.953399999999988, 1.02859999999998, 0.563199999999995, 0.663000000000011, 0.723000000000013, 0.756599999999992, 0.256599999999992, -0.837600000000009, 0.620000000000005, 0.821599999999989, 0.216600000000028, 0.205600000000004, 0.220199999999977, 0.372599999999977, 0.334400000000016, 0.928400000000011, 0.972800000000007, 0.192400000000021, 0.487199999999973, -0.413000000000011, 0.807000000000016, 0.120600000000024, 0.769000000000005, 0.870799999999974, 0.66500000000002, 0.118200000000002, 0.401200000000017, 0.635199999999998, 0.135400000000004, 0.175599999999974, 1.16059999999999, 0.34620000000001, 0.521400000000028, -0.586599999999976, -1.16480000000001, 0.968399999999974, 0.836999999999989, 0.779600000000016, 0.985799999999983};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p7[] = {91.0, 89.4934, 87.9758, 86.4574, 84.9718, 83.4954, 81.5302, 80.0756, 78.6374, 77.1782, 75.7888, 73.9522, 72.592, 71.2532, 69.9086, 68.5938, 66.9474, 65.6796, 64.4394, 63.2176, 61.9768, 60.4214, 59.2528, 58.0102, 56.8658, 55.7278, 54.3044, 53.1316, 52.093, 51.0032, 49.9092, 48.6306, 47.5294, 46.5756, 45.6508, 44.662, 43.552, 42.3724, 41.617, 40.5754, 39.7872, 38.8444, 37.7988, 36.8606, 36.2118, 35.3566, 34.4476, 33.5882, 32.6816, 32.0824, 31.0258, 30.6048, 29.4436, 28.7274, 27.957, 27.147, 26.4364, 25.7592, 25.3386, 24.781, 23.8028, 23.656, 22.6544, 21.996, 21.4718, 21.1544, 20.6098, 19.5956, 19.0616, 18.5758, 18.4878, 17.5244, 17.2146, 16.724, 15.8722, 15.5198, 15.0414, 14.941, 14.9048, 13.87, 13.4304, 13.028, 12.4708, 12.37, 12.0624, 11.4668, 11.5532, 11.4352, 11.2564, 10.2744, 10.2118, 9.74720000000002, 10.1456, 9.2928, 8.75040000000001, 8.55279999999999, 8.97899999999998, 8.21019999999999, 8.18340000000001, 7.3494, 7.32499999999999, 7.66140000000001, 6.90300000000002, 7.25439999999998, 6.9042, 7.21499999999997, 6.28640000000001, 6.08139999999997, 6.6764, 6.30099999999999, 5.13900000000001, 5.65800000000002, 5.17320000000001, 4.59019999999998, 4.9538, 5.08280000000002, 4.92200000000003, 4.99020000000002, 4.7328, 5.4538, 4.11360000000002, 4.22340000000003, 4.08780000000002, 3.70800000000003, 4.15559999999999, 4.18520000000001, 3.63720000000001, 3.68220000000002, 3.77960000000002, 3.6078, 2.49160000000001, 3.13099999999997, 2.5376, 3.19880000000001, 3.21100000000001, 2.4502, 3.52820000000003, 2.91199999999998, 3.04480000000001, 2.7432, 2.85239999999999, 2.79880000000003, 2.78579999999999, 1.88679999999999, 2.98860000000002, 2.50639999999999, 1.91239999999999, 2.66160000000002, 2.46820000000002, 1.58199999999999, 1.30399999999997, 2.27379999999999, 2.68939999999998, 1.32900000000001, 3.10599999999999, 1.69080000000002, 2.13740000000001, 2.53219999999999, 1.88479999999998, 1.33240000000001, 1.45119999999997, 1.17899999999997, 2.44119999999998, 1.60659999999996, 2.16700000000003, 0.77940000000001, 2.37900000000002, 2.06700000000001, 1.46000000000004, 2.91160000000002, 1.69200000000001, 0.954600000000028, 2.49300000000005, 2.2722, 1.33500000000004, 2.44899999999996, 1.20140000000004, 3.07380000000001, 2.09739999999999, 2.85640000000001, 2.29960000000005, 2.40899999999999, 1.97040000000004, 0.809799999999996, 1.65279999999996, 2.59979999999996, 0.95799999999997, 2.06799999999998, 2.32780000000002, 4.20159999999998, 1.96320000000003, 1.86400000000003, 1.42999999999995, 3.77940000000001, 1.27200000000005, 1.86440000000005, 2.20600000000002, 3.21900000000005, 1.5154, 2.61019999999996};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p8[] = {183.2152, 180.2454, 177.2096, 173.6652, 170.6312, 167.6822, 164.249, 161.3296, 158.0038, 155.2074, 152.4612, 149.27, 146.5178, 143.4412, 140.8032, 138.1634, 135.1688, 132.6074, 129.6946, 127.2664, 124.8228, 122.0432, 119.6824, 116.9464, 114.6268, 112.2626, 109.8376, 107.4034, 104.8956, 102.8522, 100.7638, 98.3552, 96.3556, 93.7526, 91.9292, 89.8954, 87.8198, 85.7668, 83.298, 81.6688, 79.9466, 77.9746, 76.1672, 74.3474, 72.3028, 70.8912, 69.114, 67.4646, 65.9744, 64.4092, 62.6022, 60.843, 59.5684, 58.1652, 56.5426, 55.4152, 53.5388, 52.3592, 51.1366, 49.486, 48.3918, 46.5076, 45.509, 44.3834, 43.3498, 42.0668, 40.7346, 40.1228, 38.4528, 37.7, 36.644, 36.0518, 34.5774, 33.9068, 32.432, 32.1666, 30.434, 29.6644, 28.4894, 27.6312, 26.3804, 26.292, 25.5496000000001, 25.0234, 24.8206, 22.6146, 22.4188, 22.117, 20.6762, 20.6576, 19.7864, 19.509, 18.5334, 17.9204, 17.772, 16.2924, 16.8654, 15.1836, 15.745, 15.1316, 15.0386, 14.0136, 13.6342, 12.6196, 12.1866, 12.4281999999999, 11.3324, 10.4794000000001, 11.5038, 10.129, 9.52800000000002, 10.3203999999999, 9.46299999999997, 9.79280000000006, 9.12300000000005, 8.74180000000001, 9.2192, 7.51020000000005, 7.60659999999996, 7.01840000000004, 7.22239999999999, 7.40139999999997, 6.76179999999999, 7.14359999999999, 5.65060000000005, 5.63779999999997, 5.76599999999996, 6.75139999999999, 5.57759999999996, 3.73220000000003, 5.8048, 5.63019999999995, 4.93359999999996, 3.47979999999995, 4.33879999999999, 3.98940000000005, 3.81960000000004, 3.31359999999995, 3.23080000000004, 3.4588, 3.08159999999998, 3.4076, 3.00639999999999, 2.38779999999997, 2.61900000000003, 1.99800000000005, 3.34820000000002, 2.95060000000001, 0.990999999999985, 2.11440000000005, 2.20299999999997, 2.82219999999995, 2.73239999999998, 2.7826, 3.76660000000004, 2.26480000000004, 2.31280000000004, 2.40819999999997, 2.75360000000001, 3.33759999999995, 2.71559999999999, 1.7478000000001, 1.42920000000004, 2.39300000000003, 2.22779999999989, 2.34339999999997, 0.87259999999992, 3.88400000000001, 1.80600000000004, 1.91759999999999, 1.16779999999994, 1.50320000000011, 2.52500000000009, 0.226400000000012, 2.31500000000005, 0.930000000000064, 1.25199999999995, 2.14959999999996, 0.0407999999999902, 2.5447999999999, 1.32960000000003, 0.197400000000016, 2.52620000000002, 3.33279999999991, -1.34300000000007, 0.422199999999975, 0.917200000000093, 1.12920000000008, 1.46060000000011, 1.45779999999991, 2.8728000000001, 3.33359999999993, -1.34079999999994, 1.57680000000005, 0.363000000000056, 1.40740000000005, 0.656600000000026, 0.801400000000058, -0.454600000000028, 1.51919999999996};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p9[] = {368.0, 361.8294, 355.2452, 348.6698, 342.1464, 336.2024, 329.8782, 323.6598, 317.462, 311.2826, 305.7102, 299.7416, 293.9366, 288.1046, 282.285, 277.0668, 271.306, 265.8448, 260.301, 254.9886, 250.2422, 244.8138, 239.7074, 234.7428, 229.8402, 225.1664, 220.3534, 215.594, 210.6886, 205.7876, 201.65, 197.228, 192.8036, 188.1666, 184.0818, 180.0824, 176.2574, 172.302, 168.1644, 164.0056, 160.3802, 156.7192, 152.5234, 149.2084, 145.831, 142.485, 139.1112, 135.4764, 131.76, 129.3368, 126.5538, 122.5058, 119.2646, 116.5902, 113.3818, 110.8998, 107.9532, 105.2062, 102.2798, 99.4728, 96.9582, 94.3292, 92.171, 89.7809999999999, 87.5716, 84.7048, 82.5322, 79.875, 78.3972, 75.3464, 73.7274, 71.2834, 70.1444, 68.4263999999999, 66.0166, 64.018, 62.0437999999999, 60.3399999999999, 58.6856, 57.9836, 55.0311999999999, 54.6769999999999, 52.3188, 51.4846, 49.4423999999999, 47.739, 46.1487999999999, 44.9202, 43.4059999999999, 42.5342000000001, 41.2834, 38.8954000000001, 38.3286000000001, 36.2146, 36.6684, 35.9946, 33.123, 33.4338, 31.7378000000001, 29.076, 28.9692, 27.4964, 27.0998, 25.9864, 26.7754, 24.3208, 23.4838, 22.7388000000001, 24.0758000000001, 21.9097999999999, 20.9728, 19.9228000000001, 19.9292, 16.617, 17.05, 18.2996000000001, 15.6128000000001, 15.7392, 14.5174, 13.6322, 12.2583999999999, 13.3766000000001, 11.423, 13.1232, 9.51639999999998, 10.5938000000001, 9.59719999999993, 8.12220000000002, 9.76739999999995, 7.50440000000003, 7.56999999999994, 6.70440000000008, 6.41419999999994, 6.71019999999999, 5.60940000000005, 4.65219999999999, 6.84099999999989, 3.4072000000001, 3.97859999999991, 3.32760000000007, 5.52160000000003, 3.31860000000006, 2.06940000000009, 4.35400000000004, 1.57500000000005, 0.280799999999999, 2.12879999999996, -0.214799999999968, -0.0378000000000611, -0.658200000000079, 0.654800000000023, -0.0697999999999865, 0.858400000000074, -2.52700000000004, -2.1751999999999, -3.35539999999992, -1.04019999999991, -0.651000000000067, -2.14439999999991, -1.96659999999997, -3.97939999999994, -0.604400000000169, -3.08260000000018, -3.39159999999993, -5.29640000000018, -5.38920000000007, -5.08759999999984, -4.69900000000007, -5.23720000000003, -3.15779999999995, -4.97879999999986, -4.89899999999989, -7.48880000000008, -5.94799999999987, -5.68060000000014, -6.67180000000008, -4.70499999999993, -7.27779999999984, -4.6579999999999, -4.4362000000001, -4.32139999999981, -5.18859999999995, -6.66879999999992, -6.48399999999992, -5.1260000000002, -4.4032000000002, -6.13500000000022, -5.80819999999994, -4.16719999999987, -4.15039999999999, -7.45600000000013, -7.24080000000004, -9.83179999999993, -5.80420000000004, -8.6561999999999, -6.99940000000015, -10.5473999999999, -7.34139999999979, -6.80999999999995, -6.29719999999998, -6.23199999999997};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p10[] = {737.1256, 724.4234, 711.1064, 698.4732, 685.4636, 673.0644, 660.488, 647.9654, 636.0832, 623.7864, 612.1992, 600.2176, 588.5228, 577.1716, 565.7752, 554.899, 543.6126, 532.6492, 521.9474, 511.5214, 501.1064, 490.6364, 480.2468, 470.4588, 460.3832, 451.0584, 440.8606, 431.3868, 422.5062, 413.1862, 404.463, 395.339, 386.1936, 378.1292, 369.1854, 361.2908, 353.3324, 344.8518, 337.5204, 329.4854, 321.9318, 314.552, 306.4658, 299.4256, 292.849, 286.152, 278.8956, 271.8792, 265.118, 258.62, 252.5132, 245.9322, 239.7726, 233.6086, 227.5332, 222.5918, 216.4294, 210.7662, 205.4106, 199.7338, 194.9012, 188.4486, 183.1556, 178.6338, 173.7312, 169.6264, 163.9526, 159.8742, 155.8326, 151.1966, 147.5594, 143.07, 140.037, 134.1804, 131.071, 127.4884, 124.0848, 120.2944, 117.333, 112.9626, 110.2902, 107.0814, 103.0334, 99.4832000000001, 96.3899999999999, 93.7202000000002, 90.1714000000002, 87.2357999999999, 85.9346, 82.8910000000001, 80.0264000000002, 78.3834000000002, 75.1543999999999, 73.8683999999998, 70.9895999999999, 69.4367999999999, 64.8701999999998, 65.0408000000002, 61.6738, 59.5207999999998, 57.0158000000001, 54.2302, 53.0962, 50.4985999999999, 52.2588000000001, 47.3914, 45.6244000000002, 42.8377999999998, 43.0072, 40.6516000000001, 40.2453999999998, 35.2136, 36.4546, 33.7849999999999, 33.2294000000002, 32.4679999999998, 30.8670000000002, 28.6507999999999, 28.9099999999999, 27.5983999999999, 26.1619999999998, 24.5563999999999, 23.2328000000002, 21.9484000000002, 21.5902000000001, 21.3346000000001, 17.7031999999999, 20.6111999999998, 19.5545999999999, 15.7375999999999, 17.0720000000001, 16.9517999999998, 15.326, 13.1817999999998, 14.6925999999999, 13.0859999999998, 13.2754, 10.8697999999999, 11.248, 7.3768, 4.72339999999986, 7.97899999999981, 8.7503999999999, 7.68119999999999, 9.7199999999998, 7.73919999999998, 5.6224000000002, 7.44560000000001, 6.6601999999998, 5.9058, 4.00199999999995, 4.51699999999983, 4.68240000000014, 3.86220000000003, 5.13639999999987, 5.98500000000013, 2.47719999999981, 2.61999999999989, 1.62800000000016, 4.65000000000009, 0.225599999999758, 0.831000000000131, -0.359400000000278, 1.27599999999984, -2.92559999999958, -0.0303999999996449, 2.37079999999969, -2.0033999999996, 0.804600000000391, 0.30199999999968, 1.1247999999996, -2.6880000000001, 0.0321999999996478, -1.18099999999959, -3.9402, -1.47940000000017, -0.188400000000001, -2.10720000000038, -2.04159999999956, -3.12880000000041, -4.16160000000036, -0.612799999999879, -3.48719999999958, -8.17900000000009, -5.37780000000021, -4.01379999999972, -5.58259999999973, -5.73719999999958, -7.66799999999967, -5.69520000000011, -1.1247999999996, -5.58520000000044, -8.04560000000038, -4.64840000000004, -11.6468000000004, -7.97519999999986, -5.78300000000036, -7.67420000000038, -10.6328000000003, -9.81720000000041};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p11[] = {1476.0, 1449.6014, 1423.5802, 1397.7942, 1372.3042, 1347.2062, 1321.8402, 1297.2292, 1272.9462, 1248.9926, 1225.3026, 1201.4252, 1178.0578, 1155.6092, 1132.626, 1110.5568, 1088.527, 1066.5154, 1045.1874, 1024.3878, 1003.37, 982.1972, 962.5728, 942.1012, 922.9668, 903.292, 884.0772, 864.8578, 846.6562, 828.041, 809.714, 792.3112, 775.1806, 757.9854, 740.656, 724.346, 707.5154, 691.8378, 675.7448, 659.6722, 645.5722, 630.1462, 614.4124, 600.8728, 585.898, 572.408, 558.4926, 544.4938, 531.6776, 517.282, 505.7704, 493.1012, 480.7388, 467.6876, 456.1872, 445.5048, 433.0214, 420.806, 411.409, 400.4144, 389.4294, 379.2286, 369.651, 360.6156, 350.337, 342.083, 332.1538, 322.5094, 315.01, 305.6686, 298.1678, 287.8116, 280.9978, 271.9204, 265.3286, 257.5706, 249.6014, 242.544, 235.5976, 229.583, 220.9438, 214.672, 208.2786, 201.8628, 195.1834, 191.505, 186.1816, 178.5188, 172.2294, 167.8908, 161.0194, 158.052, 151.4588, 148.1596, 143.4344, 138.5238, 133.13, 127.6374, 124.8162, 118.7894, 117.3984, 114.6078, 109.0858, 105.1036, 103.6258, 98.6018000000004, 95.7618000000002, 93.5821999999998, 88.5900000000001, 86.9992000000002, 82.8800000000001, 80.4539999999997, 74.6981999999998, 74.3644000000004, 73.2914000000001, 65.5709999999999, 66.9232000000002, 65.1913999999997, 62.5882000000001, 61.5702000000001, 55.7035999999998, 56.1764000000003, 52.7596000000003, 53.0302000000001, 49.0609999999997, 48.4694, 44.933, 46.0474000000004, 44.7165999999997, 41.9416000000001, 39.9207999999999, 35.6328000000003, 35.5276000000003, 33.1934000000001, 33.2371999999996, 33.3864000000003, 33.9228000000003, 30.2371999999996, 29.1373999999996, 25.2272000000003, 24.2942000000003, 19.8338000000003, 18.9005999999999, 23.0907999999999, 21.8544000000002, 19.5176000000001, 15.4147999999996, 16.9314000000004, 18.6737999999996, 12.9877999999999, 14.3688000000002, 12.0447999999997, 15.5219999999999, 12.5299999999997, 14.5940000000001, 14.3131999999996, 9.45499999999993, 12.9441999999999, 3.91139999999996, 13.1373999999996, 5.44720000000052, 9.82779999999912, 7.87279999999919, 3.67760000000089, 5.46980000000076, 5.55099999999948, 5.65979999999945, 3.89439999999922, 3.1275999999998, 5.65140000000065, 6.3062000000009, 3.90799999999945, 1.87060000000019, 5.17020000000048, 2.46680000000015, 0.770000000000437, -3.72340000000077, 1.16400000000067, 8.05340000000069, 0.135399999999208, 2.15940000000046, 0.766999999999825, 1.0594000000001, 3.15500000000065, -0.287399999999252, 2.37219999999979, -2.86620000000039, -1.63199999999961, -2.22979999999916, -0.15519999999924, -1.46039999999994, -0.262199999999211, -2.34460000000036, -2.8078000000005, -3.22179999999935, -5.60159999999996, -8.42200000000048, -9.43740000000071, 0.161799999999857, -10.4755999999998, -10.0823999999993};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p12[] = {2953.0, 2900.4782, 2848.3568, 2796.3666, 2745.324, 2694.9598, 2644.648, 2595.539, 2546.1474, 2498.2576, 2450.8376, 2403.6076, 2357.451, 2311.38, 2266.4104, 2221.5638, 2176.9676, 2134.193, 2090.838, 2048.8548, 2007.018, 1966.1742, 1925.4482, 1885.1294, 1846.4776, 1807.4044, 1768.8724, 1731.3732, 1693.4304, 1657.5326, 1621.949, 1586.5532, 1551.7256, 1517.6182, 1483.5186, 1450.4528, 1417.865, 1385.7164, 1352.6828, 1322.6708, 1291.8312, 1260.9036, 1231.476, 1201.8652, 1173.6718, 1145.757, 1119.2072, 1092.2828, 1065.0434, 1038.6264, 1014.3192, 988.5746, 965.0816, 940.1176, 917.9796, 894.5576, 871.1858, 849.9144, 827.1142, 805.0818, 783.9664, 763.9096, 742.0816, 724.3962, 706.3454, 688.018, 667.4214, 650.3106, 633.0686, 613.8094, 597.818, 581.4248, 563.834, 547.363, 531.5066, 520.455400000001, 505.583199999999, 488.366, 476.480799999999, 459.7682, 450.0522, 434.328799999999, 423.952799999999, 408.727000000001, 399.079400000001, 387.252200000001, 373.987999999999, 360.852000000001, 351.6394, 339.642, 330.902400000001, 322.661599999999, 311.662200000001, 301.3254, 291.7484, 279.939200000001, 276.7508, 263.215200000001, 254.811400000001, 245.5494, 242.306399999999, 234.8734, 223.787200000001, 217.7156, 212.0196, 200.793, 195.9748, 189.0702, 182.449199999999, 177.2772, 170.2336, 164.741, 158.613600000001, 155.311, 147.5964, 142.837, 137.3724, 132.0162, 130.0424, 121.9804, 120.451800000001, 114.8968, 111.585999999999, 105.933199999999, 101.705, 98.5141999999996, 95.0488000000005, 89.7880000000005, 91.4750000000004, 83.7764000000006, 80.9698000000008, 72.8574000000008, 73.1615999999995, 67.5838000000003, 62.6263999999992, 63.2638000000006, 66.0977999999996, 52.0843999999997, 58.9956000000002, 47.0912000000008, 46.4956000000002, 48.4383999999991, 47.1082000000006, 43.2392, 37.2759999999998, 40.0283999999992, 35.1864000000005, 35.8595999999998, 32.0998, 28.027, 23.6694000000007, 33.8266000000003, 26.3736000000008, 27.2008000000005, 21.3245999999999, 26.4115999999995, 23.4521999999997, 19.5013999999992, 19.8513999999996, 10.7492000000002, 18.6424000000006, 13.1265999999996, 18.2436000000016, 6.71860000000015, 3.39459999999963, 6.33759999999893, 7.76719999999841, 0.813999999998487, 3.82819999999992, 0.826199999999517, 8.07440000000133, -1.59080000000176, 5.01780000000144, 0.455399999998917, -0.24199999999837, 0.174800000000687, -9.07640000000174, -4.20160000000033, -3.77520000000004, -4.75179999999818, -5.3724000000002, -8.90680000000066, -6.10239999999976, -5.74120000000039, -9.95339999999851, -3.86339999999836, -13.7304000000004, -16.2710000000006, -7.51359999999841, -3.30679999999847, -13.1339999999982, -10.0551999999989, -6.72019999999975, -8.59660000000076, -10.9307999999983, -1.8775999999998, -4.82259999999951, -13.7788, -21.6470000000008, -10.6735999999983, -15.7799999999988};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p13[] = {5907.5052, 5802.2672, 5697.347, 5593.5794, 5491.2622, 5390.5514, 5290.3376, 5191.6952, 5093.5988, 4997.3552, 4902.5972, 4808.3082, 4715.5646, 4624.109, 4533.8216, 4444.4344, 4356.3802, 4269.2962, 4183.3784, 4098.292, 4014.79, 3932.4574, 3850.6036, 3771.2712, 3691.7708, 3615.099, 3538.1858, 3463.4746, 3388.8496, 3315.6794, 3244.5448, 3173.7516, 3103.3106, 3033.6094, 2966.5642, 2900.794, 2833.7256, 2769.81, 2707.3196, 2644.0778, 2583.9916, 2523.4662, 2464.124, 2406.073, 2347.0362, 2292.1006, 2238.1716, 2182.7514, 2128.4884, 2077.1314, 2025.037, 1975.3756, 1928.933, 1879.311, 1831.0006, 1783.2144, 1738.3096, 1694.5144, 1649.024, 1606.847, 1564.7528, 1525.3168, 1482.5372, 1443.9668, 1406.5074, 1365.867, 1329.2186, 1295.4186, 1257.9716, 1225.339, 1193.2972, 1156.3578, 1125.8686, 1091.187, 1061.4094, 1029.4188, 1000.9126, 972.3272, 944.004199999999, 915.7592, 889.965, 862.834200000001, 840.4254, 812.598399999999, 785.924200000001, 763.050999999999, 741.793799999999, 721.466, 699.040799999999, 677.997200000002, 649.866999999998, 634.911800000002, 609.8694, 591.981599999999, 570.2922, 557.129199999999, 538.3858, 521.872599999999, 502.951400000002, 495.776399999999, 475.171399999999, 459.751, 439.995200000001, 426.708999999999, 413.7016, 402.3868, 387.262599999998, 372.0524, 357.050999999999, 342.5098, 334.849200000001, 322.529399999999, 311.613799999999, 295.848000000002, 289.273000000001, 274.093000000001, 263.329600000001, 251.389599999999, 245.7392, 231.9614, 229.7952, 217.155200000001, 208.9588, 199.016599999999, 190.839199999999, 180.6976, 176.272799999999, 166.976999999999, 162.5252, 151.196400000001, 149.386999999999, 133.981199999998, 130.0586, 130.164000000001, 122.053400000001, 110.7428, 108.1276, 106.232400000001, 100.381600000001, 98.7668000000012, 86.6440000000002, 79.9768000000004, 82.4722000000002, 68.7026000000005, 70.1186000000016, 71.9948000000004, 58.998599999999, 59.0492000000013, 56.9818000000014, 47.5338000000011, 42.9928, 51.1591999999982, 37.2740000000013, 42.7220000000016, 31.3734000000004, 26.8090000000011, 25.8934000000008, 26.5286000000015, 29.5442000000003, 19.3503999999994, 26.0760000000009, 17.9527999999991, 14.8419999999969, 10.4683999999979, 8.65899999999965, 9.86720000000059, 4.34139999999752, -0.907800000000861, -3.32080000000133, -0.936199999996461, -11.9916000000012, -8.87000000000262, -6.33099999999831, -11.3366000000024, -15.9207999999999, -9.34659999999712, -15.5034000000014, -19.2097999999969, -15.357799999998, -28.2235999999975, -30.6898000000001, -19.3271999999997, -25.6083999999973, -24.409599999999, -13.6385999999984, -33.4473999999973, -32.6949999999997, -28.9063999999998, -31.7483999999968, -32.2935999999972, -35.8329999999987, -47.620600000002, -39.0855999999985, -33.1434000000008, -46.1371999999974, -37.5892000000022, -46.8164000000033, -47.3142000000007, -60.2914000000019, -37.7575999999972};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p14[] = {11816.475, 11605.0046, 11395.3792, 11188.7504, 10984.1814, 10782.0086, 10582.0072, 10384.503, 10189.178, 9996.2738, 9806.0344, 9617.9798, 9431.394, 9248.7784, 9067.6894, 8889.6824, 8712.9134, 8538.8624, 8368.4944, 8197.7956, 8031.8916, 7866.6316, 7703.733, 7544.5726, 7386.204, 7230.666, 7077.8516, 6926.7886, 6778.6902, 6631.9632, 6487.304, 6346.7486, 6206.4408, 6070.202, 5935.2576, 5799.924, 5671.0324, 5541.9788, 5414.6112, 5290.0274, 5166.723, 5047.6906, 4929.162, 4815.1406, 4699.127, 4588.5606, 4477.7394, 4369.4014, 4264.2728, 4155.9224, 4055.581, 3955.505, 3856.9618, 3761.3828, 3666.9702, 3575.7764, 3482.4132, 3395.0186, 3305.8852, 3221.415, 3138.6024, 3056.296, 2970.4494, 2896.1526, 2816.8008, 2740.2156, 2670.497, 2594.1458, 2527.111, 2460.8168, 2387.5114, 2322.9498, 2260.6752, 2194.2686, 2133.7792, 2074.767, 2015.204, 1959.4226, 1898.6502, 1850.006, 1792.849, 1741.4838, 1687.9778, 1638.1322, 1589.3266, 1543.1394, 1496.8266, 1447.8516, 1402.7354, 1361.9606, 1327.0692, 1285.4106, 1241.8112, 1201.6726, 1161.973, 1130.261, 1094.2036, 1048.2036, 1020.6436, 990.901400000002, 961.199800000002, 924.769800000002, 899.526400000002, 872.346400000002, 834.375, 810.432000000001, 780.659800000001, 756.013800000001, 733.479399999997, 707.923999999999, 673.858, 652.222399999999, 636.572399999997, 615.738599999997, 586.696400000001, 564.147199999999, 541.679600000003, 523.943599999999, 505.714599999999, 475.729599999999, 461.779600000002, 449.750800000002, 439.020799999998, 412.7886, 400.245600000002, 383.188199999997, 362.079599999997, 357.533799999997, 334.319000000003, 327.553399999997, 308.559399999998, 291.270199999999, 279.351999999999, 271.791400000002, 252.576999999997, 247.482400000001, 236.174800000001, 218.774599999997, 220.155200000001, 208.794399999999, 201.223599999998, 182.995600000002, 185.5268, 164.547400000003, 176.5962, 150.689599999998, 157.8004, 138.378799999999, 134.021200000003, 117.614399999999, 108.194000000003, 97.0696000000025, 89.6042000000016, 95.6030000000028, 84.7810000000027, 72.635000000002, 77.3482000000004, 59.4907999999996, 55.5875999999989, 50.7346000000034, 61.3916000000027, 50.9149999999936, 39.0384000000049, 58.9395999999979, 29.633600000001, 28.2032000000036, 26.0078000000067, 17.0387999999948, 9.22000000000116, 13.8387999999977, 8.07240000000456, 14.1549999999988, 15.3570000000036, 3.42660000000615, 6.24820000000182, -2.96940000000177, -8.79940000000352, -5.97860000000219, -14.4048000000039, -3.4143999999942, -13.0148000000045, -11.6977999999945, -25.7878000000055, -22.3185999999987, -24.409599999999, -31.9756000000052, -18.9722000000038, -22.8678000000073, -30.8972000000067, -32.3715999999986, -22.3907999999938, -43.6720000000059, -35.9038, -39.7492000000057, -54.1641999999993, -45.2749999999942, -42.2989999999991, -44.1089999999967, -64.3564000000042, -49.9551999999967, -42.6116000000038};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p15[] = {23634.0036, 23210.8034, 22792.4744, 22379.1524, 21969.7928, 21565.326, 21165.3532, 20770.2806, 20379.9892, 19994.7098, 19613.318, 19236.799, 18865.4382, 18498.8244, 18136.5138, 17778.8668, 17426.2344, 17079.32, 16734.778, 16397.2418, 16063.3324, 15734.0232, 15409.731, 15088.728, 14772.9896, 14464.1402, 14157.5588, 13855.5958, 13559.3296, 13264.9096, 12978.326, 12692.0826, 12413.8816, 12137.3192, 11870.2326, 11602.5554, 11340.3142, 11079.613, 10829.5908, 10583.5466, 10334.0344, 10095.5072, 9859.694, 9625.2822, 9395.7862, 9174.0586, 8957.3164, 8738.064, 8524.155, 8313.7396, 8116.9168, 7913.542, 7718.4778, 7521.65, 7335.5596, 7154.2906, 6968.7396, 6786.3996, 6613.236, 6437.406, 6270.6598, 6107.7958, 5945.7174, 5787.6784, 5635.5784, 5482.308, 5337.9784, 5190.0864, 5045.9158, 4919.1386, 4771.817, 4645.7742, 4518.4774, 4385.5454, 4262.6622, 4142.74679999999, 4015.5318, 3897.9276, 3790.7764, 3685.13800000001, 3573.6274, 3467.9706, 3368.61079999999, 3271.5202, 3170.3848, 3076.4656, 2982.38400000001, 2888.4664, 2806.4868, 2711.9564, 2634.1434, 2551.3204, 2469.7662, 2396.61139999999, 2318.9902, 2243.8658, 2171.9246, 2105.01360000001, 2028.8536, 1960.9952, 1901.4096, 1841.86079999999, 1777.54700000001, 1714.5802, 1654.65059999999, 1596.311, 1546.2016, 1492.3296, 1433.8974, 1383.84600000001, 1339.4152, 1293.5518, 1245.8686, 1193.50659999999, 1162.27959999999, 1107.19439999999, 1069.18060000001, 1035.09179999999, 999.679000000004, 957.679999999993, 925.300199999998, 888.099400000006, 848.638600000006, 818.156400000007, 796.748399999997, 752.139200000005, 725.271200000003, 692.216, 671.633600000001, 647.939799999993, 621.670599999998, 575.398799999995, 561.226599999995, 532.237999999998, 521.787599999996, 483.095799999996, 467.049599999998, 465.286399999997, 415.548599999995, 401.047399999996, 380.607999999993, 377.362599999993, 347.258799999996, 338.371599999999, 310.096999999994, 301.409199999995, 276.280799999993, 265.586800000005, 258.994399999996, 223.915999999997, 215.925399999993, 213.503800000006, 191.045400000003, 166.718200000003, 166.259000000005, 162.941200000001, 148.829400000002, 141.645999999993, 123.535399999993, 122.329800000007, 89.473399999988, 80.1962000000058, 77.5457999999926, 59.1056000000099, 83.3509999999951, 52.2906000000075, 36.3979999999865, 40.6558000000077, 42.0003999999899, 19.6630000000005, 19.7153999999864, -8.38539999999921, -0.692799999989802, 0.854800000000978, 3.23219999999856, -3.89040000000386, -5.25880000001052, -24.9052000000083, -22.6837999999989, -26.4286000000138, -34.997000000003, -37.0216000000073, -43.430400000012, -58.2390000000014, -68.8034000000043, -56.9245999999985, -57.8583999999973, -77.3097999999882, -73.2793999999994, -81.0738000000129, -87.4530000000086, -65.0254000000132, -57.296399999992, -96.2746000000043, -103.25, -96.081600000005, -91.5542000000132, -102.465200000006, -107.688599999994, -101.458000000013, -109.715800000005};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p16[] = {47270.0, 46423.3584, 45585.7074, 44757.152, 43938.8416, 43130.9514, 42330.03, 41540.407, 40759.6348, 39988.206, 39226.5144, 38473.2096, 37729.795, 36997.268, 36272.6448, 35558.665, 34853.0248, 34157.4472, 33470.5204, 32793.5742, 32127.0194, 31469.4182, 30817.6136, 30178.6968, 29546.8908, 28922.8544, 28312.271, 27707.0924, 27114.0326, 26526.692, 25948.6336, 25383.7826, 24823.5998, 24272.2974, 23732.2572, 23201.4976, 22674.2796, 22163.6336, 21656.515, 21161.7362, 20669.9368, 20189.4424, 19717.3358, 19256.3744, 18795.9638, 18352.197, 17908.5738, 17474.391, 17052.918, 16637.2236, 16228.4602, 15823.3474, 15428.6974, 15043.0284, 14667.6278, 14297.4588, 13935.2882, 13578.5402, 13234.6032, 12882.1578, 12548.0728, 12219.231, 11898.0072, 11587.2626, 11279.9072, 10973.5048, 10678.5186, 10392.4876, 10105.2556, 9825.766, 9562.5444, 9294.2222, 9038.2352, 8784.848, 8533.2644, 8301.7776, 8058.30859999999, 7822.94579999999, 7599.11319999999, 7366.90779999999, 7161.217, 6957.53080000001, 6736.212, 6548.21220000001, 6343.06839999999, 6156.28719999999, 5975.15419999999, 5791.75719999999, 5621.32019999999, 5451.66, 5287.61040000001, 5118.09479999999, 4957.288, 4798.4246, 4662.17559999999, 4512.05900000001, 4364.68539999999, 4220.77720000001, 4082.67259999999, 3957.19519999999, 3842.15779999999, 3699.3328, 3583.01180000001, 3473.8964, 3338.66639999999, 3233.55559999999, 3117.799, 3008.111, 2909.69140000001, 2814.86499999999, 2719.46119999999, 2624.742, 2532.46979999999, 2444.7886, 2370.1868, 2272.45259999999, 2196.19260000001, 2117.90419999999, 2023.2972, 1969.76819999999, 1885.58979999999, 1833.2824, 1733.91200000001, 1682.54920000001, 1604.57980000001, 1556.11240000001, 1491.3064, 1421.71960000001, 1371.22899999999, 1322.1324, 1264.7892, 1196.23920000001, 1143.8474, 1088.67240000001, 1073.60380000001, 1023.11660000001, 959.036400000012, 927.433199999999, 906.792799999996, 853.433599999989, 841.873800000001, 791.1054, 756.899999999994, 704.343200000003, 672.495599999995, 622.790399999998, 611.254799999995, 567.283200000005, 519.406599999988, 519.188400000014, 495.312800000014, 451.350799999986, 443.973399999988, 431.882199999993, 392.027000000002, 380.924200000009, 345.128999999986, 298.901400000002, 287.771999999997, 272.625, 247.253000000026, 222.490600000019, 223.590000000026, 196.407599999977, 176.425999999978, 134.725199999986, 132.4804, 110.445599999977, 86.7939999999944, 56.7038000000175, 64.915399999998, 38.3726000000024, 37.1606000000029, 46.170999999973, 49.1716000000015, 15.3362000000197, 6.71639999997569, -34.8185999999987, -39.4476000000141, 12.6830000000191, -12.3331999999937, -50.6565999999875, -59.9538000000175, -65.1054000000004, -70.7576000000117, -106.325200000021, -126.852200000023, -110.227599999984, -132.885999999999, -113.897200000007, -142.713800000027, -151.145399999979, -150.799200000009, -177.756200000003, -156.036399999983, -182.735199999996, -177.259399999981, -198.663600000029, -174.577600000019, -193.84580000001};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p17[] = {94541.0, 92848.811, 91174.019, 89517.558, 87879.9705, 86262.7565, 84663.5125, 83083.7435, 81521.7865, 79977.272, 78455.9465, 76950.219, 75465.432, 73994.152, 72546.71, 71115.2345, 69705.6765, 68314.937, 66944.2705, 65591.255, 64252.9485, 62938.016, 61636.8225, 60355.592, 59092.789, 57850.568, 56624.518, 55417.343, 54231.1415, 53067.387, 51903.526, 50774.649, 49657.6415, 48561.05, 47475.7575, 46410.159, 45364.852, 44327.053, 43318.4005, 42325.6165, 41348.4595, 40383.6265, 39436.77, 38509.502, 37594.035, 36695.939, 35818.6895, 34955.691, 34115.8095, 33293.949, 32465.0775, 31657.6715, 30877.2585, 30093.78, 29351.3695, 28594.1365, 27872.115, 27168.7465, 26477.076, 25774.541, 25106.5375, 24452.5135, 23815.5125, 23174.0655, 22555.2685, 21960.2065, 21376.3555, 20785.1925, 20211.517, 19657.0725, 19141.6865, 18579.737, 18081.3955, 17578.995, 17073.44, 16608.335, 16119.911, 15651.266, 15194.583, 14749.0495, 14343.4835, 13925.639, 13504.509, 13099.3885, 12691.2855, 12328.018, 11969.0345, 11596.5145, 11245.6355, 10917.6575, 10580.9785, 10277.8605, 9926.58100000001, 9605.538, 9300.42950000003, 8989.97850000003, 8728.73249999998, 8448.3235, 8175.31050000002, 7898.98700000002, 7629.79100000003, 7413.76199999999, 7149.92300000001, 6921.12650000001, 6677.1545, 6443.28000000003, 6278.23450000002, 6014.20049999998, 5791.20299999998, 5605.78450000001, 5438.48800000001, 5234.2255, 5059.6825, 4887.43349999998, 4682.935, 4496.31099999999, 4322.52250000002, 4191.42499999999, 4021.24200000003, 3900.64799999999, 3762.84250000003, 3609.98050000001, 3502.29599999997, 3363.84250000003, 3206.54849999998, 3079.70000000001, 2971.42300000001, 2867.80349999998, 2727.08100000001, 2630.74900000001, 2496.6165, 2440.902, 2356.19150000002, 2235.58199999999, 2120.54149999999, 2012.25449999998, 1933.35600000003, 1820.93099999998, 1761.54800000001, 1663.09350000002, 1578.84600000002, 1509.48149999999, 1427.3345, 1379.56150000001, 1306.68099999998, 1212.63449999999, 1084.17300000001, 1124.16450000001, 1060.69949999999, 1007.48849999998, 941.194499999983, 879.880500000028, 836.007500000007, 782.802000000025, 748.385499999975, 647.991500000004, 626.730500000005, 570.776000000013, 484.000500000024, 513.98550000001, 418.985499999952, 386.996999999974, 370.026500000036, 355.496999999974, 356.731499999994, 255.92200000002, 259.094000000041, 205.434499999974, 165.374500000034, 197.347500000033, 95.718499999959, 67.6165000000037, 54.6970000000438, 31.7395000000251, -15.8784999999916, 8.42500000004657, -26.3754999999655, -118.425500000012, -66.6629999999423, -42.9745000000112, -107.364999999991, -189.839000000036, -162.611499999999, -164.964999999967, -189.079999999958, -223.931499999948, -235.329999999958, -269.639500000048, -249.087999999989, -206.475499999942, -283.04449999996, -290.667000000016, -304.561499999953, -336.784499999951, -380.386500000022, -283.280499999993, -364.533000000054, -389.059499999974, -364.454000000027, -415.748000000021, -417.155000000028};
_CUDAX_CUCO_HLL_TUNING_ARR_DECL __bias_data_p18[] = {189083.0, 185696.913, 182348.774, 179035.946, 175762.762, 172526.444, 169329.754, 166166.099, 163043.269, 159958.91, 156907.912, 153906.845, 150924.199, 147996.568, 145093.457, 142239.233, 139421.475, 136632.27, 133889.588, 131174.2, 128511.619, 125868.621, 123265.385, 120721.061, 118181.769, 115709.456, 113252.446, 110840.198, 108465.099, 106126.164, 103823.469, 101556.618, 99308.004, 97124.508, 94937.803, 92833.731, 90745.061, 88677.627, 86617.47, 84650.442, 82697.833, 80769.132, 78879.629, 77014.432, 75215.626, 73384.587, 71652.482, 69895.93, 68209.301, 66553.669, 64921.981, 63310.323, 61742.115, 60205.018, 58698.658, 57190.657, 55760.865, 54331.169, 52908.167, 51550.273, 50225.254, 48922.421, 47614.533, 46362.049, 45098.569, 43926.083, 42736.03, 41593.473, 40425.26, 39316.237, 38243.651, 37170.617, 36114.609, 35084.19, 34117.233, 33206.509, 32231.505, 31318.728, 30403.404, 29540.0550000001, 28679.236, 27825.862, 26965.216, 26179.148, 25462.08, 24645.952, 23922.523, 23198.144, 22529.128, 21762.4179999999, 21134.779, 20459.117, 19840.818, 19187.04, 18636.3689999999, 17982.831, 17439.7389999999, 16874.547, 16358.2169999999, 15835.684, 15352.914, 14823.681, 14329.313, 13816.897, 13342.874, 12880.882, 12491.648, 12021.254, 11625.392, 11293.7610000001, 10813.697, 10456.209, 10099.074, 9755.39000000001, 9393.18500000006, 9047.57900000003, 8657.98499999999, 8395.85900000005, 8033.0, 7736.95900000003, 7430.59699999995, 7258.47699999996, 6924.58200000005, 6691.29399999999, 6357.92500000005, 6202.05700000003, 5921.19700000004, 5628.28399999999, 5404.96799999999, 5226.71100000001, 4990.75600000005, 4799.77399999998, 4622.93099999998, 4472.478, 4171.78700000001, 3957.46299999999, 3868.95200000005, 3691.14300000004, 3474.63100000005, 3341.67200000002, 3109.14000000001, 3071.97400000005, 2796.40399999998, 2756.17799999996, 2611.46999999997, 2471.93000000005, 2382.26399999997, 2209.22400000005, 2142.28399999999, 2013.96100000001, 1911.18999999994, 1818.27099999995, 1668.47900000005, 1519.65800000005, 1469.67599999998, 1367.13800000004, 1248.52899999998, 1181.23600000003, 1022.71900000004, 1088.20700000005, 959.03600000008, 876.095999999903, 791.183999999892, 703.337000000058, 731.949999999953, 586.86400000006, 526.024999999907, 323.004999999888, 320.448000000091, 340.672999999952, 309.638999999966, 216.601999999955, 102.922999999952, 19.2399999999907, -0.114000000059605, -32.6240000000689, -89.3179999999702, -153.497999999905, -64.2970000000205, -143.695999999996, -259.497999999905, -253.017999999924, -213.948000000091, -397.590000000084, -434.006000000052, -403.475000000093, -297.958000000101, -404.317000000039, -528.898999999976, -506.621000000043, -513.205000000075, -479.351000000024, -596.139999999898, -527.016999999993, -664.681000000099, -680.306000000099, -704.050000000047, -850.486000000034, -757.43200000003, -713.308999999892};
//! @brief Get bias data array for a given precision
//!
//! @param __precision The precision value (4-18)
//! @return Pointer to the bias data array for the given precision
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr const double* __bias_data(::cuda::std::int32_t __precision) noexcept {
switch (__precision) {
case 4: return __bias_data_p4;
case 5: return __bias_data_p5;
case 6: return __bias_data_p6;
case 7: return __bias_data_p7;
case 8: return __bias_data_p8;
case 9: return __bias_data_p9;
case 10: return __bias_data_p10;
case 11: return __bias_data_p11;
case 12: return __bias_data_p12;
case 13: return __bias_data_p13;
case 14: return __bias_data_p14;
case 15: return __bias_data_p15;
case 16: return __bias_data_p16;
case 17: return __bias_data_p17;
case 18: return __bias_data_p18;
default: return nullptr;
}
}
// clang-format on
} // namespace cuda::experimental::cuco::__hyperloglog_ns
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_HYPERLOGLOG_TUNING_CUH

View File

@@ -0,0 +1,294 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_KERNELS_CUH
#define _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_KERNELS_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cub/block/block_reduce.cuh>
#include <cuda/__atomic/atomic.h>
#include <cuda/std/__iterator/iterator_traits.h>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/std/__type_traits/void_t.h>
#include <cuda/experimental/__cuco/detail/utility/cuda.cuh>
#include <cooperative_groups.h>
#include <cooperative_groups/reduce.h>
#include <cuda/std/__cccl/prologue.h>
#if _CCCL_CUDA_COMPILATION()
_CCCL_DIAG_PUSH
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
namespace cuda::experimental::cuco::__open_addressing
{
//! @brief Scalar (cooperative-group size 1) functor inserting `first[i]` when `pred(stencil[i])` holds.
template <class _InputIt, class _StencilIt, class _Predicate, class _Ref>
struct __insert_if_fn
{
_InputIt __first;
_StencilIt __stencil;
_Predicate __pred;
_Ref __ref;
_CCCL_DEVICE_API void operator()(detail::__index_type __idx)
{
if (__pred(*(__stencil + __idx)))
{
__ref.insert(*(__first + __idx));
}
}
};
template <class _InputIt, class _StencilIt, class _Predicate, class _Ref>
__insert_if_fn(_InputIt, _StencilIt, _Predicate, _Ref) -> __insert_if_fn<_InputIt, _StencilIt, _Predicate, _Ref>;
//! @brief Scalar (cooperative-group size 1) functor writing `pred(stencil[i]) ? contains(first[i]) : false`.
template <class _InputIt, class _StencilIt, class _Predicate, class _OutputIt, class _Ref>
struct __contains_if_fn
{
_InputIt __first;
_StencilIt __stencil;
_Predicate __pred;
_OutputIt __output_begin;
_Ref __ref;
_CCCL_DEVICE_API void operator()(detail::__index_type __idx) const
{
*(__output_begin + __idx) = __pred(*(__stencil + __idx)) ? __ref.contains(*(__first + __idx)) : false;
}
};
template <class _InputIt, class _StencilIt, class _Predicate, class _OutputIt, class _Ref>
__contains_if_fn(_InputIt, _StencilIt, _Predicate, _OutputIt, _Ref)
-> __contains_if_fn<_InputIt, _StencilIt, _Predicate, _OutputIt, _Ref>;
//! @brief Inserts all elements in the range `[first, first + n)` and returns the number of
//! successful insertions if `pred` of the corresponding stencil returns true.
template <int _CgSize, int _BlockSize, class _InputIt, class _StencilIt, class _Predicate, class _Ref>
_CCCL_KERNEL_ATTRIBUTES _CCCL_LAUNCH_BOUNDS(_BlockSize) void __insert_if_n(
_InputIt __first,
detail::__index_type __n,
_StencilIt __stencil,
_Predicate __pred,
typename _Ref::size_type* __num_successes,
_Ref __ref)
{
using __block_reduce = CUB_NS_QUALIFIER::BlockReduce<typename _Ref::size_type, _BlockSize>;
__shared__ typename __block_reduce::TempStorage __temp_storage;
typename _Ref::size_type __thread_num_successes = 0;
const auto __loop_stride = detail::__grid_stride() / _CgSize;
auto __idx = detail::__global_thread_id() / _CgSize;
while (__idx < __n)
{
if (__pred(*(__stencil + __idx)))
{
using __value_t = typename ::cuda::std::iterator_traits<_InputIt>::value_type;
const __value_t __insert_element{*(__first + __idx)};
if constexpr (_CgSize == 1)
{
if (__ref.insert(__insert_element))
{
__thread_num_successes++;
}
}
else
{
const auto __tile = ::cooperative_groups::tiled_partition<_CgSize, ::cooperative_groups::thread_block>(
::cooperative_groups::this_thread_block());
if (__ref.insert(__tile, __insert_element) && __tile.thread_rank() == 0)
{
__thread_num_successes++;
}
}
}
__idx += __loop_stride;
}
const auto __block_num_successes = __block_reduce(__temp_storage).Sum(__thread_num_successes);
if (threadIdx.x == 0)
{
::cuda::atomic_ref<typename _Ref::size_type, _Ref::thread_scope>{*__num_successes}.fetch_add(
__block_num_successes, ::cuda::std::memory_order_relaxed);
}
}
//! @brief Inserts all elements in the range `[first, first + n)` if `pred` of the corresponding
//! stencil returns true.
template <int _CgSize, int _BlockSize, class _InputIt, class _StencilIt, class _Predicate, class _Ref>
_CCCL_KERNEL_ATTRIBUTES _CCCL_LAUNCH_BOUNDS(_BlockSize) void
__insert_if_n(_InputIt __first, detail::__index_type __n, _StencilIt __stencil, _Predicate __pred, _Ref __ref)
{
const auto __loop_stride = detail::__grid_stride() / _CgSize;
auto __idx = detail::__global_thread_id() / _CgSize;
while (__idx < __n)
{
if (__pred(*(__stencil + __idx)))
{
using __value_t = typename ::cuda::std::iterator_traits<_InputIt>::value_type;
const __value_t __insert_element{*(__first + __idx)};
const auto __tile = ::cooperative_groups::tiled_partition<_CgSize, ::cooperative_groups::thread_block>(
::cooperative_groups::this_thread_block());
__ref.insert(__tile, __insert_element);
}
__idx += __loop_stride;
}
}
//! @brief Contains test with predicate.
template <int _CgSize, int _BlockSize, class _InputIt, class _StencilIt, class _Predicate, class _OutputIt, class _Ref>
_CCCL_KERNEL_ATTRIBUTES _CCCL_LAUNCH_BOUNDS(_BlockSize) void __contains_if_n(
_InputIt __first,
detail::__index_type __n,
_StencilIt __stencil,
_Predicate __pred,
_OutputIt __output_begin,
_Ref __ref)
{
const auto __block = ::cooperative_groups::this_thread_block();
const auto __loop_stride = detail::__grid_stride() / _CgSize;
auto __idx = detail::__global_thread_id() / _CgSize;
while (__idx < __n)
{
const auto __tile = ::cooperative_groups::tiled_partition<_CgSize, ::cooperative_groups::thread_block>(__block);
using __value_t = typename ::cuda::std::iterator_traits<_InputIt>::value_type;
const __value_t __key = *(__first + __idx);
const auto __found = __pred(*(__stencil + __idx)) ? __ref.contains(__tile, __key) : false;
if (__tile.thread_rank() == 0)
{
*(__output_begin + __idx) = __found;
}
__idx += __loop_stride;
}
}
//! @brief Helper to determine the buffer type for the find kernel.
template <class _Container, class = void>
struct __find_buffer
{
using type = typename _Container::key_type;
};
//! @brief Helper to determine the buffer type for the find kernel when `mapped_type` exists.
template <class _Container>
struct __find_buffer<_Container, ::cuda::std::void_t<typename _Container::mapped_type>>
{
using type = typename _Container::mapped_type;
};
//! @brief Converts a find result to the output value or the appropriate empty sentinel.
template <class _Ref, class _Iterator>
[[nodiscard]] _CCCL_DEVICE_API typename __find_buffer<_Ref>::type __find_output(_Ref const& __ref, _Iterator __found)
{
constexpr bool __has_payload = !::cuda::std::is_same_v<typename _Ref::key_type, typename _Ref::value_type>;
if constexpr (__has_payload)
{
return __found == __ref.end() ? __ref.empty_value_sentinel() : __found->second;
}
else
{
return __found == __ref.end() ? __ref.empty_key_sentinel() : *__found;
}
}
//! @brief Find with predicate.
template <int _CgSize, int _BlockSize, class _InputIt, class _StencilIt, class _Predicate, class _OutputIt, class _Ref>
_CCCL_KERNEL_ATTRIBUTES _CCCL_LAUNCH_BOUNDS(_BlockSize) void __find_if_n(
_InputIt __first,
detail::__index_type __n,
_StencilIt __stencil,
_Predicate __pred,
_OutputIt __output_begin,
_Ref __ref)
{
const auto __block = ::cooperative_groups::this_thread_block();
const auto __thread_idx = __block.thread_rank();
const auto __loop_stride = detail::__grid_stride() / _CgSize;
auto __idx = detail::__global_thread_id() / _CgSize;
using __output_type = typename __find_buffer<_Ref>::type;
__shared__ __output_type __output_buffer[_BlockSize / _CgSize];
while ((__idx - __thread_idx / _CgSize) < __n)
{
if constexpr (_CgSize == 1)
{
if (__idx < __n)
{
using __value_t = typename ::cuda::std::iterator_traits<_InputIt>::value_type;
const __value_t __key = *(__first + __idx);
const auto __selected = __pred(*(__stencil + __idx));
const auto __found = __selected ? __ref.find(__key) : __ref.end();
/*
* The ld.relaxed.gpu instruction causes L1 to flush more frequently, causing increased
* sector stores from L2 to global memory. By writing results to shared memory and then
* synchronizing before writing back to global, we no longer rely on L1, preventing the
* increase in sector stores from L2 to global and improving performance.
*/
__output_buffer[__thread_idx] = __find_output(__ref, __found);
}
__block.sync();
if (__idx < __n)
{
*(__output_begin + __idx) = __output_buffer[__thread_idx];
}
}
else
{
const auto __tile = ::cooperative_groups::tiled_partition<_CgSize, ::cooperative_groups::thread_block>(__block);
if (__idx < __n)
{
using __value_t = typename ::cuda::std::iterator_traits<_InputIt>::value_type;
const __value_t __key = *(__first + __idx);
bool __selected = false;
if (__tile.thread_rank() == 0)
{
__selected = __pred(*(__stencil + __idx));
}
__selected = __tile.shfl(__selected, 0);
const auto __found = __selected ? __ref.find(__tile, __key) : __ref.end();
if (__tile.thread_rank() == 0)
{
*(__output_begin + __idx) = __find_output(__ref, __found);
}
}
}
__idx += __loop_stride;
}
}
} // namespace cuda::experimental::cuco::__open_addressing
_CCCL_DIAG_POP
#endif // _CCCL_CUDA_COMPILATION()
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_KERNELS_CUH

View File

@@ -0,0 +1,426 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_IMPL_CUH
#define _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_IMPL_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cub/device/device_for.cuh>
#include <cub/device/device_transform.cuh>
#include <cuda/__container/buffer.h>
#include <cuda/__driver/driver_api.h>
#include <cuda/__iterator/constant_iterator.h>
#include <cuda/__runtime/api_wrapper.h>
#include <cuda/__type_traits/is_bitwise_comparable.h>
#include <cuda/std/__exception/exception_macros.h>
#include <cuda/std/__functional/identity.h>
#include <cuda/std/__type_traits/is_base_of.h>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/experimental/__cuco/capacity.cuh>
#include <cuda/experimental/__cuco/detail/open_addressing/kernels.cuh>
#include <cuda/experimental/__cuco/detail/open_addressing/slot_storage_ref.cuh>
#include <cuda/experimental/__cuco/detail/utility/cuda.cuh>
#include <cuda/experimental/__cuco/probing_scheme.cuh>
#include <cuda/std/__cccl/prologue.h>
#if !_CCCL_COMPILER(NVRTC)
namespace cuda::experimental::cuco::__open_addressing
{
//! @brief Open addressing implementation class.
//!
//! @note This class should NOT be used directly.
//!
//! @throw If the size of the given key type is larger than 8 bytes
//! @throw If the size of the given slot type is larger than 16 bytes
//! @throw If the given key type doesn't have unique object representations, i.e.,
//! `cuda::is_bitwise_comparable_v<_Key> == false`
//! @throw If the probing scheme type is not inherited from
//! `cuda::experimental::cuco::detail::__probing_scheme_base`
//!
//! @tparam _Key Type used for keys. Requires `cuda::is_bitwise_comparable_v<_Key>`
//! @tparam _Value Type used for storage values
//! @tparam _Scope The scope in which operations will be performed by individual threads
//! @tparam _KeyEqual Binary callable type used to compare two keys for equality
//! @tparam _ProbingScheme Probing scheme type
//! @tparam _BucketSize Number of slots per bucket
//! @tparam _MemoryResource Type of memory resource used for device storage
template <class _Key,
class _Value,
::cuda::thread_scope _Scope,
class _KeyEqual,
class _ProbingScheme,
int _BucketSize,
class _MemoryResource>
class __open_addressing_impl
{
public:
using __key_type = _Key;
using __value_type = _Value;
using __probing_scheme_type = _ProbingScheme;
using __hasher = typename __probing_scheme_type::hasher;
using __size_type = ::cuda::std::size_t;
using __key_equal = _KeyEqual;
using __storage_ref_type = __slot_storage_ref<__value_type, _BucketSize>;
static constexpr auto __has_payload = !::cuda::std::is_same_v<_Key, _Value>;
static constexpr auto __cg_size = _ProbingScheme::cg_size;
static constexpr auto __bucket_size = _BucketSize;
static constexpr auto __thread_scope = _Scope;
static_assert(sizeof(_Key) <= 8, "Container does not support key types larger than 8 bytes.");
static_assert(sizeof(_Value) <= 16, "Container does not support slot types larger than 16 bytes.");
static_assert(::cuda::is_bitwise_comparable_v<_Key>,
"Key type must have unique object representations or have been explicitly declared as safe for "
"bitwise comparison via specialization of cuda::is_bitwise_comparable_v<Key>.");
static_assert(::cuda::std::is_base_of_v<detail::__probing_scheme_base<_ProbingScheme::cg_size>, _ProbingScheme>,
"ProbingScheme must inherit from cuda::experimental::cuco::detail::__probing_scheme_base");
private:
__value_type __empty_slot_sentinel;
__key_type __erased_key_sentinel;
__key_equal __predicate;
__probing_scheme_type __probing_scheme;
mutable _MemoryResource __memory_resource;
::cuda::device_buffer<__value_type> __slots;
//! @brief Computes the number of buckets for a requested capacity.
[[nodiscard]] _CCCL_HOST_API static __size_type __compute_num_buckets(__size_type __requested_capacity)
{
return make_valid_capacity<_ProbingScheme, _BucketSize>(__requested_capacity) / _BucketSize;
}
//! @brief Computes the number of buckets for a given number of keys and load factor.
[[nodiscard]] _CCCL_HOST_API static __size_type __compute_num_buckets(__size_type __n, double __load_factor)
{
return make_valid_capacity<_ProbingScheme, _BucketSize>(__n, __load_factor) / _BucketSize;
}
//! @brief Extracts the key from a slot.
[[nodiscard]] _CCCL_HOST_API constexpr const __key_type& __extract_key(const __value_type& __slot) const noexcept
{
if constexpr (__has_payload)
{
return __slot.first;
}
else
{
return __slot;
}
}
//! @brief Allocates and zero-initializes an RAII device counter.
[[nodiscard]] _CCCL_HOST_API ::cuda::device_buffer<__size_type> __make_counter(::cuda::stream_ref __stream) const
{
return ::cuda::device_buffer<__size_type>{__stream, __memory_resource, {__size_type{0}}};
}
//! @brief Reads a device counter to host.
[[nodiscard]] _CCCL_HOST_API __size_type
__read_counter(const ::cuda::device_buffer<__size_type>& __counter, ::cuda::stream_ref __stream) const
{
__size_type __result;
::cuda::__driver::__memcpyAsync(&__result, __counter.data(), sizeof(__size_type), __stream.get());
__stream.sync();
return __result;
}
public:
//! @brief Constructs an open addressing implementation with the given capacity.
_CCCL_HOST_API __open_addressing_impl(
::cuda::stream_ref __stream,
_MemoryResource __mr,
__size_type __capacity,
__value_type __empty_slot_sentinel,
const _KeyEqual& __pred,
const _ProbingScheme& __probing_scheme)
: __empty_slot_sentinel{__empty_slot_sentinel}
, __erased_key_sentinel{__extract_key(__empty_slot_sentinel)}
, __predicate{__pred}
, __probing_scheme{__probing_scheme}
, __memory_resource{__mr}
, __slots{__stream, __mr, __compute_num_buckets(__capacity) * _BucketSize, ::cuda::no_init}
{
clear_async(__stream);
}
//! @brief Constructs an open addressing implementation with capacity derived from desired load
//! factor.
_CCCL_HOST_API __open_addressing_impl(
::cuda::stream_ref __stream,
_MemoryResource __mr,
__size_type __n,
double __desired_load_factor,
__value_type __empty_slot_sentinel,
const _KeyEqual& __pred,
const _ProbingScheme& __probing_scheme)
: __empty_slot_sentinel{__empty_slot_sentinel}
, __erased_key_sentinel{__extract_key(__empty_slot_sentinel)}
, __predicate{__pred}
, __probing_scheme{__probing_scheme}
, __memory_resource{__mr}
, __slots{__stream, __mr, __compute_num_buckets(__n, __desired_load_factor) * _BucketSize, ::cuda::no_init}
{
clear_async(__stream);
}
//! @brief Constructs an open addressing implementation with erasure support.
_CCCL_HOST_API __open_addressing_impl(
::cuda::stream_ref __stream,
_MemoryResource __mr,
__size_type __capacity,
__value_type __empty_slot_sentinel,
__key_type __erased_key_sentinel,
const _KeyEqual& __pred,
const _ProbingScheme& __probing_scheme)
: __empty_slot_sentinel{__empty_slot_sentinel}
, __erased_key_sentinel{__erased_key_sentinel}
, __predicate{__pred}
, __probing_scheme{__probing_scheme}
, __memory_resource{__mr}
, __slots{__stream, __mr, __compute_num_buckets(__capacity) * _BucketSize, ::cuda::no_init}
{
if (empty_key_sentinel() == erased_key_sentinel())
{
_CCCL_THROW(::std::invalid_argument, "The empty key sentinel and erased key sentinel cannot be the same value.");
}
clear_async(__stream);
}
//! @brief Fills all slots with the empty sentinel.
_CCCL_HOST_API void clear(::cuda::stream_ref __stream)
{
clear_async(__stream);
__stream.sync();
}
//! @brief Asynchronously fills all slots with the empty sentinel.
//!
//! @throws cuda_error if the clear operation fails to launch
_CCCL_HOST_API void clear_async(::cuda::stream_ref __stream)
{
const auto __n = capacity();
if (__n == 0)
{
return;
}
_CCCL_TRY_CUDA_API(
CUB_NS_QUALIFIER::DeviceTransform::Fill,
"cuco: failed to clear slot storage",
__slots.data(),
static_cast<detail::__index_type>(__n),
__empty_slot_sentinel,
__stream);
}
//! @brief Inserts keys in `[first, last)` and returns the number of successful insertions.
template <class _InputIt, class _Ref>
_CCCL_HOST_API __size_type insert(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last, _Ref __container_ref)
{
const auto __num_keys = detail::__distance(__first, __last);
if (__num_keys == 0)
{
return 0;
}
auto __counter = __make_counter(__stream);
const auto __grid_size = detail::__grid_size(__num_keys, __cg_size);
__open_addressing::__insert_if_n<__cg_size, detail::__default_block_size>
<<<static_cast<unsigned>(__grid_size), detail::__default_block_size, 0, __stream.get()>>>(
__first,
__num_keys,
::cuda::constant_iterator<bool>{true},
::cuda::std::identity{},
__counter.data(),
__container_ref);
return __read_counter(__counter, __stream);
}
//! @brief Asynchronously inserts keys in `[first, last)`.
//!
//! @throws cuda_error if the insert operation fails to launch
template <class _InputIt, class _Ref>
_CCCL_HOST_API void insert_async(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last, _Ref __container_ref)
{
const auto __num_keys = detail::__distance(__first, __last);
if (__num_keys == 0)
{
return;
}
if constexpr (__cg_size == 1)
{
__open_addressing::__insert_if_fn __op{
__first, ::cuda::constant_iterator<bool>{true}, ::cuda::std::identity{}, __container_ref};
_CCCL_TRY_CUDA_API(CUB_NS_QUALIFIER::DeviceFor::Bulk, "cuco: failed to insert keys", __num_keys, __op, __stream);
}
else
{
const auto __grid_size = detail::__grid_size(__num_keys, __cg_size);
__open_addressing::__insert_if_n<__cg_size, detail::__default_block_size>
<<<static_cast<unsigned>(__grid_size), detail::__default_block_size, 0, __stream.get()>>>(
__first, __num_keys, ::cuda::constant_iterator<bool>{true}, ::cuda::std::identity{}, __container_ref);
}
}
//! @brief Asynchronously checks if keys in `[first, last)` exist in the container.
//!
//! @throws cuda_error if the query operation fails to launch
template <class _InputIt, class _OutputIt, class _Ref>
_CCCL_HOST_API void contains_async(
::cuda::stream_ref __stream, _InputIt __first, _InputIt __last, _OutputIt __output_begin, _Ref __container_ref) const
{
const auto __num_keys = detail::__distance(__first, __last);
if (__num_keys == 0)
{
return;
}
if constexpr (__cg_size == 1)
{
__open_addressing::__contains_if_fn __op{
__first, ::cuda::constant_iterator<bool>{true}, ::cuda::std::identity{}, __output_begin, __container_ref};
_CCCL_TRY_CUDA_API(CUB_NS_QUALIFIER::DeviceFor::Bulk, "cuco: failed to query keys", __num_keys, __op, __stream);
}
else
{
const auto __grid_size = detail::__grid_size(__num_keys, __cg_size);
__open_addressing::__contains_if_n<__cg_size, detail::__default_block_size>
<<<static_cast<unsigned>(__grid_size), detail::__default_block_size, 0, __stream.get()>>>(
__first,
__num_keys,
::cuda::constant_iterator<bool>{true},
::cuda::std::identity{},
__output_begin,
__container_ref);
}
}
//! @brief Asynchronously finds payloads for keys in `[first, last)` whose stencil satisfies `pred`.
//!
//! For each key `first[i]` with `pred(stencil[i]) == true` that is present, the associated payload is
//! written to the corresponding output position; otherwise the empty value sentinel is written.
//!
//! @throws cuda_error if the query operation fails to launch
template <class _InputIt, class _StencilIt, class _Predicate, class _OutputIt, class _Ref>
_CCCL_HOST_API void find_if_async(
::cuda::stream_ref __stream,
_InputIt __first,
_InputIt __last,
_StencilIt __stencil,
_Predicate __pred,
_OutputIt __output_begin,
_Ref __container_ref) const
{
const auto __num_keys = detail::__distance(__first, __last);
if (__num_keys == 0)
{
return;
}
const auto __grid_size = detail::__grid_size(__num_keys, __cg_size);
__open_addressing::__find_if_n<__cg_size, detail::__default_block_size>
<<<static_cast<unsigned>(__grid_size), detail::__default_block_size, 0, __stream.get()>>>(
__first, __num_keys, __stencil, __pred, __output_begin, __container_ref);
}
//! @brief Asynchronously finds the payloads for keys in `[first, last)`.
//!
//! For each key that is present, the associated payload is written to the corresponding output
//! position; for each key that is absent, the empty value sentinel is written instead.
//!
//! @throws cuda_error if the query operation fails to launch
template <class _InputIt, class _OutputIt, class _Ref>
_CCCL_HOST_API void find_async(
::cuda::stream_ref __stream, _InputIt __first, _InputIt __last, _OutputIt __output_begin, _Ref __container_ref) const
{
this->find_if_async(
__stream,
__first,
__last,
::cuda::constant_iterator<bool>{true},
::cuda::std::identity{},
__output_begin,
__container_ref);
}
//! @brief Returns the total number of slots.
[[nodiscard]] _CCCL_HOST_API constexpr __size_type capacity() const noexcept
{
return static_cast<__size_type>(__slots.size());
}
//! @brief Returns a pointer to the underlying slot array.
[[nodiscard]] _CCCL_HOST_API __value_type* data() const noexcept
{
return const_cast<__value_type*>(__slots.data());
}
//! @brief Returns the empty key sentinel.
[[nodiscard]] _CCCL_HOST_API constexpr __key_type empty_key_sentinel() const noexcept
{
return __extract_key(__empty_slot_sentinel);
}
//! @brief Returns the erased key sentinel.
[[nodiscard]] _CCCL_HOST_API constexpr __key_type erased_key_sentinel() const noexcept
{
return __erased_key_sentinel;
}
//! @brief Returns the key comparison function.
[[nodiscard]] _CCCL_HOST_API constexpr __key_equal key_eq() const noexcept
{
return __predicate;
}
//! @brief Returns the probing scheme.
[[nodiscard]] _CCCL_HOST_API constexpr __probing_scheme_type probing_scheme() const noexcept
{
return __probing_scheme;
}
//! @brief Returns the hash function.
[[nodiscard]] _CCCL_HOST_API constexpr __hasher hash_function() const noexcept
{
return probing_scheme().hash_function();
}
//! @brief Returns a non-owning reference to the stored slots.
[[nodiscard]] _CCCL_HOST_API __storage_ref_type storage_ref() const noexcept
{
return __storage_ref_type{const_cast<__value_type*>(__slots.data()), capacity()};
}
};
} // namespace cuda::experimental::cuco::__open_addressing
#endif // !_CCCL_COMPILER(NVRTC)
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_IMPL_CUH

View File

@@ -0,0 +1,126 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_SLOT_STORAGE_REF_CUH
#define _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_SLOT_STORAGE_REF_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__memory/is_aligned.h>
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/__mdspan/extents.h>
#include <cuda/std/span>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::__open_addressing
{
//! @brief Lightweight non-owning reference to a contiguous slot array with bucket abstraction.
//!
//! Provides indexing into the slot array organized as buckets; within each bucket there are
//! `_BucketSize` value-typed slots. The total slot count is carried as a `cuda::std::extents`, so a
//! static `_Capacity` folds the probing reduction to a constant while a dynamic `_Capacity` stores
//! the slot count. The probing layer works in slot offsets bounded by `capacity()`.
//!
//! @tparam _Value The slot value type (e.g. `::cuda::std::pair<Key, T>`)
//! @tparam _BucketSize Number of slots per bucket (compile-time constant)
//! @tparam _Capacity Valid total slot count, or `cuda::std::dynamic_extent` for runtime sizing
template <class _Value, int _BucketSize, ::cuda::std::size_t _Capacity = ::cuda::std::dynamic_extent>
struct __slot_storage_ref
{
using __size_type = ::cuda::std::size_t;
using __value_type = _Value;
using __capacity_extent_type = ::cuda::std::extents<__size_type, _Capacity>;
using __iterator = _Value*;
using __const_iterator = const _Value*;
static constexpr int __bucket_size = _BucketSize;
using __bucket_type = ::cuda::std::span<_Value, _BucketSize>;
static_assert(_BucketSize > 0, "bucket size must be greater than zero");
static_assert(_Capacity == ::cuda::std::dynamic_extent || _Capacity % _BucketSize == 0,
"static capacity must be divisible by the bucket size");
_Value* __data_;
_CCCL_NO_UNIQUE_ADDRESS __capacity_extent_type __capacity_;
//! @brief Constructs a slot storage ref.
//!
//! @param __data Pointer to the first slot
//! @param __capacity Total slot count (must equal the static `_Capacity` when it is static)
_CCCL_HOST_DEVICE_API constexpr __slot_storage_ref(_Value* __data, __size_type __capacity) noexcept
: __data_{__data}
, __capacity_{__capacity}
{}
//! @brief Returns the bucket at position `__i`.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __bucket_type operator[](__size_type __i) const noexcept
{
return __bucket_type{__data_ + __i, typename __bucket_type::size_type{_BucketSize}};
}
//! @brief Returns the total number of slots.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __size_type capacity() const noexcept
{
return __capacity_.extent(0);
}
//! @brief Returns the number of buckets.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __size_type num_buckets() const noexcept
{
return capacity() / __size_type{_BucketSize};
}
//! @brief Returns the total slot count as a `cuda::std::extents` (the probing reduction bound).
//!
//! Returning the extent rather than a plain size keeps the static slot count in the type, so the
//! probing iterator's modular reduction folds to a constant for static `_Capacity`.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __capacity_extent_type capacity_extent() const noexcept
{
return __capacity_;
}
//! @brief Returns a pointer to the underlying slot array.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr _Value* data() const noexcept
{
return __data_;
}
[[nodiscard]] _CCCL_HOST_DEVICE_API bool __is_packed_cas_aligned() const noexcept
{
return ::cuda::is_aligned(__data_, sizeof(_Value));
}
//! @brief Returns an iterator to the first slot.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __iterator begin() const noexcept
{
return __data_;
}
//! @brief Returns an iterator to one past the last slot.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __iterator end() const noexcept
{
return __data_ + capacity();
}
};
} // namespace cuda::experimental::cuco::__open_addressing
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_OPEN_ADDRESSING_SLOT_STORAGE_REF_CUH

View File

@@ -0,0 +1,173 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_PRIME_CUH
#define _CUDAX___CUCO_DETAIL_PRIME_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__numeric/add_overflow.h>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::detail
{
//! @brief Modular multiplication: `(__n1 * __n2) % __m` without overflow.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint64_t
__mod_mul(::cuda::std::uint64_t __n1, ::cuda::std::uint64_t __n2, ::cuda::std::uint64_t __m) noexcept
{
#if _CCCL_HAS_INT128()
auto __r = static_cast<__uint128_t>(__n1) * __n2;
return static_cast<::cuda::std::uint64_t>(__r % __m);
#else
// Fallback: Russian-peasant multiplication in modular arithmetic.
::cuda::std::uint64_t __r = 0;
__n1 %= __m;
__n2 %= __m;
while (__n2 > 0)
{
const ::cuda::std::uint64_t __mod_diff = __m - __n1;
if (__n2 & 1)
{
__r = (__r >= __mod_diff) ? __r - __mod_diff : __r + __n1;
}
__n1 = (__n1 >= __mod_diff) ? __n1 - __mod_diff : __n1 + __n1;
__n2 >>= 1;
}
return __r;
#endif // _CCCL_HAS_INT128()
}
//! @brief Modular exponentiation: `(__b ^ __e) % __m` via binary exponentiation.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint64_t
__mod_pow(::cuda::std::uint64_t __b, ::cuda::std::uint64_t __e, ::cuda::std::uint64_t __m) noexcept
{
::cuda::std::uint64_t __r = 1;
__b %= __m;
for (; __e > 0; __e >>= 1)
{
if (__e & 1)
{
__r = detail::__mod_mul(__r, __b, __m);
}
__b = detail::__mod_mul(__b, __b, __m);
}
return __r;
}
//! @brief Single Miller-Rabin witness test.
//!
//! Given `n - 1 == 2^s * d`, checks whether `a^d == 1 (mod n)` or
//! `a^(2^r * d) == n - 1 (mod n)` for some `0 <= r < s`.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr bool __miller_rabin_test(
::cuda::std::uint64_t __n, ::cuda::std::uint64_t __a, ::cuda::std::uint64_t __d, ::cuda::std::uint32_t __s) noexcept
{
::cuda::std::uint64_t __x = detail::__mod_pow(__a % __n, __d, __n);
const auto __neg_one = __n - 1;
if (__x == 1 || __x == __neg_one)
{
return true;
}
for (::cuda::std::uint32_t __i = 1; __i < __s; ++__i)
{
__x = detail::__mod_mul(__x, __x, __n);
if (__x == __neg_one)
{
return true;
}
}
return false;
}
//! @brief Deterministic primality test for all 64-bit integers.
//!
//! Uses trial division by small primes followed by Miller-Rabin with a fixed
//! set of bases that make the test deterministic for every `uint64_t`.
//! Bases from https://cp-algorithms.com/algebra/primality_tests.html.
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr bool __is_prime(::cuda::std::uint64_t __n) noexcept
{
if (__n < 2)
{
return false;
}
// Trial division by small primes.
constexpr ::cuda::std::uint64_t __small_primes[]{
2ull, 3ull, 5ull, 7ull, 11ull, 13ull, 17ull, 19ull, 23ull, 29ull, 31ull, 37ull};
for (::cuda::std::uint64_t __p : __small_primes)
{
if (__n % __p == 0)
{
return __n == __p;
}
}
// Decompose `__n - 1 == 2^__s * __d`.
::cuda::std::uint64_t __d = __n - 1;
::cuda::std::uint32_t __s = 0;
while ((__d & 1) == 0)
{
__d >>= 1;
++__s;
}
// Deterministic witness bases for all `uint64_t` values.
constexpr ::cuda::std::uint64_t __witnesses[]{2ull, 325ull, 9375ull, 28178ull, 450775ull, 9780504ull, 1795265022ull};
for (::cuda::std::uint64_t __a : __witnesses)
{
if (!detail::__miller_rabin_test(__n, __a, __d, __s))
{
return false;
}
}
return true;
}
//! @brief Returns the smallest prime `>= __n`.
//!
//! For `__n <= 2`, returns 2. Otherwise searches odd numbers starting from
//! `__n` (or `__n + 1` if `__n` is even).
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::uint64_t __next_prime(::cuda::std::uint64_t __n) noexcept
{
if (__n <= 2)
{
return 2;
}
__n |= 1; // make odd
while (!detail::__is_prime(__n))
{
const auto __next = ::cuda::add_overflow(__n, ::cuda::std::uint64_t{2});
if (__next.overflow)
{
return __n;
}
__n = __next.value;
}
return __n;
}
} // namespace cuda::experimental::cuco::detail
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_PRIME_CUH

View File

@@ -0,0 +1,89 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_PROBING_SCHEME_BASE_CUH
#define _CUDAX___CUCO_DETAIL_PROBING_SCHEME_BASE_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::detail
{
//! @brief Base class of public probing schemes.
//!
//! @tparam _CgSize Cooperative group size
template <int _CgSize>
class __probing_scheme_base
{
public:
static constexpr int __cg_size = _CgSize;
};
//! @brief Probing iterator class.
//!
//! Yields slot offsets and wraps modulo the total capacity (in slots). The capacity is held as a
//! `cuda::std::extents` so a static slot count folds the reduction to a constant.
//!
//! @tparam _CapacityExtent Capacity extent type (total slots), a `cuda::std::extents`
//! @tparam _StepExtent Probe-step extent type, a `cuda::std::extents` (static for linear probing)
template <class _CapacityExtent, class _StepExtent>
class __probing_iterator
{
public:
using __capacity_extent_type = _CapacityExtent;
using __step_extent_type = _StepExtent;
using __size_type = typename _CapacityExtent::index_type;
_CCCL_HOST_DEVICE_API constexpr __probing_iterator(
__size_type __start, _StepExtent __step, _CapacityExtent __capacity) noexcept
: __curr_index{__start}
, __step_{__step}
, __capacity_{__capacity}
{}
#if _CCCL_CUDA_COMPILATION()
_CCCL_DEVICE_API constexpr auto operator*() const noexcept
{
return __curr_index;
}
_CCCL_DEVICE_API constexpr auto operator++() noexcept
{
__curr_index = (__curr_index + __step_.extent(0)) % __capacity_.extent(0);
return *this;
}
_CCCL_DEVICE_API constexpr auto operator++(int) noexcept
{
auto __temp = *this;
++(*this);
return __temp;
}
#endif // _CCCL_CUDA_COMPILATION()
private:
__size_type __curr_index;
_CCCL_NO_UNIQUE_ADDRESS _StepExtent __step_;
_CCCL_NO_UNIQUE_ADDRESS _CapacityExtent __capacity_;
};
} // namespace cuda::experimental::cuco::detail
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_PROBING_SCHEME_BASE_CUH

View File

@@ -0,0 +1,88 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_UTILITY_CUDA_CUH
#define _CUDAX___CUCO_DETAIL_UTILITY_CUDA_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__cmath/ceil_div.h>
#include <cuda/__hierarchy/hierarchy_levels.h>
#include <cuda/std/__iterator/concepts.h>
#include <cuda/std/__iterator/distance.h>
#include <cuda/std/cstdint>
#include <cooperative_groups.h>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::detail
{
using __index_type = ::cuda::std::int64_t;
#if _CCCL_CUDA_COMPILATION()
[[nodiscard]] _CCCL_DEVICE_API inline __index_type __global_thread_id() noexcept
{
return ::cuda::gpu_thread.rank_as<__index_type>(::cuda::grid);
}
[[nodiscard]] _CCCL_DEVICE_API inline __index_type __grid_stride() noexcept
{
return ::cuda::gpu_thread.count_as<__index_type>(::cuda::grid);
}
#endif // _CCCL_CUDA_COMPILATION()
inline constexpr int __default_block_size = 128;
inline constexpr int __default_stride = 1;
inline constexpr int __warp_size = 32;
template <class _Tile>
struct __tile_size;
template <::cuda::std::uint32_t _Size, class _ParentCG>
struct __tile_size<::cooperative_groups::thread_block_tile<_Size, _ParentCG>>
{
static constexpr int __value = _Size;
};
template <class _Tile>
inline constexpr int __tile_size_v = __tile_size<_Tile>::__value;
constexpr _CCCL_HOST_DEVICE_API __index_type __grid_size(
__index_type __num,
int __cg_size = 1,
int __stride = __default_stride,
int __block_size = __default_block_size) noexcept
{
return ::cuda::ceil_div(__cg_size * __num, __stride * __block_size);
}
//! @brief Distance helper requiring random access iterators.
template <class _Iterator>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr __index_type __distance(_Iterator __begin, _Iterator __end)
{
static_assert(::cuda::std::random_access_iterator<_Iterator>, "Input iterator should be a random access iterator.");
return __index_type{::cuda::std::distance(__begin, __end)};
}
} // namespace cuda::experimental::cuco::detail
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_UTILITY_CUDA_CUH

View File

@@ -0,0 +1,64 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_UTILITY_STRONG_TYPE_CUH
#define _CUDAX___CUCO_DETAIL_UTILITY_STRONG_TYPE_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief A strong type wrapper.
//!
//! @tparam _Tp Type of the underlying value
template <class _Tp>
struct __strong_type
{
//! @brief Constructs a strong type.
//!
//! @param __v Value to be wrapped as a strong type
_CCCL_HOST_DEVICE_API explicit constexpr __strong_type(_Tp __v)
: __value{__v}
{}
//! @brief Implicit conversion operator to the underlying value.
//!
//! @return The underlying value
_CCCL_HOST_DEVICE_API constexpr operator _Tp() const noexcept
{
return __value;
}
_Tp __value; //!< Underlying data value
};
} // namespace cuda::experimental::cuco
//! Convenience wrapper for defining a strong type
#define CUDAX_CUCO_DEFINE_STRONG_TYPE(Name, Type) \
struct Name : __strong_type<Type> \
{ \
_CCCL_HOST_DEVICE_API explicit constexpr Name(Type __value) \
: __strong_type<Type>(__value) \
{} \
};
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_UTILITY_STRONG_TYPE_CUH

View File

@@ -0,0 +1,45 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_DETAIL_UTILITY_TRAITS_CUH
#define _CUDAX___CUCO_DETAIL_UTILITY_TRAITS_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <thrust/device_reference.h>
#include <cuda/std/__tuple_dir/tuple_like.h>
#include <cuda/std/__type_traits/remove_reference.h>
#include <cuda/std/__utility/declval.h>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco::detail
{
//! @brief Trait value indicating whether `_Tp`, after unwrapping any thrust reference, is a pair-like
//! type (tuple-like with exactly two elements).
//!
//! @tparam _Tp Type to inspect
template <class _Tp>
inline constexpr bool __is_pair_like_v = ::cuda::std::__pair_like<
::cuda::std::remove_reference_t<decltype(::thrust::raw_reference_cast(::cuda::std::declval<_Tp>()))>>;
} // namespace cuda::experimental::cuco::detail
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_DETAIL_UTILITY_TRAITS_CUH

View File

@@ -0,0 +1,538 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_FIXED_CAPACITY_MAP_CUH
#define _CUDAX___CUCO_FIXED_CAPACITY_MAP_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__memory_pool/device_memory_pool.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/__fwd/extents.h>
#include <cuda/std/__memory/unique_ptr.h>
#include <cuda/std/__utility/pair.h>
#include <cuda/experimental/__cuco/capacity.cuh>
#include <cuda/experimental/__cuco/detail/bitwise_compare.cuh>
#include <cuda/experimental/__cuco/detail/open_addressing/open_addressing_impl.cuh>
#include <cuda/experimental/__cuco/fixed_capacity_map_ref.cuh>
#include <cuda/experimental/__cuco/hash_functions.cuh>
#include <cuda/experimental/__cuco/probing_scheme.cuh>
#include <cuda/experimental/__cuco/types.cuh>
#include <cuda/std/__cccl/prologue.h>
#if !_CCCL_COMPILER(NVRTC)
namespace cuda::experimental::cuco
{
//! @brief A GPU-accelerated, unordered, associative container of key-value pairs with unique keys.
//!
//! Allows constant-time inserts and lookups from device code. Many threads may perform
//! the same kind of operation concurrently (e.g. concurrent inserts, or concurrent lookups).
//! Storage is bulk-allocated ahead of time and requires the user to provide sentinel values
//! for empty and, optionally, erased keys.
//!
//! @note Concurrent modification (insert) and lookup (contains) on the same map are not
//! supported: lookups perform non-atomic loads, so a lookup that overlaps a concurrent insert
//! is a data race and results in undefined behavior. Concurrent inserts (with other inserts)
//! and concurrent lookups (with other lookups) are supported; the two kinds must not be mixed.
//! @note `_Capacity` is a span-style `size_t` non-type parameter holding the *valid* (post-rounding)
//! slot count, or `cuda::std::dynamic_extent` (the default) for runtime-sized maps. Obtain a valid
//! value with `cuco::make_valid_capacity`.
//!
//! @tparam _Key Key type. Requires `cuda::is_bitwise_comparable_v<_Key>`
//! @tparam _Tp Mapped value type
//! @tparam _Capacity Requested slot count, or `cuda::std::dynamic_extent` for runtime sizing
//! @tparam _Scope Thread scope for atomic operations
//! @tparam _KeyEqual Key equality comparator
//! @tparam _ProbingScheme Probing scheme type
//! @tparam _BucketSize Slots per bucket
//! @tparam _MemoryResource Memory resource for device storage
template <class _Key,
class _Tp,
::cuda::std::size_t _Capacity = ::cuda::std::dynamic_extent,
::cuda::thread_scope _Scope = ::cuda::thread_scope_device,
class _KeyEqual = ::cuda::std::equal_to<_Key>,
class _ProbingScheme = linear_probing<4, hash<_Key>>,
int _BucketSize = 1,
class _MemoryResource = ::cuda::device_memory_pool_ref>
class fixed_capacity_map
{
public:
using key_type = _Key; ///< Key type
using mapped_type = _Tp; ///< Payload (mapped value) type
using value_type = ::cuda::std::pair<_Key, _Tp>; ///< Key-payload pair type
using size_type = ::cuda::std::size_t; ///< Size type
using key_equal = _KeyEqual; ///< Key equality comparator type
using probing_scheme_type = _ProbingScheme; ///< Probing scheme type
using hasher = typename probing_scheme_type::hasher; ///< Hash function type
static constexpr auto cg_size = _ProbingScheme::cg_size; ///< Cooperative-group size used for probing
static constexpr auto bucket_size = _BucketSize; ///< Number of slots per bucket
static constexpr auto thread_scope = _Scope; ///< CUDA thread scope for atomic operations
static_assert(_Capacity == ::cuda::std::dynamic_extent || is_valid_capacity<_ProbingScheme, _BucketSize>(_Capacity),
"Capacity must be a valid open-addressing capacity; obtain it via cuco::make_valid_capacity");
//! @brief Valid (post-rounding) slot count; `cuda::std::dynamic_extent` for dynamic maps.
static constexpr size_type capacity_v = _Capacity;
using ref_type =
fixed_capacity_map_ref<_Key, _Tp, _Scope, _KeyEqual, _ProbingScheme, _BucketSize, _Capacity>; ///< Device
///< non-owning
///< ref type
private:
using __impl_type = __open_addressing::
__open_addressing_impl<_Key, value_type, _Scope, _KeyEqual, _ProbingScheme, _BucketSize, _MemoryResource>;
::cuda::std::unique_ptr<__impl_type> __impl;
mapped_type __empty_value_sentinel;
//! @brief Synchronizes the CUDA stream.
static void __sync(::cuda::stream_ref __stream)
{
__stream.sync();
}
public:
//! @brief Constructs a map with static capacity (encoded in `_Capacity`) and no erasure.
//!
//! @param __stream Stream used for allocation and initialization
//! @param __mr Memory resource for device storage
//! @param __empty_key_sentinel Sentinel indicating an empty key slot
//! @param __empty_value_sentinel Sentinel indicating an empty payload
//! @param __pred Key equality binary callable
//! @param __probing_scheme Probing scheme
_CCCL_TEMPLATE(::cuda::std::size_t _C = _Capacity)
_CCCL_REQUIRES((_C != ::cuda::std::dynamic_extent))
_CCCL_HOST_API fixed_capacity_map(
::cuda::stream_ref __stream,
_MemoryResource __mr,
empty_key<_Key> __empty_key_sentinel,
empty_value<_Tp> __empty_value_sentinel,
const _KeyEqual& __pred = {},
const _ProbingScheme& __probing_scheme = {})
: __impl{::cuda::std::make_unique<__impl_type>(
__stream,
__mr,
_Capacity,
value_type{key_type(__empty_key_sentinel), mapped_type(__empty_value_sentinel)},
__pred,
__probing_scheme)}
, __empty_value_sentinel{mapped_type(__empty_value_sentinel)}
{}
//! @brief Constructs a map with dynamic capacity and no erasure.
//!
//! @param __stream Stream used for allocation and initialization
//! @param __mr Memory resource for device storage
//! @param __capacity Requested slot count (prime/stride-adjusted internally)
//! @param __empty_key_sentinel Sentinel indicating an empty key slot
//! @param __empty_value_sentinel Sentinel indicating an empty payload
//! @param __pred Key equality binary callable
//! @param __probing_scheme Probing scheme
_CCCL_TEMPLATE(::cuda::std::size_t _C = _Capacity)
_CCCL_REQUIRES((_C == ::cuda::std::dynamic_extent))
_CCCL_HOST_API fixed_capacity_map(
::cuda::stream_ref __stream,
_MemoryResource __mr,
size_type __capacity,
empty_key<_Key> __empty_key_sentinel,
empty_value<_Tp> __empty_value_sentinel,
const _KeyEqual& __pred = {},
const _ProbingScheme& __probing_scheme = {})
: __impl{::cuda::std::make_unique<__impl_type>(
__stream,
__mr,
__capacity,
value_type{key_type(__empty_key_sentinel), mapped_type(__empty_value_sentinel)},
__pred,
__probing_scheme)}
, __empty_value_sentinel{mapped_type(__empty_value_sentinel)}
{}
//! @brief Constructs a map sized by a target load factor (dynamic capacity only).
//!
//! @param __stream Stream used for allocation and initialization
//! @param __mr Memory resource for device storage
//! @param __n Expected number of keys
//! @param __desired_load_factor Target load factor in (0, 1]
//! @param __empty_key_sentinel Sentinel indicating an empty key slot
//! @param __empty_value_sentinel Sentinel indicating an empty payload
//! @param __pred Key equality binary callable
//! @param __probing_scheme Probing scheme
_CCCL_TEMPLATE(::cuda::std::size_t _C = _Capacity)
_CCCL_REQUIRES((_C == ::cuda::std::dynamic_extent))
_CCCL_HOST_API fixed_capacity_map(
::cuda::stream_ref __stream,
_MemoryResource __mr,
size_type __n,
double __desired_load_factor,
empty_key<_Key> __empty_key_sentinel,
empty_value<_Tp> __empty_value_sentinel,
const _KeyEqual& __pred = {},
const _ProbingScheme& __probing_scheme = {})
: __impl{::cuda::std::make_unique<__impl_type>(
__stream,
__mr,
__n,
__desired_load_factor,
value_type{key_type(__empty_key_sentinel), mapped_type(__empty_value_sentinel)},
__pred,
__probing_scheme)}
, __empty_value_sentinel{mapped_type(__empty_value_sentinel)}
{}
//! @brief Constructs a map with static capacity and erasure support.
//!
//! @param __stream Stream used for allocation and initialization
//! @param __mr Memory resource for device storage
//! @param __empty_key_sentinel Sentinel indicating an empty key slot
//! @param __empty_value_sentinel Sentinel indicating an empty payload
//! @param __erased_key_sentinel Sentinel indicating an erased key slot
//! @param __pred Key equality binary callable
//! @param __probing_scheme Probing scheme
_CCCL_TEMPLATE(::cuda::std::size_t _C = _Capacity)
_CCCL_REQUIRES((_C != ::cuda::std::dynamic_extent))
_CCCL_HOST_API fixed_capacity_map(
::cuda::stream_ref __stream,
_MemoryResource __mr,
empty_key<_Key> __empty_key_sentinel,
empty_value<_Tp> __empty_value_sentinel,
erased_key<_Key> __erased_key_sentinel,
const _KeyEqual& __pred = {},
const _ProbingScheme& __probing_scheme = {})
: __impl{::cuda::std::make_unique<__impl_type>(
__stream,
__mr,
_Capacity,
value_type{key_type(__empty_key_sentinel), mapped_type(__empty_value_sentinel)},
key_type(__erased_key_sentinel),
__pred,
__probing_scheme)}
, __empty_value_sentinel{mapped_type(__empty_value_sentinel)}
{}
//! @brief Constructs a map with dynamic capacity and erasure support.
//!
//! @param __stream Stream used for allocation and initialization
//! @param __mr Memory resource for device storage
//! @param __capacity Requested slot count (prime/stride-adjusted internally)
//! @param __empty_key_sentinel Sentinel indicating an empty key slot
//! @param __empty_value_sentinel Sentinel indicating an empty payload
//! @param __erased_key_sentinel Sentinel indicating an erased key slot
//! @param __pred Key equality binary callable
//! @param __probing_scheme Probing scheme
_CCCL_TEMPLATE(::cuda::std::size_t _C = _Capacity)
_CCCL_REQUIRES((_C == ::cuda::std::dynamic_extent))
_CCCL_HOST_API fixed_capacity_map(
::cuda::stream_ref __stream,
_MemoryResource __mr,
size_type __capacity,
empty_key<_Key> __empty_key_sentinel,
empty_value<_Tp> __empty_value_sentinel,
erased_key<_Key> __erased_key_sentinel,
const _KeyEqual& __pred = {},
const _ProbingScheme& __probing_scheme = {})
: __impl{::cuda::std::make_unique<__impl_type>(
__stream,
__mr,
__capacity,
value_type{key_type(__empty_key_sentinel), mapped_type(__empty_value_sentinel)},
key_type(__erased_key_sentinel),
__pred,
__probing_scheme)}
, __empty_value_sentinel{mapped_type(__empty_value_sentinel)}
{}
// ===== Clear =====
//! @brief Erases all elements from the container. After this call, `size()` returns zero.
//!
//! @param __stream CUDA stream this operation is executed in
void clear(::cuda::stream_ref __stream)
{
__impl->clear(__stream);
}
//! @brief Asynchronously erases all elements from the container. After this call, `size()`
//! returns zero.
//!
//! @param __stream CUDA stream this operation is executed in
void clear_async(::cuda::stream_ref __stream) noexcept
{
__impl->clear_async(__stream);
}
// ===== Insert =====
//! @brief Inserts all keys in the range `[__first, __last)` and returns the number of successful
//! insertions.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `insert_async`.
//!
//! @tparam _InputIt Device accessible random access input iterator whose `value_type` is
//! convertible to the map's `value_type`
//!
//! @param __stream CUDA stream used for insert
//! @param __first Beginning of the sequence of keys
//! @param __last End of the sequence of keys
//!
//! @return Number of successful insertions
template <class _InputIt>
size_type insert(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last)
{
return __impl->insert(__stream, __first, __last, ref());
}
//! @brief Asynchronously inserts all keys in the range `[__first, __last)`.
//!
//! @tparam _InputIt Device accessible random access input iterator whose `value_type` is
//! convertible to the map's `value_type`
//!
//! @param __stream CUDA stream used for insert
//! @param __first Beginning of the sequence of keys
//! @param __last End of the sequence of keys
template <class _InputIt>
void insert_async(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last) noexcept
{
__impl->insert_async(__stream, __first, __last, ref());
}
// ===== Contains =====
//! @brief Indicates whether each key in `[__first, __last)` is contained in the map.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `contains_async`.
//!
//! @tparam _InputIt Device accessible input iterator
//! @tparam _OutputIt Device accessible output iterator assignable from `bool`
//!
//! @param __stream CUDA stream used for executing the kernels
//! @param __first Beginning of the sequence of keys
//! @param __last End of the sequence of keys
//! @param __output_begin Beginning of the output sequence of booleans
template <class _InputIt, class _OutputIt>
void contains(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last, _OutputIt __output_begin) const
{
contains_async(__stream, __first, __last, __output_begin);
__sync(__stream);
}
//! @brief Asynchronously indicates whether each key in `[__first, __last)` is contained in the map.
//!
//! @tparam _InputIt Device accessible input iterator
//! @tparam _OutputIt Device accessible output iterator assignable from `bool`
//!
//! @param __stream CUDA stream used for executing the kernels
//! @param __first Beginning of the sequence of keys
//! @param __last End of the sequence of keys
//! @param __output_begin Beginning of the output sequence of booleans
template <class _InputIt, class _OutputIt>
void contains_async(
::cuda::stream_ref __stream, _InputIt __first, _InputIt __last, _OutputIt __output_begin) const noexcept
{
__impl->contains_async(__stream, __first, __last, __output_begin, ref());
}
// ===== Find =====
//! @brief For each key in `[__first, __last)` writes the associated payload, or `empty_value_sentinel()`
//! if the key is not present.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use `find_async`.
//!
//! @tparam _InputIt Device accessible input iterator
//! @tparam _OutputIt Device accessible output iterator assignable from `mapped_type`
//!
//! @param __stream CUDA stream used for executing the kernels
//! @param __first Beginning of the sequence of keys
//! @param __last End of the sequence of keys
//! @param __output_begin Beginning of the output sequence of payloads
template <class _InputIt, class _OutputIt>
void find(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last, _OutputIt __output_begin) const
{
find_async(__stream, __first, __last, __output_begin);
__sync(__stream);
}
//! @brief Asynchronously, for each key in `[__first, __last)` writes the associated payload, or
//! `empty_value_sentinel()` if the key is not present.
//!
//! @tparam _InputIt Device accessible input iterator
//! @tparam _OutputIt Device accessible output iterator assignable from `mapped_type`
//!
//! @param __stream CUDA stream used for executing the kernels
//! @param __first Beginning of the sequence of keys
//! @param __last End of the sequence of keys
//! @param __output_begin Beginning of the output sequence of payloads
template <class _InputIt, class _OutputIt>
void
find_async(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last, _OutputIt __output_begin) const noexcept
{
__impl->find_async(__stream, __first, __last, __output_begin, ref());
}
//! @brief For each key `__first[i]` with `__pred(__stencil[i]) == true` writes the associated payload,
//! or `empty_value_sentinel()` if the key is not present; writes `empty_value_sentinel()` for the rest.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use `find_if_async`.
//!
//! @tparam _InputIt Device accessible input iterator
//! @tparam _StencilIt Device accessible random access iterator whose value type is convertible to
//! `_Predicate`'s argument type
//! @tparam _Predicate Unary callable returning `bool`
//! @tparam _OutputIt Device accessible output iterator assignable from `mapped_type`
//!
//! @param __stream CUDA stream used for executing the kernels
//! @param __first Beginning of the sequence of keys
//! @param __last End of the sequence of keys
//! @param __stencil Beginning of the stencil sequence
//! @param __pred Predicate applied to the stencil to determine which keys to query
//! @param __output_begin Beginning of the output sequence of payloads
template <class _InputIt, class _StencilIt, class _Predicate, class _OutputIt>
void find_if(::cuda::stream_ref __stream,
_InputIt __first,
_InputIt __last,
_StencilIt __stencil,
_Predicate __pred,
_OutputIt __output_begin) const
{
find_if_async(__stream, __first, __last, __stencil, __pred, __output_begin);
__sync(__stream);
}
//! @brief Asynchronous version of `find_if`.
//!
//! @tparam _InputIt Device accessible input iterator
//! @tparam _StencilIt Device accessible random access iterator whose value type is convertible to
//! `_Predicate`'s argument type
//! @tparam _Predicate Unary callable returning `bool`
//! @tparam _OutputIt Device accessible output iterator assignable from `mapped_type`
//!
//! @param __stream CUDA stream used for executing the kernels
//! @param __first Beginning of the sequence of keys
//! @param __last End of the sequence of keys
//! @param __stencil Beginning of the stencil sequence
//! @param __pred Predicate applied to the stencil to determine which keys to query
//! @param __output_begin Beginning of the output sequence of payloads
template <class _InputIt, class _StencilIt, class _Predicate, class _OutputIt>
void find_if_async(
::cuda::stream_ref __stream,
_InputIt __first,
_InputIt __last,
_StencilIt __stencil,
_Predicate __pred,
_OutputIt __output_begin) const noexcept
{
__impl->find_if_async(__stream, __first, __last, __stencil, __pred, __output_begin, ref());
}
// ===== Accessors =====
//! @brief Returns the total number of slots the map can hold (the prime/stride-adjusted capacity).
//!
//! @return Total slot count
[[nodiscard]] constexpr size_type capacity() const noexcept
{
return __impl->capacity();
}
//! @brief Gets a device pointer to the underlying slot storage.
//!
//! @return Pointer to the underlying slot storage
[[nodiscard]] _CCCL_HOST_API value_type* data() const
{
return __impl->data();
}
//! @brief Gets the sentinel value used to represent an empty key slot.
//!
//! @return The sentinel value used to represent an empty key slot
[[nodiscard]] constexpr key_type empty_key_sentinel() const noexcept
{
return __impl->empty_key_sentinel();
}
//! @brief Gets the sentinel value used to represent an empty payload slot.
//!
//! @return The sentinel value used to represent an empty payload slot
[[nodiscard]] constexpr mapped_type empty_value_sentinel() const noexcept
{
return __empty_value_sentinel;
}
//! @brief Gets the sentinel value used to represent an erased key slot.
//!
//! @return The sentinel value used to represent an erased key slot
[[nodiscard]] constexpr key_type erased_key_sentinel() const noexcept
{
return __impl->erased_key_sentinel();
}
//! @brief Gets the function used to compare keys for equality.
//!
//! @return The function used to compare keys for equality
[[nodiscard]] constexpr key_equal key_eq() const noexcept
{
return __impl->key_eq();
}
//! @brief Gets the function(s) used to hash keys.
//!
//! @return The function(s) used to hash keys
[[nodiscard]] constexpr hasher hash_function() const noexcept
{
return __impl->hash_function();
}
//! @brief Gets a device-usable non-owning reference to this map.
//!
//! The returned ref borrows the map's slot storage and sentinel values and is trivially copyable
//! — safe to pass by value to kernels. The ref's lifetime must not exceed the map's lifetime.
//!
//! @return A `ref_type` referring to this map
[[nodiscard]] auto ref() const noexcept -> ref_type
{
auto __slots = typename ref_type::storage_span_type{__impl->storage_ref().data(), __impl->capacity()};
return detail::__bitwise_compare(empty_key_sentinel(), erased_key_sentinel())
? ref_type{empty_key{empty_key_sentinel()},
empty_value{empty_value_sentinel()},
__impl->key_eq(),
__impl->probing_scheme(),
__slots}
: ref_type{empty_key{empty_key_sentinel()},
empty_value{empty_value_sentinel()},
erased_key{erased_key_sentinel()},
__impl->key_eq(),
__impl->probing_scheme(),
__slots};
}
};
} // namespace cuda::experimental::cuco
#endif // !_CCCL_COMPILER(NVRTC)
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_FIXED_CAPACITY_MAP_CUH

View File

@@ -0,0 +1,354 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_FIXED_CAPACITY_MAP_REF_CUH
#define _CUDAX___CUCO_FIXED_CAPACITY_MAP_REF_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__atomic/atomic.h>
#include <cuda/__cmath/pow2.h>
#include <cuda/__type_traits/is_bitwise_comparable.h>
#include <cuda/std/__mdspan/extents.h>
#include <cuda/std/__utility/pair.h>
#include <cuda/std/span>
#include <cuda/experimental/__cuco/capacity.cuh>
#include <cuda/experimental/__cuco/detail/open_addressing/open_addressing_ref_impl.cuh>
#include <cuda/experimental/__cuco/detail/open_addressing/slot_storage_ref.cuh>
#include <cuda/experimental/__cuco/probing_scheme.cuh>
#include <cuda/experimental/__cuco/types.cuh>
#include <cooperative_groups.h>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief Device non-owning reference type for `fixed_capacity_map`.
//!
//! This lightweight, trivially-copyable reference is intended to be passed by value to device code
//! for performing insert and lookup operations on the hash map.
//!
//! @note Concurrent modify and lookup on the same map are not supported: lookups perform non-atomic
//! loads, so a lookup must not run concurrently with an insert (doing so is a data race).
//! @note cuCollections data structures always place the slot keys on the right-hand side when
//! invoking the key comparison predicate, i.e., `__pred(__query_key, __slot_key)`.
//! @note `_ProbingScheme::cg_size` indicates how many threads are used to handle one independent
//! device operation. `cg_size == 1` uses the scalar (or non-CG) code paths.
//! @note `_Capacity` is a span-style `size_t` non-type parameter encoding the *requested* slot
//! count. Pass `cuda::std::dynamic_extent` (the default) for runtime-sized maps; any concrete
//! value encodes the requested slot count at compile time. The actual slot count is the
//! prime/stride-adjusted value exposed as `capacity_v` and matches the owning map's
//! `fixed_capacity_map::capacity_v` for the same parameters.
//!
//! @tparam _Key Type used for keys
//! @tparam _Tp Type used for mapped values
//! @tparam _Scope The scope in which operations will be performed by individual threads
//! @tparam _KeyEqual Binary callable type used to compare two keys for equality
//! @tparam _ProbingScheme Probing scheme type
//! @tparam _BucketSize Number of slots per bucket
//! @tparam _Capacity Requested slot count, or `cuda::std::dynamic_extent` for runtime sizing
template <class _Key,
class _Tp,
::cuda::thread_scope _Scope,
class _KeyEqual,
class _ProbingScheme,
int _BucketSize,
::cuda::std::size_t _Capacity = ::cuda::std::dynamic_extent>
class fixed_capacity_map_ref
{
static_assert(sizeof(_Key) <= 8, "Container does not support key types larger than 8 bytes.");
static_assert(::cuda::is_power_of_two(sizeof(_Key)), "key_type size must be a power of two");
static_assert(sizeof(_Tp) <= 8, "sizeof(mapped_type) must be no larger than 8 bytes.");
static_assert(::cuda::is_power_of_two(sizeof(::cuda::std::pair<_Key, _Tp>)),
"value_type size must be a power of two");
static_assert(::cuda::is_bitwise_comparable_v<_Key>,
"Key type must have unique object representations or have been explicitly declared as safe for "
"bitwise comparison via specialization of cuda::is_bitwise_comparable_v<Key>.");
static constexpr bool __allows_duplicates = false;
static_assert(_Capacity == ::cuda::std::dynamic_extent || is_valid_capacity<_ProbingScheme, _BucketSize>(_Capacity),
"Capacity must be a valid open-addressing capacity; obtain it via cuco::make_valid_capacity");
public:
using key_type = _Key; ///< Key type
using mapped_type = _Tp; ///< Payload (mapped value) type
using value_type = ::cuda::std::pair<_Key, _Tp>; ///< Key-payload pair type
using probing_scheme_type = _ProbingScheme; ///< Probing scheme type
using hasher = typename probing_scheme_type::hasher; ///< Hash function type
using size_type = ::cuda::std::size_t; ///< Size type
using key_equal = _KeyEqual; ///< Key equality comparator type
using iterator = value_type*; ///< Slot iterator
using const_iterator = const value_type*; ///< Const slot iterator
static constexpr auto cg_size = probing_scheme_type::cg_size; ///< Cooperative-group size for probing
static constexpr auto bucket_size = _BucketSize; ///< Number of slots per bucket
static constexpr auto thread_scope = _Scope; ///< CUDA thread scope for atomic operations
//! @brief Compile-time adjusted slot count; `cuda::std::dynamic_extent` when `_Capacity` is dynamic.
static constexpr size_type capacity_v = _Capacity;
//! @brief Slot-storage span type. For static `_Capacity`, the span carries the adjusted
//! `capacity_v` extent at compile time; for dynamic `_Capacity`, the extent is dynamic.
using storage_span_type = ::cuda::std::span<value_type, capacity_v>;
private:
// Internal adapter to the open-addressing impl. The storage's `_Capacity` template arg receives
// the (already valid) `capacity_v`, so when `_Capacity` is static the slot count travels through
// the storage's extent at compile time and the probing iterator's modular reduction folds to a
// constant.
using __storage_ref_type = __open_addressing::__slot_storage_ref<value_type, _BucketSize, capacity_v>;
//! @brief Returns the slot count of the given span, validating it for the dynamic case.
//!
//! @param __slots Span over the slot storage
//!
//! @return The total slot count
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr size_type __checked_capacity(storage_span_type __slots) noexcept
{
if constexpr (_Capacity == ::cuda::std::dynamic_extent)
{
_CCCL_ASSERT((is_valid_capacity<_ProbingScheme, _BucketSize>(__slots.size())),
"storage size is not a valid capacity");
}
return __slots.size();
}
using __impl_type = __open_addressing::
__open_addressing_ref_impl<_Key, _Scope, _KeyEqual, _ProbingScheme, __storage_ref_type, __allows_duplicates>;
__impl_type __impl;
public:
//! @brief Constructs a ref without erasure support.
//!
//! @param __empty_key_sentinel Sentinel indicating an empty key slot
//! @param __empty_value_sentinel Sentinel indicating an empty payload
//! @param __predicate Key equality binary callable
//! @param __probing_scheme Probing scheme
//! @param __slots Span over the slot storage; must contain `capacity()` slots
_CCCL_HOST_DEVICE_API explicit constexpr fixed_capacity_map_ref(
empty_key<_Key> __empty_key_sentinel,
empty_value<_Tp> __empty_value_sentinel,
const _KeyEqual& __predicate,
const _ProbingScheme& __probing_scheme,
storage_span_type __slots) noexcept
: __impl{value_type{key_type(__empty_key_sentinel), mapped_type(__empty_value_sentinel)},
__predicate,
__probing_scheme,
__storage_ref_type{__slots.data(), __checked_capacity(__slots)}}
{}
//! @brief Constructs a ref with erasure support.
//!
//! @param __empty_key_sentinel Sentinel indicating an empty key slot
//! @param __empty_value_sentinel Sentinel indicating an empty payload
//! @param __erased_key_sentinel Sentinel indicating an erased key slot
//! @param __predicate Key equality binary callable
//! @param __probing_scheme Probing scheme
//! @param __slots Span over the slot storage; must contain `capacity()` slots
_CCCL_HOST_DEVICE_API explicit constexpr fixed_capacity_map_ref(
empty_key<_Key> __empty_key_sentinel,
empty_value<_Tp> __empty_value_sentinel,
erased_key<_Key> __erased_key_sentinel,
const _KeyEqual& __predicate,
const _ProbingScheme& __probing_scheme,
storage_span_type __slots) noexcept
: __impl{value_type{key_type(__empty_key_sentinel), mapped_type(__empty_value_sentinel)},
key_type(__erased_key_sentinel),
__predicate,
__probing_scheme,
__storage_ref_type{__slots.data(), __checked_capacity(__slots)}}
{}
// ===== Accessors =====
//! @brief Returns the total number of slots.
//!
//! @return Total slot count (equal to the owning map's `capacity()`)
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr size_type capacity() const noexcept
{
return __impl.capacity();
}
//! @brief Returns the sentinel value used to represent an empty key slot.
//!
//! @return The sentinel value used to represent an empty key slot
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr key_type empty_key_sentinel() const noexcept
{
return __impl.empty_key_sentinel();
}
//! @brief Returns the sentinel value used to represent an empty payload slot.
//!
//! @return The sentinel value used to represent an empty payload slot
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr mapped_type empty_value_sentinel() const noexcept
{
return __impl.empty_value_sentinel();
}
//! @brief Returns the sentinel value used to represent an erased key slot.
//!
//! @return The sentinel value used to represent an erased key slot
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr key_type erased_key_sentinel() const noexcept
{
return __impl.erased_key_sentinel();
}
//! @brief Returns the function used to compare keys for equality.
//!
//! @return The key equality comparator
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr key_equal key_eq() const noexcept
{
return __impl.key_eq();
}
//! @brief Returns the function(s) used to hash keys.
//!
//! @return The hasher used by this ref's probing scheme
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr hasher hash_function() const noexcept
{
return __impl.hash_function();
}
//! @brief Returns the probing scheme used to resolve hash collisions.
//!
//! @return The probing scheme object
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr probing_scheme_type probing_scheme() const noexcept
{
return __impl.probing_scheme();
}
//! @brief Returns a const iterator to one past the last slot (the end sentinel).
//!
//! @return Past-the-end const iterator
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr const_iterator end() const noexcept
{
return __impl.end();
}
//! @brief Returns an iterator to one past the last slot (the end sentinel).
//!
//! @return Past-the-end iterator
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr iterator end() noexcept
{
return __impl.end();
}
//! @brief Returns a span over the slot storage backing this ref.
//!
//! @return Span of `capacity()` slots
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr storage_span_type storage_span() const noexcept
{
return storage_span_type{__impl.storage_ref().data(), __impl.capacity()};
}
#if _CCCL_CUDA_COMPILATION()
// ===== Insert operations =====
//! @brief Inserts a key-value pair.
//!
//! @param __value The key-value pair to insert
//!
//! @return `true` if the pair was inserted, `false` if the key already exists
_CCCL_DEVICE_API bool insert(value_type __value) noexcept
{
return __impl.insert(__value);
}
//! @brief Inserts a key-value pair using a cooperative group.
//!
//! @tparam _ParentCG Parent cooperative group type
//!
//! @param __group The cooperative group used for this operation
//! @param __value The key-value pair to insert
//!
//! @return `true` if the pair was inserted, `false` if the key already exists
template <class _ParentCG>
_CCCL_DEVICE_API bool
insert(::cooperative_groups::thread_block_tile<cg_size, _ParentCG> __group, value_type __value) noexcept
{
return __impl.insert(__group, __value);
}
// ===== Lookup operations =====
//! @brief Checks if a key exists in the map.
//!
//! @param __key The key to search for
//!
//! @return `true` if the key is found
template <class _ProbeKey = key_type>
[[nodiscard]] _CCCL_DEVICE_API bool contains(_ProbeKey __key) const noexcept
{
return __impl.contains(__key);
}
//! @brief Cooperative-group variant of `contains`.
//!
//! @tparam _ParentCG Parent cooperative group type
//! @tparam _ProbeKey Probe key type (defaults to `key_type`)
//!
//! @param __group Cooperative group of size `cg_size` performing this lookup
//! @param __key The key to search for
//!
//! @return `true` if the key is found
template <class _ParentCG, class _ProbeKey = key_type>
[[nodiscard]] _CCCL_DEVICE_API bool
contains(::cooperative_groups::thread_block_tile<cg_size, _ParentCG> __group, _ProbeKey __key) const noexcept
{
return __impl.contains(__group, __key);
}
//! @brief Finds the slot associated with a key.
//!
//! @tparam _ProbeKey Probe key type (defaults to `key_type`)
//!
//! @param __key The key to search for
//!
//! @return An iterator to the slot holding `__key`, or `end()` if the key is not found
template <class _ProbeKey = key_type>
[[nodiscard]] _CCCL_DEVICE_API iterator find(_ProbeKey __key) const noexcept
{
return __impl.find(__key);
}
//! @brief Cooperative-group variant of `find`.
//!
//! @tparam _ParentCG Parent cooperative group type
//! @tparam _ProbeKey Probe key type (defaults to `key_type`)
//!
//! @param __group Cooperative group of size `cg_size` performing this lookup
//! @param __key The key to search for
//!
//! @return An iterator to the slot holding `__key`, or `end()` if the key is not found
template <class _ParentCG, class _ProbeKey = key_type>
[[nodiscard]] _CCCL_DEVICE_API iterator
find(::cooperative_groups::thread_block_tile<cg_size, _ParentCG> __group, _ProbeKey __key) const noexcept
{
return __impl.find(__group, __key);
}
#endif // _CCCL_CUDA_COMPILATION()
};
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_FIXED_CAPACITY_MAP_REF_CUH

View File

@@ -0,0 +1,97 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_HASH_FUNCTIONS_CUH
#define _CUDAX___CUCO_HASH_FUNCTIONS_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/experimental/__cuco/detail/hash_functions/murmurhash3.cuh>
#include <cuda/experimental/__cuco/detail/hash_functions/xxhash.cuh>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
enum class hash_algorithm
{
xxhash_32,
xxhash_64,
murmurhash3_32
#if _CCCL_HAS_INT128()
,
murmurhash3_x86_128,
murmurhash3_x64_128
#endif // _CCCL_HAS_INT128()
};
//! @brief A hash function class specialized for different hash algorithms.
//!
//! @tparam _Key The type of the values to hash
//! @tparam _S The hash strategy to use, defaults to `hash_algorithm::xxhash_32`
template <typename _Key, hash_algorithm _S = hash_algorithm::xxhash_32>
class hash;
template <typename _Key>
class hash<_Key, hash_algorithm::xxhash_32> : private ::cuda::experimental::cuco::_XXHash_32<_Key>
{
public:
using ::cuda::experimental::cuco::_XXHash_32<_Key>::_XXHash_32;
using ::cuda::experimental::cuco::_XXHash_32<_Key>::operator();
};
template <typename _Key>
class hash<_Key, hash_algorithm::xxhash_64> : private ::cuda::experimental::cuco::_XXHash_64<_Key>
{
public:
using ::cuda::experimental::cuco::_XXHash_64<_Key>::_XXHash_64;
using ::cuda::experimental::cuco::_XXHash_64<_Key>::operator();
};
template <typename _Key>
class hash<_Key, hash_algorithm::murmurhash3_32> : private ::cuda::experimental::cuco::_MurmurHash3_32<_Key>
{
public:
using ::cuda::experimental::cuco::_MurmurHash3_32<_Key>::_MurmurHash3_32;
using ::cuda::experimental::cuco::_MurmurHash3_32<_Key>::operator();
};
#if _CCCL_HAS_INT128()
template <typename _Key>
class hash<_Key, hash_algorithm::murmurhash3_x86_128> : private ::cuda::experimental::cuco::_MurmurHash3_x86_128<_Key>
{
public:
using ::cuda::experimental::cuco::_MurmurHash3_x86_128<_Key>::_MurmurHash3_x86_128;
using ::cuda::experimental::cuco::_MurmurHash3_x86_128<_Key>::operator();
};
template <typename _Key>
class hash<_Key, hash_algorithm::murmurhash3_x64_128> : private ::cuda::experimental::cuco::_MurmurHash3_x64_128<_Key>
{
public:
using ::cuda::experimental::cuco::_MurmurHash3_x64_128<_Key>::_MurmurHash3_x64_128;
using ::cuda::experimental::cuco::_MurmurHash3_x64_128<_Key>::operator();
};
#endif // _CCCL_HAS_INT128()
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_HASH_FUNCTIONS_CUH

View File

@@ -0,0 +1,26 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_HLL_POLICIES_CUH
#define _CUDAX___CUCO_HLL_POLICIES_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/experimental/__cuco/detail/hyperloglog/default_policy.cuh>
#endif // _CUDAX___CUCO_HLL_POLICIES_CUH

View File

@@ -0,0 +1,437 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_HYPERLOGLOG_CUH
#define _CUDAX___CUCO_HYPERLOGLOG_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__container/buffer.h>
#include <cuda/__memory_pool/device_memory_pool.h>
#include <cuda/__memory_resource/legacy_pinned_memory_resource.h>
#include <cuda/__stream/stream_ref.h>
#include <cuda/__utility/in_range.h>
#include <cuda/__utility/no_init.h>
#include <cuda/std/__bit/countr.h>
#include <cuda/std/__cccl/assert.h>
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/__host_stdlib/stdexcept>
#include <cuda/std/__utility/forward.h>
#include <cuda/std/span>
#include <cuda/experimental/__cuco/hll_policies.cuh>
#include <cuda/experimental/__cuco/hyperloglog_ref.cuh>
#include <cuda/std/__cccl/prologue.h>
#if !_CCCL_COMPILER(NVRTC)
namespace cuda::experimental::cuco
{
//! @brief A GPU-accelerated utility for approximating the number of distinct items in a multiset.
//!
//! @note This implementation is based on the HyperLogLog++ algorithm:
//! https://static.googleusercontent.com/media/research.google.com/de//pubs/archive/40671.pdf.
//!
//! @tparam _Tp Type of items to count
//! @tparam _MemoryResource Type of memory resource used for device storage
//! @tparam _Scope The scope in which operations will be performed by individual threads
//! @tparam _Policy Policy bundling hash function, bit-slicing rule, and finalizer
template <class _Tp,
class _MemoryResource = ::cuda::device_memory_pool_ref,
::cuda::thread_scope _Scope = ::cuda::thread_scope_device,
class _Policy = ::cuda::experimental::cuco::default_hll_policy<_Tp>>
class hyperloglog
{
public:
static constexpr auto thread_scope = _Scope; ///< CUDA thread scope
template <::cuda::thread_scope _NewScope = thread_scope>
using ref_type = hyperloglog_ref<_Tp, _NewScope, _Policy>; ///< Non-owning reference type
using value_type = typename ref_type<>::value_type; ///< Type of items to count
using policy_type = typename ref_type<>::policy_type; ///< Policy type
using hasher = typename ref_type<>::hasher; ///< Hash function type
using register_type = typename ref_type<>::register_type; ///< HLL register type
//! A strong type wrapper `sketch_size_kb` of `double`, for specifying the upper-bound
//! sketch size of `cuda::experimental::cuco::hyperloglog(_ref)` in KB.
//!
//! @note Valid sketch sizes are in [0.0625 KB, 1024 KB], which correspond to precision [4, 18].
using sketch_size_kb = ::cuda::experimental::cuco::__sketch_size_kb_t;
//! A strong type wrapper `standard_deviation` of `double`, for specifying the desired
//! standard deviation for the cardinality estimate of `cuda::experimental::cuco::hyperloglog(_ref)`.
//!
//! @note Valid standard deviations are approximately in [0.00216, 0.2765], which correspond to
//! precision [4, 18].
using standard_deviation = ::cuda::experimental::cuco::__standard_deviation_t;
//! A strong type wrapper `precision` of `int`, for specifying the HyperLogLog precision
//! parameter of `cuda::experimental::cuco::hyperloglog(_ref)`.
//!
//! @note Valid precision values are in [4, 18], which correspond to sketch sizes in
//! [0.0625 KB, 1024 KB] and standard deviations approximately in [0.00216, 0.2765].
using precision = ::cuda::experimental::cuco::__precision_t;
private:
::cuda::device_buffer<register_type> __sketch_buffer; ///< Storage for sketch
ref_type<> __ref; ///< Device ref of the current `hyperloglog` object
// Needs to be friends with other instantiations of this class template to have access to their
// storage
template <class _Tp_, class _MemoryResource_, ::cuda::thread_scope _Scope_, class _Policy_>
friend class hyperloglog;
public:
// TODO enable CTAD
//! @brief Constructs a `hyperloglog` host object.
//!
//! @note Construction is stream-ordered: the initial clear is enqueued on `__stream` without
//! synchronizing it.
//!
//! @param __stream CUDA stream used to initialize the object
//! @param __memory_resource A memory resource used for allocating device storage
//! @param __sketch_size_kb Maximum sketch size in KB
//! @param __policy The policy used to hash items and finalize the estimate
//!
//! @throw If sketch size implies precision outside [4, 18].
template <typename _MemoryResource_ = _MemoryResource>
_CCCL_HOST_API constexpr hyperloglog(
::cuda::stream_ref __stream,
_MemoryResource_&& __memory_resource,
sketch_size_kb __sketch_size_kb = sketch_size_kb{32.0},
const _Policy& __policy = {})
: hyperloglog{__stream,
::cuda::std::forward<_MemoryResource_>(__memory_resource),
__to_precision(__sketch_size_kb),
__policy}
{}
//! @brief Constructs a `hyperloglog` host object.
//!
//! @note Construction is stream-ordered: the initial clear is enqueued on `__stream` without
//! synchronizing it.
//!
//! @param __stream CUDA stream used to initialize the object
//! @param __memory_resource A memory resource used for allocating device storage
//! @param __sd Desired standard deviation for the approximation error
//! @param __policy The policy used to hash items and finalize the estimate
//!
//! @throw If standard deviation implies precision outside [4, 18].
template <typename _MemoryResource_ = _MemoryResource>
_CCCL_HOST_API constexpr hyperloglog(
::cuda::stream_ref __stream,
_MemoryResource_&& __memory_resource,
standard_deviation __sd,
const _Policy& __policy = {})
: hyperloglog{__stream, ::cuda::std::forward<_MemoryResource_>(__memory_resource), __to_precision(__sd), __policy}
{}
//! @brief Constructs a `hyperloglog` host object.
//!
//! @note Construction is stream-ordered: the initial clear is enqueued on `__stream` without
//! synchronizing it.
//!
//! @param __stream CUDA stream used to initialize the object
//! @param __memory_resource A memory resource used for allocating device storage
//! @param __precision HyperLogLog precision parameter (determines number of registers as 2^precision)
//! @param __policy The policy used to hash items and finalize the estimate
//!
//! @throw If precision is outside [4, 18].
template <typename _MemoryResource_ = _MemoryResource>
_CCCL_HOST_API constexpr hyperloglog(
::cuda::stream_ref __stream,
_MemoryResource_&& __memory_resource,
precision __precision,
const _Policy& __policy = {})
: __sketch_buffer{__stream,
::cuda::std::forward<_MemoryResource_>(__memory_resource),
ref_type<>::sketch_bytes(
__precision_in_bounds(__precision, "HyperLogLog precision must be in [4, 18]"))
/ sizeof(register_type),
::cuda::no_init}
, __ref{::cuda::std::as_writable_bytes(::cuda::std::span{__sketch_buffer.data(), __sketch_buffer.size()}),
__policy}
{
clear_async(__stream);
}
_CCCL_HIDE_FROM_ABI ~hyperloglog() = default;
hyperloglog(const hyperloglog&) = delete;
//! @brief Copy-assignment operator.
//!
//! @return Copy of `*this`
hyperloglog& operator=(const hyperloglog&) = delete;
_CCCL_HIDE_FROM_ABI hyperloglog(hyperloglog&&) = default; ///< Move constructor
_CCCL_HIDE_FROM_ABI hyperloglog& operator=(hyperloglog&&) = default;
//! @brief Asynchronously resets the estimator, i.e., clears the current count estimate.
//!
//! @param __stream CUDA stream this operation is executed in
_CCCL_HOST_API constexpr void clear_async(::cuda::stream_ref __stream) noexcept
{
__ref.clear_async(__stream);
}
//! @brief Resets the estimator, i.e., clears the current count estimate.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `clear_async`.
//!
//! @param __stream CUDA stream this operation is executed in
_CCCL_HOST_API constexpr void clear(::cuda::stream_ref __stream)
{
__ref.clear(__stream);
}
//! @brief Asynchronously adds to be counted items to the estimator.
//!
//! @tparam _InputIt Device accessible random access input iterator where
//! <tt>std::is_convertible<std::iterator_traits<_InputIt>::value_type,
//! _Tp></tt> is `true`
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __first Beginning of the sequence of items
//! @param __last End of the sequence of items
template <class _InputIt>
_CCCL_HOST_API constexpr void add_async(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last)
{
__ref.add_async(__stream, __first, __last);
}
//! @brief Adds to be counted items to the estimator.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `add_async`.
//!
//! @tparam _InputIt Device accessible random access input iterator where
//! <tt>std::is_convertible<std::iterator_traits<_InputIt>::value_type,
//! _Tp></tt> is `true`
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __first Beginning of the sequence of items
//! @param __last End of the sequence of items
template <class _InputIt>
_CCCL_HOST_API constexpr void add(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last)
{
__ref.add(__stream, __first, __last);
}
//! @brief Asynchronously merges the result of `other` estimator into `*this` estimator.
//!
//! @throw If sketch_bytes() != __other.sketch_bytes()
//!
//! @tparam _OtherScope Thread scope of `other` estimator
//! @tparam _OtherMemoryResource Memory resource type of `other` estimator
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __other Other estimator to be merged into `*this`
template <::cuda::thread_scope _OtherScope, class _OtherMemoryResource>
_CCCL_HOST_API constexpr void
merge_async(::cuda::stream_ref __stream, const hyperloglog<_Tp, _OtherMemoryResource, _OtherScope, _Policy>& __other)
{
__ref.merge_async(__stream, __other.__ref);
}
//! @brief Merges the result of `other` estimator into `*this` estimator.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `merge_async`.
//!
//! @throw If sketch_bytes() != __other.sketch_bytes()
//!
//! @tparam _OtherScope Thread scope of `other` estimator
//! @tparam _OtherMemoryResource Memory resource type of `other` estimator
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __other Other estimator to be merged into `*this`
template <::cuda::thread_scope _OtherScope, class _OtherMemoryResource>
_CCCL_HOST_API constexpr void
merge(::cuda::stream_ref __stream, const hyperloglog<_Tp, _OtherMemoryResource, _OtherScope, _Policy>& __other)
{
__ref.merge(__stream, __other.__ref);
}
//! @brief Asynchronously merges the result of `other` estimator reference into `*this` estimator.
//!
//! @throw If sketch_bytes() != __other.sketch_bytes()
//!
//! @tparam _OtherScope Thread scope of `other` estimator
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __other_ref Other estimator reference to be merged into `*this`
template <::cuda::thread_scope _OtherScope>
_CCCL_HOST_API constexpr void merge_async(::cuda::stream_ref __stream, const ref_type<_OtherScope>& __other_ref)
{
__ref.merge_async(__stream, __other_ref);
}
//! @brief Merges the result of `other` estimator reference into `*this` estimator.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `merge_async`.
//!
//! @throw If sketch_bytes() != __other.sketch_bytes()
//!
//! @tparam _OtherScope Thread scope of `other` estimator
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __other_ref Other estimator reference to be merged into `*this`
template <::cuda::thread_scope _OtherScope>
_CCCL_HOST_API constexpr void merge(::cuda::stream_ref __stream, const ref_type<_OtherScope>& __other_ref)
{
__ref.merge(__stream, __other_ref);
}
//! @brief Compute the estimated distinct items count.
//!
//! @note This function synchronizes the given stream.
//!
//! @tparam _MemoryResource Host memory resource used for allocating the host buffer required to
//! compute the final estimate by copying the sketch from device to host
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __host_mr Host memory resource used for copying the sketch
//!
//! @return Approximate distinct items count
template <typename _HostMemoryResource = ::cuda::mr::legacy_pinned_memory_resource>
[[nodiscard]] _CCCL_HOST_API constexpr double
estimate(::cuda::stream_ref __stream, _HostMemoryResource __host_mr = {}) const
{
return __ref.estimate(__stream, __host_mr);
}
//! @brief Get device ref.
//!
//! @return Device ref object of the current `hyperloglog` host object
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ref_type<> ref() const noexcept
{
return {sketch(), policy()};
}
//! @brief Get hash function.
//!
//! @return The hash function
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto hash_function() const noexcept
{
return __ref.hash_function();
}
//! @brief Get the policy.
//!
//! @return The policy
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr const _Policy& policy() const noexcept
{
return __ref.policy();
}
//! @brief Gets the span of the sketch.
//!
//! @return The ::cuda::std::span of the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::span<::cuda::std::byte> sketch() const noexcept
{
return __ref.sketch();
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::size_t sketch_bytes() const noexcept
{
return __ref.sketch_bytes();
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param __sketch_size_kb Upper bound sketch size in KB
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t
sketch_bytes(sketch_size_kb __sketch_size_kb) noexcept
{
return ref_type<>::sketch_bytes(__sketch_size_kb);
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param __standard_deviation Upper bound standard deviation for approximation error
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t
sketch_bytes(standard_deviation __standard_deviation) noexcept
{
return ref_type<>::sketch_bytes(__standard_deviation);
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param __precision HyperLogLog precision parameter
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t sketch_bytes(precision __precision) noexcept
{
return ref_type<>::sketch_bytes(__precision);
}
//! @brief Gets the alignment required for the sketch storage.
//!
//! @return The required alignment
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t sketch_alignment() noexcept
{
return ref_type<>::sketch_alignment();
}
private:
[[nodiscard]] _CCCL_HOST_API static constexpr precision
__precision_in_bounds(precision __precision, const char* __message)
{
const auto __value = static_cast<::cuda::std::int32_t>(__precision);
const auto __in_range = ::cuda::in_range(__value, 4, 18);
if (!__in_range)
{
_CCCL_THROW(::std::invalid_argument, __message);
}
return __precision;
}
[[nodiscard]] _CCCL_HOST_API static constexpr precision __to_precision(sketch_size_kb __sketch_size_kb)
{
const auto __bytes = ref_type<>::sketch_bytes(__sketch_size_kb) / sizeof(register_type);
const auto __precision = static_cast<int>(::cuda::std::countr_zero(static_cast<::cuda::std::size_t>(__bytes)));
return __precision_in_bounds(
precision{__precision}, "HyperLogLog sketch size must be in range [0.0625 KB, 1024 KB]");
}
[[nodiscard]] _CCCL_HOST_API static constexpr precision __to_precision(standard_deviation __standard_deviation)
{
const auto __bytes = ref_type<>::sketch_bytes(__standard_deviation) / sizeof(register_type);
const auto __precision = static_cast<int>(::cuda::std::countr_zero(static_cast<::cuda::std::size_t>(__bytes)));
return __precision_in_bounds(
precision{__precision}, "HyperLogLog standard deviation must be in range [0.00216, 0.2765]");
}
};
} // namespace cuda::experimental::cuco
#endif // !_CCCL_COMPILER(NVRTC)
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_HYPERLOGLOG_CUH

View File

@@ -0,0 +1,364 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_HYPERLOGLOG_REF_CUH
#define _CUDAX___CUCO_HYPERLOGLOG_REF_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__stream/stream_ref.h>
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/__type_traits/is_convertible.h>
#include <cuda/std/span>
#include <cuda/experimental/__cuco/detail/hyperloglog/hyperloglog_impl.cuh>
#include <cuda/experimental/__cuco/hll_policies.cuh>
#include <cooperative_groups.h>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief A non-owning reference to a HyperLogLog sketch for approximating the number of distinct
//! items in a multiset.
//!
//! @note This implementation is based on the HyperLogLog++ algorithm:
//! https://static.googleusercontent.com/media/research.google.com/de//pubs/archive/40671.pdf.
//!
//! @tparam _Tp Type of items to count
//! @tparam _Scope The scope in which operations will be performed by individual threads
//! @tparam _Policy Policy bundling hash function, bit-slicing rule, and finalizer
template <class _Tp,
::cuda::thread_scope _Scope = ::cuda::thread_scope_device,
class _Policy = ::cuda::experimental::cuco::default_hll_policy<_Tp>>
class hyperloglog_ref
{
using __impl_type = ::cuda::experimental::cuco::__hyperloglog_impl<_Tp, _Scope, _Policy>;
__impl_type __impl; ///< Implementation object
template <class _Tp_, ::cuda::thread_scope _Scope_, class _Policy_>
friend class hyperloglog_ref;
public:
static constexpr auto thread_scope = __impl_type::__thread_scope; ///< CUDA thread scope
using value_type = typename __impl_type::__value_type; ///< Type of items to count
using policy_type = typename __impl_type::__policy_type; ///< Policy type
using hasher = typename __impl_type::__hasher; ///< Type of hash function
using register_type = typename __impl_type::__register_type; ///< HLL register type
//! A strong type wrapper `sketch_size_kb` of `double`, for specifying the upper-bound
//! sketch size of `cuda::experimental::cuco::hyperloglog(_ref)` in KB.
//!
//! @note Valid sketch sizes are in [0.0625 KB, 1024 KB], which correspond to precision [4, 18].
using sketch_size_kb = ::cuda::experimental::cuco::__sketch_size_kb_t;
//! A strong type wrapper `standard_deviation` of `double`, for specifying the desired
//! standard deviation for the cardinality estimate of `cuda::experimental::cuco::hyperloglog(_ref)`.
//!
//! @note Valid standard deviations are approximately in [0.00216, 0.2765], which correspond to
//! precision [4, 18].
using standard_deviation = ::cuda::experimental::cuco::__standard_deviation_t;
//! A strong type wrapper `precision` of `int`, for specifying the HyperLogLog precision
//! parameter of `cuda::experimental::cuco::hyperloglog(_ref)`.
//!
//! @note Valid precision values are in [4, 18], which correspond to sketch sizes in
//! [0.0625 KB, 1024 KB] and standard deviations approximately in [0.00216, 0.2765].
using precision = ::cuda::experimental::cuco::__precision_t;
template <::cuda::thread_scope _NewScope>
using rebind_scope = hyperloglog_ref<_Tp, _NewScope, _Policy>; ///< Ref type with different thread scope
//! @brief Constructs a non-owning `hyperloglog_ref` object.
//!
//! @throw If sketch size < 0.0625KB or 64B or standard deviation > 0.2765. Throws if called from
//! host; __trap() if called from device.
//! @throw If sketch size or standard deviation imply precision outside [4, 18].
//! @throw If sketch storage has insufficient alignment. Throws if called from host; __trap() if called
//! from device.
//!
//! @param __sketch_span Reference to sketch storage
//! @param __policy The policy used to hash items and finalize the estimate
_CCCL_HOST_DEVICE_API constexpr hyperloglog_ref(::cuda::std::span<::cuda::std::byte> __sketch_span,
const _Policy& __policy = {})
: __impl{__sketch_span, __policy}
{}
//! @brief Resets the estimator, i.e., clears the current count estimate.
//!
//! @tparam _CG CUDA Cooperative Group type
//!
//! @param __group CUDA Cooperative group this operation is executed in
_CCCL_TEMPLATE(class _CG)
_CCCL_REQUIRES((!::cuda::std::is_convertible_v<_CG, ::cuda::stream_ref>) )
_CCCL_DEVICE_API constexpr void clear(_CG __group) noexcept
{
// The constraint above is to work around an incompatibility between host and device
// overload preference for clang and NVCC. See
// https://llvm.org/docs/CompileCudaWithLLVM.html#overloading-based-on-host-and-device-attributes
// for further reading, but the bottom line is when:
//
// 1. Compiling in device mode (and clang compiles CUDA in a "hybrid" host-device mode,
// also explained by the link above).
// 2. And the current function is __host__ __device__.
// 3. And the function whose overload needs to be resolved has both a __host__ __device__,
// and __device__ (and/or __host__) overload.
//
// Then clang will prefer these overloads (assuming they have equal priority under C++
// rules) in the following order:
//
// 1. __host__ __device__
// 2. __device__
// 3. __host__
//
// In this particular case, `clear(_CG)` conflicts with `clear(::cuda::stream_ref)` when called
// from `hyperloglog::clear(::cuda::stream_ref)`. `hyperloglog::clear(::cuda::stream_ref)`
// is constexpr, and therefore implicitly __host__ __device__. Since
// `clear(::cuda::stream_ref)` on this class is only __host__, it will take lower priority
// that `clear(_CG)`, and we get:
//
// cudax/include/cuda/experimental/__cuco/detail/hyperloglog/hyperloglog_impl.cuh:131:28: error: no member named
// 'thread_rank' in 'cuda::stream_ref' [clang-diagnostic-error]
//
// 131 | for (int __i = __group.thread_rank(); __i < __sketch.size(); __i += __group.size())
// | ~~~~~~~ ^
__impl.__clear(__group);
}
//! @brief Asynchronously resets the estimator, i.e., clears the current count estimate.
//!
//! @param __stream CUDA stream this operation is executed in
_CCCL_HOST_API constexpr void clear_async(::cuda::stream_ref __stream) noexcept
{
__impl.__clear_async(__stream);
}
//! @brief Resets the estimator, i.e., clears the current count estimate.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `clear_async`.
//!
//! @param __stream CUDA stream this operation is executed in
_CCCL_HOST_API constexpr void clear(::cuda::stream_ref __stream)
{
__impl.__clear(__stream);
}
//! @brief Adds an item to the estimator.
//!
//! @param __item The item to be counted
_CCCL_DEVICE_API constexpr void add(const _Tp& __item) noexcept
{
__impl.__add(__item);
}
//! @brief Asynchronously adds to be counted items to the estimator.
//!
//! @tparam _InputIt Device accessible random access input iterator where
//! <tt>std::is_convertible<std::iterator_traits<_InputIt>::value_type,
//! _Tp></tt> is `true`
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __first Beginning of the sequence of items
//! @param __last End of the sequence of items
template <class _InputIt>
_CCCL_HOST_API constexpr void add_async(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last)
{
__impl.__add_async(__first, __last, __stream);
}
//! @brief Adds to be counted items to the estimator.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `add_async`.
//!
//! @tparam _InputIt Device accessible random access input iterator where
//! <tt>std::is_convertible<std::iterator_traits<_InputIt>::value_type,
//! _Tp></tt> is `true`
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __first Beginning of the sequence of items
//! @param __last End of the sequence of items
template <class _InputIt>
_CCCL_HOST_API constexpr void add(::cuda::stream_ref __stream, _InputIt __first, _InputIt __last)
{
__impl.__add(__first, __last, __stream);
}
//! @brief Merges the result of `other` estimator reference into `*this` estimator reference.
//!
//! @throw If sketch_bytes() != __other.sketch_bytes(), then terminates execution with a device __trap()
//!
//! @tparam _CG CUDA Cooperative Group type
//! @tparam _OtherScope Thread scope of `other` estimator
//!
//! @param __group CUDA Cooperative group this operation is executed in
//! @param __other Other estimator reference to be merged into `*this`
_CCCL_TEMPLATE(class _CG, ::cuda::thread_scope _OtherScope)
_CCCL_REQUIRES((!::cuda::std::is_convertible_v<_CG, ::cuda::stream_ref>) )
_CCCL_DEVICE_API constexpr void merge(_CG __group, const hyperloglog_ref<_Tp, _OtherScope, _Policy>& __other)
{
// The constraint above works around the same host/device overload preference issue as
// documented in `clear(_CG)`: `merge(_CG, ...)` would otherwise conflict with
// `merge(::cuda::stream_ref, ...)` when called from `hyperloglog::merge(::cuda::stream_ref, ...)`.
__impl.__merge(__group, __other.__impl);
}
//! @brief Asynchronously merges the result of `other` estimator reference into `*this`
//! estimator.
//!
//! @throw If sketch_bytes() != __other.sketch_bytes()
//!
//! @tparam _OtherScope Thread scope of `other` estimator
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __other Other estimator reference to be merged into `*this`
template <::cuda::thread_scope _OtherScope>
_CCCL_HOST_API constexpr void
merge_async(::cuda::stream_ref __stream, const hyperloglog_ref<_Tp, _OtherScope, _Policy>& __other)
{
__impl.__merge_async(__other.__impl, __stream);
}
//! @brief Merges the result of `other` estimator reference into `*this` estimator.
//!
//! @note This function synchronizes the given stream. For asynchronous execution use
//! `merge_async`.
//!
//! @throw If sketch_bytes() != __other.sketch_bytes()
//!
//! @tparam _OtherScope Thread scope of `other` estimator
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __other Other estimator reference to be merged into `*this`
template <::cuda::thread_scope _OtherScope>
_CCCL_HOST_API constexpr void
merge(::cuda::stream_ref __stream, const hyperloglog_ref<_Tp, _OtherScope, _Policy>& __other)
{
__impl.__merge(__other.__impl, __stream);
}
//! @brief Compute the estimated distinct items count.
//!
//! @param __group CUDA thread block group this operation is executed in
//!
//! @return Approximate distinct items count
[[nodiscard]] _CCCL_DEVICE_API double estimate(const ::cooperative_groups::thread_block& __group) const noexcept
{
return __impl.__estimate(__group);
}
//! @brief Compute the estimated distinct items count.
//!
//! @note This function synchronizes the given stream.
//!
//! @tparam _HostMemoryResource Host memory resource used for allocating the host buffer required to
//! compute the final estimate by copying the sketch from device to host
//!
//! @param __stream CUDA stream this operation is executed in
//! @param __host_mr Host memory resource used for copying the sketch
//!
//! @return Approximate distinct items count
template <typename _HostMemoryResource = ::cuda::mr::legacy_pinned_memory_resource>
[[nodiscard]] _CCCL_HOST_API constexpr double
estimate(::cuda::stream_ref __stream, _HostMemoryResource __host_mr = {}) const
{
return __impl.__estimate(__host_mr, __stream);
}
//! @brief Gets the hash function.
//!
//! @return The hash function
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto hash_function() const noexcept
{
return __impl.__hash_function();
}
//! @brief Gets the policy.
//!
//! @return The policy
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr const _Policy& policy() const noexcept
{
return __impl.__policy_();
}
//! @brief Gets the span of the sketch.
//!
//! @return The ::cuda::std::span of the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::span<::cuda::std::byte> sketch() const noexcept
{
return __impl.__sketch_span();
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr ::cuda::std::size_t sketch_bytes() const noexcept
{
return __impl.__sketch_bytes();
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param __sketch_size_kb Upper bound sketch size in KB
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t
sketch_bytes(sketch_size_kb __sketch_size_kb) noexcept
{
return __impl_type::__sketch_bytes(__sketch_size_kb);
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param __standard_deviation Upper bound standard deviation for approximation error
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t
sketch_bytes(standard_deviation __standard_deviation) noexcept
{
return __impl_type::sketch_bytes(__standard_deviation);
}
//! @brief Gets the number of bytes required for the sketch storage.
//!
//! @param __precision HyperLogLog precision parameter
//!
//! @return The number of bytes required for the sketch
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t sketch_bytes(precision __precision) noexcept
{
return __impl_type::sketch_bytes(__precision);
}
//! @brief Gets the alignment required for the sketch storage.
//!
//! @return The required alignment
[[nodiscard]] _CCCL_HOST_DEVICE_API static constexpr ::cuda::std::size_t sketch_alignment() noexcept
{
return __impl_type::__sketch_alignment();
}
};
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_HYPERLOGLOG_REF_CUH

View File

@@ -0,0 +1,273 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_PROBING_SCHEME_CUH
#define _CUDAX___CUCO_PROBING_SCHEME_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__mdspan/extents.h>
#include <cuda/std/__tuple_dir/get.h>
#include <cuda/std/__tuple_dir/tuple.h>
#include <cuda/std/__tuple_dir/tuple_like.h>
#include <cuda/std/__tuple_dir/tuple_size.h>
#include <cuda/std/__type_traits/decay.h>
#include <cuda/experimental/__cuco/detail/probing_scheme_base.cuh>
#include <cooperative_groups.h>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief Public linear probing scheme class.
//!
//! @note Linear probing is efficient when few collisions are present, e.g., low occupancy or low
//! multiplicity.
//!
//! @note `_Hash` should be a callable object type.
//!
//! @tparam _CgSize Cooperative group size
//! @tparam _Hash Hash functor type
template <int _CgSize, class _Hash>
class linear_probing : detail::__probing_scheme_base<_CgSize>
{
using __base_type = detail::__probing_scheme_base<_CgSize>;
public:
static constexpr int cg_size = __base_type::__cg_size;
using hasher = _Hash;
//! @brief Constructs a linear probing scheme with the given hasher callable.
//!
//! @param __hash Hasher
_CCCL_HOST_DEVICE_API constexpr linear_probing(const _Hash& __hash = {})
: __hash{__hash}
{}
//! @brief Makes a copy of the current probing scheme with the given hasher.
//!
//! @tparam _NewHash New hasher type
//!
//! @param __hash Hasher
//!
//! @return Copy of the current probing scheme
template <class _NewHash>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto rebind_hash_function(const _NewHash& __hash) const noexcept
{
return linear_probing<cg_size, _NewHash>{__hash};
}
//! @brief Returns a probing iterator.
//!
//! @tparam _BucketSize Size of the bucket
//! @tparam _ProbeKey Type of probing key
//! @tparam _Capacity Capacity extent type (total slots)
//!
//! @param __probe_key The probing key
//! @param __cap Capacity extent (total slots) bounding the iteration
//!
//! @return An iterator whose value_type is convertible to the slot index type
template <int _BucketSize, class _ProbeKey, class _Capacity>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto make_iterator(_ProbeKey __probe_key, _Capacity __cap) const noexcept
{
using __size_type = typename _Capacity::index_type;
using __step_extent = ::cuda::std::extents<__size_type, _BucketSize>;
const __size_type __init = __hash(__probe_key) % (__cap.extent(0) / _BucketSize) * _BucketSize;
return detail::__probing_iterator<_Capacity, __step_extent>{__init, __step_extent{}, __cap};
}
//! @brief Returns a cooperative group based probing iterator.
//!
//! @tparam _BucketSize Size of the bucket
//! @tparam _ProbeKey Type of probing key
//! @tparam _Capacity Capacity extent type (total slots)
//! @tparam _ParentCG Type of parent cooperative group
//!
//! @param __group The cooperative group used to generate the probing iterator
//! @param __probe_key The probing key
//! @param __cap Capacity extent (total slots) bounding the iteration
//!
//! @return An iterator whose value_type is convertible to the slot index type
template <int _BucketSize, class _ProbeKey, class _Capacity, class _ParentCG>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto
make_iterator(::cooperative_groups::thread_block_tile<cg_size, _ParentCG> __group,
_ProbeKey __probe_key,
_Capacity __cap) const noexcept
{
using __size_type = typename _Capacity::index_type;
constexpr __size_type __stride = cg_size * _BucketSize;
using __step_extent = ::cuda::std::extents<__size_type, __stride>;
const __size_type __init =
__hash(__probe_key) % (__cap.extent(0) / __stride) * __stride + __size_type{__group.thread_rank() * _BucketSize};
return detail::__probing_iterator<_Capacity, __step_extent>{__init, __step_extent{}, __cap};
}
//! @brief Gets the function used to hash keys.
//!
//! @return The function used to hash keys
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr hasher hash_function() const noexcept
{
return __hash;
}
private:
_Hash __hash;
};
//! @brief Public double hashing scheme class.
//!
//! @note Default probing scheme for cuco data structures. It shows superior performance over linear
//! probing especially when dealing with high multiplicity and/or high occupancy use cases.
//!
//! @note `_Hash1` and `_Hash2` should be callable object types.
//!
//! @note `_Hash2` needs to be able to construct from an integer value to avoid secondary clustering.
//!
//! @tparam _CgSize Cooperative group size
//! @tparam _Hash1 First hash functor
//! @tparam _Hash2 Second hash functor
template <int _CgSize, class _Hash1, class _Hash2 = _Hash1>
class double_hashing : detail::__probing_scheme_base<_CgSize>
{
using __base_type = detail::__probing_scheme_base<_CgSize>;
public:
static constexpr int cg_size = __base_type::__cg_size;
using hasher = ::cuda::std::tuple<_Hash1, _Hash2>;
//! @brief Constructs a double hashing probing scheme with the two hasher callables.
//!
//! @param __hash1 First hasher
//! @param __hash2 Second hasher
_CCCL_HOST_DEVICE_API constexpr double_hashing(const _Hash1& __hash1 = {}, const _Hash2& __hash2 = {1})
: __hash1{__hash1}
, __hash2{__hash2}
{}
//! @brief Constructs a double hashing probing scheme with the given hasher tuple.
//!
//! @param __hash Hasher tuple
_CCCL_HOST_DEVICE_API constexpr double_hashing(const ::cuda::std::tuple<_Hash1, _Hash2>& __hash)
: __hash1{::cuda::std::get<0>(__hash)}
, __hash2{::cuda::std::get<1>(__hash)}
{}
//! @brief Makes a copy of the current probing scheme with the given hasher.
//!
//! @tparam _NewHash Tuple-like new hasher type
//!
//! @param __hash Hasher
//!
//! @return Copy of the current probing scheme
_CCCL_TEMPLATE(class _NewHash)
_CCCL_REQUIRES(::cuda::std::__tuple_like<_NewHash>)
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto rebind_hash_function(const _NewHash& __hash) const
{
static_assert(::cuda::std::__tuple_like<_NewHash> && ::cuda::std::tuple_size<_NewHash>::value == 2,
"The given hasher must be a tuple-like object with exactly two elements");
const auto& [__hash1, __hash2] = __hash;
using __hash1_type = ::cuda::std::decay_t<decltype(__hash1)>;
using __hash2_type = ::cuda::std::decay_t<decltype(__hash2)>;
return double_hashing<cg_size, __hash1_type, __hash2_type>{__hash1, __hash2};
}
//! @brief Returns a probing iterator.
//!
//! @tparam _BucketSize Size of the bucket
//! @tparam _ProbeKey Type of probing key
//! @tparam _Capacity Capacity extent type (total slots)
//!
//! @param __probe_key The probing key
//! @param __cap Capacity extent (total slots) bounding the iteration
//!
//! @return An iterator whose value_type is convertible to the slot index type
template <int _BucketSize, class _ProbeKey, class _Capacity>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto make_iterator(_ProbeKey __probe_key, _Capacity __cap) const noexcept
{
using __size_type = typename _Capacity::index_type;
using __step_extent = ::cuda::std::extents<__size_type, ::cuda::std::dynamic_extent>;
return detail::__probing_iterator<_Capacity, __step_extent>{
__size_type{__hash1(__probe_key)} % (__cap.extent(0) / _BucketSize) * _BucketSize,
__step_extent{__size_type{(__hash2(__probe_key) % (__cap.extent(0) / _BucketSize - 1) + 1) * _BucketSize}},
__cap};
}
//! @brief Returns a cooperative group based probing iterator.
//!
//! @tparam _BucketSize Size of the bucket
//! @tparam _ProbeKey Type of probing key
//! @tparam _Capacity Capacity extent type (total slots)
//! @tparam _ParentCG Type of parent cooperative group
//!
//! @param __group The cooperative group used to generate the probing iterator
//! @param __probe_key The probing key
//! @param __cap Capacity extent (total slots) bounding the iteration
//!
//! @return An iterator whose value_type is convertible to the slot index type
template <int _BucketSize, class _ProbeKey, class _Capacity, class _ParentCG>
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr auto
make_iterator(::cooperative_groups::thread_block_tile<cg_size, _ParentCG> __group,
_ProbeKey __probe_key,
_Capacity __cap) const noexcept
{
using __size_type = typename _Capacity::index_type;
constexpr __size_type __stride = cg_size * _BucketSize;
using __step_extent = ::cuda::std::extents<__size_type, ::cuda::std::dynamic_extent>;
return detail::__probing_iterator<_Capacity, __step_extent>{
__size_type{__hash1(__probe_key)} % (__cap.extent(0) / __stride) * __stride
+ __size_type{__group.thread_rank() * _BucketSize},
__step_extent{__size_type{(__hash2(__probe_key) % (__cap.extent(0) / __stride - 1) + 1) * __stride}},
__cap};
}
//! @brief Gets the functions used to hash keys.
//!
//! @return The functions used to hash keys
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr hasher hash_function() const noexcept
{
return {__hash1, __hash2};
}
private:
_Hash1 __hash1;
_Hash2 __hash2;
};
//! @brief Trait value indicating whether a probing scheme is double hashing.
//!
//! @tparam _Tp Input probing scheme type
template <class _Tp>
inline constexpr bool is_double_hashing_v = false;
//! @brief Specialization indicating that `double_hashing` is a double hashing scheme.
//!
//! @tparam _CgSize Cooperative group size
//! @tparam _Hash1 First hash functor
//! @tparam _Hash2 Second hash functor
template <int _CgSize, class _Hash1, class _Hash2>
inline constexpr bool is_double_hashing_v<double_hashing<_CgSize, _Hash1, _Hash2>> = true;
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_PROBING_SCHEME_CUH

View File

@@ -0,0 +1,66 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDAX___CUCO_TYPES_CUH
#define _CUDAX___CUCO_TYPES_CUH
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/experimental/__cuco/detail/utility/strong_type.cuh>
#include <cuda/std/__cccl/prologue.h>
namespace cuda::experimental::cuco
{
//! @brief Strong type wrapper for an empty key sentinel.
//!
//! @tparam _Key The key type
template <class _Key>
struct empty_key : __strong_type<_Key>
{
_CCCL_HOST_DEVICE_API explicit constexpr empty_key(_Key __value) noexcept
: __strong_type<_Key>(__value)
{}
};
//! @brief Strong type wrapper for an empty value sentinel.
//!
//! @tparam _Tp The mapped value type
template <class _Tp>
struct empty_value : __strong_type<_Tp>
{
_CCCL_HOST_DEVICE_API explicit constexpr empty_value(_Tp __value) noexcept
: __strong_type<_Tp>(__value)
{}
};
//! @brief Strong type wrapper for an erased key sentinel.
//!
//! @tparam _Key The key type
template <class _Key>
struct erased_key : __strong_type<_Key>
{
_CCCL_HOST_DEVICE_API explicit constexpr erased_key(_Key __value) noexcept
: __strong_type<_Key>(__value)
{}
};
} // namespace cuda::experimental::cuco
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDAX___CUCO_TYPES_CUH