[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,49 @@
if (NOT CCCL_ENABLE_LIBCUDACXX)
include(cmake/libcudacxxAddSubdir.cmake)
return()
endif()
cmake_minimum_required(VERSION 3.21)
set(PACKAGE_NAME libcudacxx)
set(PACKAGE_VERSION 11.0)
set(PACKAGE_STRING "${PACKAGE_NAME} ${PACKAGE_VERSION}")
project(libcudacxx LANGUAGES CXX)
# Add codegen module
option(
libcudacxx_ENABLE_CODEGEN
"Enable libcudacxx's atomics backend codegen and tests."
OFF
)
if (libcudacxx_ENABLE_CODEGEN)
add_subdirectory(codegen)
endif()
list(PREPEND CMAKE_MODULE_PATH "${libcudacxx_SOURCE_DIR}/cmake")
set(LLVM_PATH "${libcudacxx_SOURCE_DIR}" CACHE STRING "" FORCE)
# Configuration options.
option(LIBCUDACXX_ENABLE_CUDA "Enable the CUDA language support." ON)
if (LIBCUDACXX_ENABLE_CUDA)
enable_language(CUDA)
endif()
option(LIBCUDACXX_ENABLE_LIBCUDACXX_TESTS "Enable libcu++ tests." ON)
if (LIBCUDACXX_ENABLE_LIBCUDACXX_TESTS)
enable_testing()
# Create the compiler targets with the common compile flags
include(cmake/LibcudacxxBuildCompilerTargets.cmake)
libcudacxx_build_compiler_targets()
# Test all public and internal headers
include(cmake/LibcudacxxInternalHeaderTesting.cmake)
include(cmake/LibcudacxxPublicHeaderTesting.cmake)
include(cmake/LibcudacxxPublicHeaderTestingHost.cmake)
add_subdirectory(test)
endif()
if (CCCL_ENABLE_BENCHMARKS)
add_subdirectory(benchmarks)
endif()

View File

@@ -0,0 +1,311 @@
==============================================================================
libcu++ is under the Apache License v2.0 with LLVM Exceptions:
==============================================================================
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
---- LLVM Exceptions to the Apache 2.0 License ----
As an exception, if, as a result of your compiling your source code, portions
of this Software are embedded into an Object form of such source code, you
may redistribute such embedded portions in such Object form without complying
with the conditions of Sections 4(a), 4(b) and 4(d) of the License.
In addition, if you combine or link compiled forms of this Software with
software that is licensed under the GPLv2 ("Combined Software") and if a
court of competent jurisdiction determines that the patent provision (Section
3), the indemnity provision (Section 9) or other Section of the License
conflicts with the conditions of the GPLv2, you may retroactively and
prospectively choose to deem waived or otherwise exclude such Section(s) of
the License, but only in their entirety and only with respect to the Combined
Software.
==============================================================================
Software from third parties included in the LLVM Project:
==============================================================================
The LLVM Project contains third party software which is under different license
terms. All such code will be identified clearly using at least one of two
mechanisms:
1) It will be in a separate directory tree with its own `LICENSE.txt` or
`LICENSE` file at the top containing the specific license and restrictions
which apply to that software, or
2) It will contain specific license and restriction terms at the top of every
file.
==============================================================================
Legacy LLVM License (https://llvm.org/docs/DeveloperPolicy.html#legacy):
==============================================================================
The libc++ library is dual licensed under both the University of Illinois
"BSD-Like" license and the MIT license. As a user of this code you may choose
to use it under either license. As a contributor, you agree to allow your code
to be used under both.
Full text of the relevant licenses is included below.
==============================================================================
University of Illinois/NCSA
Open Source License
Copyright (c) 2009-2019 by the contributors listed in CREDITS.TXT
All rights reserved.
Developed by:
LLVM Team
University of Illinois at Urbana-Champaign
http://llvm.org
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal with
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
of the Software, and to permit persons to whom the Software is furnished to do
so, subject to the following conditions:
* Redistributions of source code must retain the above copyright notice,
this list of conditions and the following disclaimers.
* Redistributions in binary form must reproduce the above copyright notice,
this list of conditions and the following disclaimers in the
documentation and/or other materials provided with the distribution.
* Neither the names of the LLVM Team, University of Illinois at
Urbana-Champaign, nor the names of its contributors may be used to
endorse or promote products derived from this Software without specific
prior written permission.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS WITH THE
SOFTWARE.
==============================================================================
Copyright (c) 2009-2014 by the contributors listed in CREDITS.TXT
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.

View File

@@ -0,0 +1,92 @@
include(${CMAKE_SOURCE_DIR}/benchmarks/cmake/CCCLBenchmarkRegistry.cmake)
cccl_get_nvbench_helper()
set(benches_root "${CMAKE_CURRENT_LIST_DIR}")
if (NOT CMAKE_BUILD_TYPE STREQUAL "Release")
set(message_type FATAL_ERROR)
if (CCCL_ENABLE_CLANG_TIDY)
# We are here because CI has force-enabled clang-tidy. We must use a debug build for
# this because certain clang-tidy checks (such as out of bounds or clang static
# analyzer) work better when they see assert()'s. In this case we don't actually
# intend to run any of the benchmarks, we just need them to be compilable, so a simple
# warning is enough.
#
# We don't ignore this outright (by making it say, DEBUG or VERBOSE), because it's
# possible that a user may accidentally stumble into enabling the option.
set(message_type WARNING)
endif()
message(${message_type} "libcu++ benchmarks must be built in release mode.")
endif()
if (NOT DEFINED CMAKE_CUDA_ARCHITECTURES)
message(
FATAL_ERROR
"CMAKE_CUDA_ARCHITECTURES must be set to build libcu++ benchmarks."
)
endif()
set(benches_meta_target libcudacxx.all.benches)
add_custom_target(${benches_meta_target})
function(get_recursive_subdirs subdirs)
set(dirs)
file(
GLOB_RECURSE contents
CONFIGURE_DEPENDS
LIST_DIRECTORIES ON
"${CMAKE_CURRENT_LIST_DIR}/bench/*"
)
foreach (test_dir IN LISTS contents)
if (IS_DIRECTORY "${test_dir}")
list(APPEND dirs "${test_dir}")
endif()
endforeach()
set(${subdirs} "${dirs}" PARENT_SCOPE)
endfunction()
create_benchmark_registry()
function(add_bench target_name bench_name bench_src)
set(bench_target ${bench_name})
set(${target_name} ${bench_target} PARENT_SCOPE)
cccl_add_executable(${bench_target} SOURCES "${bench_src}")
target_link_libraries(
${bench_target}
PRIVATE libcudacxx::libcudacxx cccl.nvbench_helper nvbench::main
)
endfunction()
function(add_bench_dir bench_dir)
file(GLOB bench_srcs CONFIGURE_DEPENDS "${bench_dir}/*.cu")
file(RELATIVE_PATH bench_prefix "${benches_root}" "${bench_dir}")
file(TO_CMAKE_PATH "${bench_prefix}" bench_prefix)
string(REPLACE "/" "." bench_prefix "${bench_prefix}")
foreach (bench_src IN LISTS bench_srcs)
# base tuning
get_filename_component(bench_name "${bench_src}" NAME_WLE)
string(PREPEND bench_name "libcudacxx.${bench_prefix}.")
set(base_bench_name "${bench_name}.base")
add_bench(base_bench_target ${base_bench_name} "${bench_src}")
add_dependencies(${benches_meta_target} ${base_bench_target})
target_compile_definitions(${base_bench_target} PRIVATE TUNE_BASE=1)
target_compile_options(
${base_bench_target}
PRIVATE "$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--extended-lambda>"
)
# benchmarking
register_cccl_benchmark("${bench_name}" "")
endforeach()
endfunction()
get_recursive_subdirs(subdirs)
foreach (subdir IN LISTS subdirs)
add_bench_dir("${subdir}")
endforeach()

View File

@@ -0,0 +1,67 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/adjacent_difference.h>
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> out(elements);
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc;
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::adjacent_difference(cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> out(elements);
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::adjacent_difference(
cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin(), ::cuda::std::greater<T>{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,77 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/sequence.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
thrust::device_vector<T> in(elements, thrust::no_init);
thrust::sequence(in.begin(), in.end(), 0);
in[mismatch_point] = in[mismatch_point + 1];
state.add_element_count(elements);
state.add_global_memory_reads<T>(mismatch_point);
state.add_global_memory_writes<T>(0);
caching_allocator_t alloc;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
thrust::device_vector<T> in(elements, thrust::no_init);
thrust::sequence(in.begin(), in.end(), 0);
in[mismatch_point] = in[mismatch_point + 1];
state.add_element_count(elements);
state.add_global_memory_reads<T>(mismatch_point);
state.add_global_memory_writes<T>(0);
caching_allocator_t alloc;
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend(), ::cuda::std::greater<T>{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,48 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::all_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,48 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::any_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,70 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("contiguous")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void random_access(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::copy(
cuda_policy(alloc, launch),
cuda::counting_iterator<std::size_t>{0},
cuda::counting_iterator{elements},
out.begin()));
});
}
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
.set_name("random_access")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,51 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct is_even
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return static_cast<int>(val) % 2 == 0;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,67 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::copy_n(cuda_policy(alloc, launch), in.begin(), elements, out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("contiguous")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void random_access(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::copy_n(cuda_policy(alloc, launch), cuda::counting_iterator<std::size_t>{0}, elements, out.begin()));
});
}
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
.set_name("random_access")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,41 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::count(cuda_policy(alloc, launch), in.begin(), in.end(), T{42}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,50 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct equal_to_42
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return val == 42;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::count_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,82 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/iterator>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::equal(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
});
}
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base_range_iter")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void range_range(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::equal(
cuda_policy(alloc, launch),
dinput.begin(),
dinput.end(),
cuda::constant_iterator<T>{0},
cuda::constant_iterator<T>{0, elements}));
});
}
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base_range_range")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,68 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter_init(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
});
}
NVBENCH_BENCH_TYPES(range_iter_init, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_init")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::exclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, ::cuda::std::plus<T>{}));
});
}
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_init_op")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,43 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, max_t{}));
});
}
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_init_op")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,40 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> output(elements);
state.add_element_count(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::fill(cuda_policy(alloc, launch), output.begin(), output.end(), T{42});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,40 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> output(elements);
state.add_element_count(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::fill_n(cuda_policy(alloc, launch), output.begin(), elements, T{42}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,46 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::find(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), val));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,48 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::find_if(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,48 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::find_if_not(
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::not_fn(cuda::equal_to_value{val})));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,52 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct square_t
{
__device__ void operator()(T& x) const
{
x = x * x;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in(elements, T{1});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
square_t<T> op{};
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::for_each(cuda_policy(alloc, launch), in.begin(), in.end(), op);
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,52 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct square_t
{
__device__ void operator()(T& x) const
{
x = x * x;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in(elements, T{1});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
square_t<T> op{};
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::for_each_n(cuda_policy(alloc, launch), in.begin(), elements, op));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,42 @@
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
struct generator
{
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
{
return 42;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> output(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::generate(cuda_policy(alloc, launch), output.begin(), output.end(), generator<T>{});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,42 @@
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
struct generator
{
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
{
return 42;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> output(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::generate_n(cuda_policy(alloc, launch), output.begin(), elements, generator<T>{});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,94 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::inclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}));
});
}
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_op")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::inclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}, T{42}));
});
}
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_op_init")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,69 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}));
});
}
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_op")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}, T{42}));
});
}
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_op_init")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,85 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/fill.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
// All-zero is a valid heap; setting one element to 1 forces a violation at
// that child index since its parent is still 0.
template <typename T>
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
{
thrust::fill(d.begin(), d.end(), T{0});
if (violation_point >= 1 && violation_point < d.size())
{
d[violation_point] = T{1};
}
}
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto violation_frac = state.get_float64("ViolationAt");
const auto violation_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
prepare_input(dinput, violation_point);
state.add_global_memory_reads<T>(2 * violation_point);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto violation_frac = state.get_float64("ViolationAt");
const auto violation_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
prepare_input(dinput, violation_point);
state.add_global_memory_reads<T>(2 * violation_point);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,85 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/fill.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
// All-zero is a valid heap; setting one element to 1 forces a violation at
// that child index since its parent is still 0.
template <typename T>
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
{
thrust::fill(d.begin(), d.end(), T{0});
if (violation_point >= 1 && violation_point < d.size())
{
d[violation_point] = T{1};
}
}
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto violation_frac = state.get_float64("ViolationAt");
const auto violation_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
prepare_input(dinput, violation_point);
state.add_global_memory_reads<T>(2 * violation_point);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto violation_frac = state.get_float64("ViolationAt");
const auto violation_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
prepare_input(dinput, violation_point);
state.add_global_memory_reads<T>(2 * violation_point);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,50 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/partition.h>
#include <thrust/sequence.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
using select_op_t = less_then_t<T>;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
state.add_global_memory_reads<T>(2 * elements);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_partitioned(
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), select_op_t{static_cast<T>(mismatch_point)}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,78 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/sequence.h>
#include <thrust/sort.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
dinput[mismatch_point] = T{-1};
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
dinput[mismatch_point] = T{-1};
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::greater<>{}));
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,78 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/sequence.h>
#include <thrust/sort.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
dinput[mismatch_point] = T{-1};
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
dinput[mismatch_point] = T{-1};
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,66 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/extrema.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,95 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/execution_policy.h>
#include <thrust/merge.h>
#include <thrust/sort.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
const auto entropy = str_to_entropy(state.get_string("Entropy"));
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
thrust::device_vector<T> out(elements);
thrust::device_vector<T> in = generate(elements, entropy);
thrust::sort(in.begin(), in.begin() + elements_in_lhs);
thrust::sort(in.begin() + elements_in_lhs, in.end());
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::merge(
cuda_policy(alloc, launch),
in.cbegin(),
in.cbegin() + elements_in_lhs,
in.cbegin() + elements_in_lhs,
in.cend(),
out.begin());
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.201"})
.add_int64_axis("InputSizeRatio", {25, 50, 75});
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
const auto entropy = str_to_entropy(state.get_string("Entropy"));
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
thrust::device_vector<T> out(elements);
thrust::device_vector<T> in = generate(elements, entropy);
thrust::sort(in.begin(), in.begin() + elements_in_lhs, ::cuda::std::greater<T>{});
thrust::sort(in.begin() + elements_in_lhs, in.end(), ::cuda::std::greater<T>{});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::merge(
cuda_policy(alloc, launch),
in.cbegin(),
in.cbegin() + elements_in_lhs,
in.cbegin() + elements_in_lhs,
in.cend(),
out.begin(),
::cuda::std::greater<T>{});
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.201"})
.add_int64_axis("InputSizeRatio", {25, 50, 75});

View File

@@ -0,0 +1,66 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/extrema.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,82 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/iterator>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::mismatch(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
});
}
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base_range_iter")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void range_range(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::mismatch(
cuda_policy(alloc, launch),
dinput.begin(),
dinput.end(),
cuda::constant_iterator<T>{0},
cuda::constant_iterator<T>{0, elements}));
});
}
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base_range_range")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,48 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::none_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -0,0 +1,48 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/partition.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
using select_op_t = less_then_t<T>;
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
select_op_t select_op{val};
thrust::device_vector<T> input = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});

View File

@@ -0,0 +1,55 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/partition.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
using select_op_t = less_then_t<T>;
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
select_op_t select_op{val};
thrust::device_vector<T> input = generate(elements);
thrust::device_vector<T> output(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::partition_copy(
cuda_policy(alloc, launch),
input.begin(),
input.end(),
output.begin(),
cuda::std::make_reverse_iterator(output.begin() + elements),
select_op));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});

View File

@@ -0,0 +1,41 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::reduce(cuda_policy(alloc, launch), in.begin(), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,43 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/complex>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements - count);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::remove(cuda_policy(alloc, launch), in.begin(), in.end(), T{42});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,44 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/complex>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements - count);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::remove_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,52 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct is_even
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return static_cast<int>(val) % 2 == 0;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::remove_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,50 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct is_even
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return static_cast<int>(val) % 2 == 0;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::remove_if(cuda_policy(alloc, launch), in.begin(), in.end(), is_even{});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,41 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::replace(cuda_policy(alloc, launch), in.begin(), in.end(), 42, 1337);
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,42 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::replace_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), 42, 1337));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,52 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct equal_to_42
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return val == static_cast<T>(42);
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::replace_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), equal_to_42{}, 1337));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,50 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct equal_to_42
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return val == static_cast<T>(42);
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::replace_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}, 1337);
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,42 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/reverse.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::reverse(cuda_policy(alloc, launch), in.begin(), in.end());
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,43 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/reverse.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::reverse_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,44 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto midpoint_float = state.get_float64("MidpointAt");
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::rotate(cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});

View File

@@ -0,0 +1,45 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto midpoint_float = state.get_float64("MidpointAt");
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::rotate_copy(
cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});

View File

@@ -0,0 +1,43 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto midpoint_float = state.get_float64("ShiftedTo");
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements - midpoint);
state.add_global_memory_writes<T>(elements - midpoint);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::shift_left(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.5, 0.01});

View File

@@ -0,0 +1,42 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto midpoint_float = state.get_float64("ShiftedTo");
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements - midpoint);
state.add_global_memory_writes<T>(elements - midpoint);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::shift_right(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.6, 0.45, 0.01});

View File

@@ -0,0 +1,81 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/sort.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
thrust::device_vector<T> in = generate(elements, entropy);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end());
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.201"});
struct fake_less
{
template <class T, class U>
[[nodiscard]] _CCCL_API constexpr bool operator()(const T& t, const U& u) const
{
// complex is not less than comparable, so just compare the first element
if constexpr (cuda::std::__is_cpp17_less_than_comparable_v<T, U>)
{
return t < u;
}
else
{
return cuda::std::get<0>(t) < cuda::std::get<0>(u);
}
}
};
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
thrust::device_vector<T> in = generate(elements, entropy);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end(), fake_less{});
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(all_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.201"});

View File

@@ -0,0 +1,48 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/partition.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
using select_op_t = less_then_t<T>;
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
select_op_t select_op{val};
thrust::device_vector<T> input = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::stable_partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});

View File

@@ -0,0 +1,72 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/swap.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in1 = generate(elements);
thrust::device_vector<T> in2 = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(2 * elements);
state.add_global_memory_writes<T>(2 * elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::swap_ranges(cuda_policy(alloc, launch), in1.begin(), in1.end(), in2.begin());
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_iter_swap(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in1 = generate(elements);
thrust::device_vector<T> in2 = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(2 * elements);
state.add_global_memory_writes<T>(2 * elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::swap_ranges(
cuda_policy(alloc, launch),
cuda::std::reverse_iterator{in1.end()},
cuda::std::reverse_iterator{in1.begin()},
cuda::std::reverse_iterator{in2.end()});
});
}
NVBENCH_BENCH_TYPES(with_iter_swap, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_iter_swap")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,156 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/execution_policy.h>
#include <thrust/iterator/zip_iterator.h>
#include <cuda/functional>
#include <cuda/iterator>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include <nvbench_helper.cuh>
// The benchmarks are inspired by the BabelStream thrust version:
// https://github.com/UoB-HPC/BabelStream/blob/main/src/thrust/ThrustStream.cu
// Modified from BabelStream to also work for integers
constexpr auto startA = 1; // BabelStream: 0.1
constexpr auto startB = 2; // BabelStream: 0.2
constexpr auto startC = 3; // BabelStream: 0.1
constexpr auto startScalar = 4; // BabelStream: 0.4
using element_types = nvbench::type_list<std::int8_t, std::int16_t, float, double, __int128>;
// Different benchmarks use a different number of buffers. H200/B200 can fit 2^31 elements for all benchmarks and types.
// Upstream BabelStream uses 2^25. Allocation failure just skips the benchmark
auto array_size_powers = std::vector<std::int64_t>{25, 31};
template <typename T>
static void mul(nvbench::state& state, nvbench::type_list<T>)
{
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> b(n, startB);
thrust::device_vector<T> c(n, startC);
state.add_element_count(n);
state.add_global_memory_reads<T>(n);
state.add_global_memory_writes<T>(n);
caching_allocator_t alloc{};
const T scalar = startScalar;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform(
cuda_policy(alloc, launch), c.begin(), c.end(), b.begin(), [=] _CCCL_HOST_DEVICE(const T& ci) {
return ci * scalar;
}));
});
}
NVBENCH_BENCH_TYPES(mul, NVBENCH_TYPE_AXES(element_types))
.set_name("mul")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", array_size_powers);
template <typename T>
static void add(nvbench::state& state, nvbench::type_list<T>)
{
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> a(n, startA);
thrust::device_vector<T> b(n, startB);
thrust::device_vector<T> c(n, startC);
state.add_element_count(n);
state.add_global_memory_reads<T>(2 * n);
state.add_global_memory_writes<T>(n);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform(
cuda_policy(alloc, launch), a.begin(), a.end(), b.begin(), c.begin(), cuda::std::plus<T>{}));
});
}
NVBENCH_BENCH_TYPES(add, NVBENCH_TYPE_AXES(element_types))
.set_name("add")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", array_size_powers);
template <typename T>
static void triad(nvbench::state& state, nvbench::type_list<T>)
{
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> a(n, startA);
thrust::device_vector<T> b(n, startB);
thrust::device_vector<T> c(n, startC);
state.add_element_count(n);
state.add_global_memory_reads<T>(2 * n);
state.add_global_memory_writes<T>(n);
caching_allocator_t alloc{};
const T scalar = startScalar;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform(
cuda_policy(alloc, launch),
b.begin(),
b.end(),
c.begin(),
a.begin(),
[=] _CCCL_HOST_DEVICE(const T& bi, const T& ci) {
return bi + scalar * ci;
}));
});
}
NVBENCH_BENCH_TYPES(triad, NVBENCH_TYPE_AXES(element_types))
.set_name("triad")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", array_size_powers);
template <typename T>
static void nstream(nvbench::state& state, nvbench::type_list<T>)
{
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> a(n, startA);
thrust::device_vector<T> b(n, startB);
thrust::device_vector<T> c(n, startC);
state.add_element_count(n);
state.add_global_memory_reads<T>(3 * n);
state.add_global_memory_writes<T>(n);
caching_allocator_t alloc{};
const T scalar = startScalar;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform(
cuda_policy(alloc, launch),
cuda::make_zip_iterator(a.begin(), b.begin(), c.begin()),
cuda::make_zip_iterator(a.end(), b.end(), c.end()),
a.begin(),
cuda::zip_function{[=] _CCCL_HOST_DEVICE(const T& ai, const T& bi, const T& ci) {
return ai + bi + scalar * ci;
}}));
});
}
NVBENCH_BENCH_TYPES(nstream, NVBENCH_TYPE_AXES(element_types))
.set_name("nstream")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", array_size_powers);

View File

@@ -0,0 +1,75 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/execution_policy.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include <nvbench_helper.cuh>
template <class InT, class OutT>
struct fib_t
{
__device__ OutT operator()(InT n)
{
OutT t1 = 0;
OutT t2 = 1;
if (n <= 1)
{
return t1;
}
else if (n == 2)
{
return t2;
}
for (InT i = 3; i <= n; ++i)
{
const auto next = t1 + t2;
t1 = t2;
t2 = next;
}
return t2;
}
};
template <typename T>
static void fib(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> input = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> output(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<nvbench::uint32_t>(elements);
fib_t<T, nvbench::uint32_t> op{};
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::transform(cuda_policy(alloc, launch), input.cbegin(), input.cend(), output.begin(), op));
});
}
using types = nvbench::type_list<nvbench::uint32_t, nvbench::uint64_t>;
NVBENCH_BENCH_TYPES(fib, NVBENCH_TYPE_AXES(types))
.set_name("fib")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,53 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/transform_scan.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct times_two
{
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
{
return 2 * val;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_exclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, cuda::std::plus<T>{}, times_two<T>{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,79 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/transform_scan.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct times_two
{
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
{
return 2 * val;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_inclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("basic")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_init(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_inclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}, T{42}));
});
}
NVBENCH_BENCH_TYPES(with_init, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_init")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,49 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/iterator>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void binary(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_reduce(
cuda_policy(alloc, launch),
in.begin(),
in.end(),
cuda::constant_iterator<int>{42},
42,
cuda::std::plus<T>{},
cuda::std::multiplies<T>{}));
});
}
NVBENCH_BENCH_TYPES(binary, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,52 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct plus_one
{
template <class U>
[[nodiscard]] __device__ constexpr T operator()(const U val) const noexcept
{
return static_cast<T>(val + 1);
}
};
template <typename T>
static void unary(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_reduce(
cuda_policy(alloc, launch), in.begin(), in.end(), 42, cuda::std::plus<T>{}, plus_one<T>{}));
});
}
NVBENCH_BENCH_TYPES(unary, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,88 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/iterator/counting_iterator.h>
#include <thrust/transform.h>
#include <thrust/unique.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream_ref>
#include "nvbench_helper.cuh"
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
template <typename T>
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
{
in.resize(elements);
thrust::transform(
thrust::counting_iterator<std::size_t>(0),
thrust::counting_iterator<std::size_t>(elements),
in.begin(),
[] __device__(std::size_t i) {
// This seems like a clang-tidy bug. Yes we end up converting to double, but the division
// is done entirely in integer land...
return static_cast<T>(i / 2ULL); // NOLINT(bugprone-integer-division)
});
}
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in;
make_unique_input(in, elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
// unique writes at most elements
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in;
make_unique_input(in, elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
// unique writes at most elements
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end(), cuda::std::equal_to<T>{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -0,0 +1,90 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/iterator/counting_iterator.h>
#include <thrust/transform.h>
#include <thrust/unique.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream_ref>
#include "nvbench_helper.cuh"
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
template <typename T>
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
{
in.resize(elements);
thrust::transform(
thrust::counting_iterator<std::size_t>(0),
thrust::counting_iterator<std::size_t>(elements),
in.begin(),
[] __device__(std::size_t i) {
const auto run = i / 2;
return static_cast<T>(run);
});
}
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in;
make_unique_input(in, elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
// unique_copy writes at most elements
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::unique_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in;
make_unique_input(in, elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
// unique_copy writes at most elements
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::unique_copy(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::equal_to<T>{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,11 @@
# Determine if the compiler has GCC-compatible command-line syntax.
if (NOT DEFINED LLVM_COMPILER_IS_GCC_COMPATIBLE)
if (CMAKE_COMPILER_IS_GNUCXX)
set(LLVM_COMPILER_IS_GCC_COMPATIBLE ON)
elseif (MSVC)
set(LLVM_COMPILER_IS_GCC_COMPATIBLE OFF)
elseif ("${CMAKE_CXX_COMPILER_ID}" MATCHES "Clang")
set(LLVM_COMPILER_IS_GCC_COMPATIBLE ON)
endif()
endif()

View File

@@ -0,0 +1,35 @@
# Returns the host triple.
# Invokes config.guess
function(get_host_triple var)
if (MSVC)
if (CMAKE_SIZEOF_VOID_P EQUAL 8)
set(value "x86_64-pc-windows-msvc")
else()
set(value "i686-pc-windows-msvc")
endif()
elseif (MINGW AND NOT MSYS)
if (CMAKE_SIZEOF_VOID_P EQUAL 8)
set(value "x86_64-w64-windows-gnu")
else()
set(value "i686-pc-windows-gnu")
endif()
else(MSVC)
if (CMAKE_HOST_SYSTEM_NAME STREQUAL Windows AND NOT MSYS)
message(WARNING "unable to determine host target triple")
else()
set(config_guess ${LLVM_PATH}/cmake/config.guess)
execute_process(
COMMAND sh ${config_guess}
RESULT_VARIABLE TT_RV
OUTPUT_VARIABLE TT_OUT
OUTPUT_STRIP_TRAILING_WHITESPACE
)
if (NOT TT_RV EQUAL 0)
message(FATAL_ERROR "Failed to execute ${config_guess}")
endif(NOT TT_RV EQUAL 0)
set(value ${TT_OUT})
endif()
endif(MSVC)
set(${var} ${value} PARENT_SCOPE)
endfunction(get_host_triple var)

View File

@@ -0,0 +1,376 @@
function(get_system_libs return_var)
message(AUTHOR_WARNING "get_system_libs no longer needed")
set(${return_var} "" PARENT_SCOPE)
endfunction()
function(link_system_libs target)
message(AUTHOR_WARNING "link_system_libs no longer needed")
endfunction()
# is_llvm_target_library(
# library
# Name of the LLVM library to check
# return_var
# Output variable name
# ALL_TARGETS;INCLUDED_TARGETS;OMITTED_TARGETS
# ALL_TARGETS - default looks at the full list of known targets
# INCLUDED_TARGETS - looks only at targets being configured
# OMITTED_TARGETS - looks only at targets that are not being configured
# )
function(is_llvm_target_library library return_var)
cmake_parse_arguments(
ARG
"ALL_TARGETS;INCLUDED_TARGETS;OMITTED_TARGETS"
""
""
${ARGN}
)
# Sets variable `return_var' to ON if `library' corresponds to a
# LLVM supported target. To OFF if it doesn't.
set(${return_var} OFF PARENT_SCOPE)
string(TOUPPER "${library}" capitalized_lib)
if (ARG_INCLUDED_TARGETS)
string(TOUPPER "${LLVM_TARGETS_TO_BUILD}" targets)
elseif (ARG_OMITTED_TARGETS)
set(omitted_targets ${LLVM_ALL_TARGETS})
list(REMOVE_ITEM omitted_targets ${LLVM_TARGETS_TO_BUILD})
string(TOUPPER "${omitted_targets}" targets)
else()
string(TOUPPER "${LLVM_ALL_TARGETS}" targets)
endif()
foreach (t ${targets})
if (
capitalized_lib STREQUAL t
OR capitalized_lib STREQUAL "${t}"
OR capitalized_lib STREQUAL "${t}DESC"
OR capitalized_lib STREQUAL "${t}CODEGEN"
OR capitalized_lib STREQUAL "${t}ASMPARSER"
OR capitalized_lib STREQUAL "${t}ASMPRINTER"
OR capitalized_lib STREQUAL "${t}DISASSEMBLER"
OR capitalized_lib STREQUAL "${t}INFO"
OR capitalized_lib STREQUAL "${t}UTILS"
)
set(${return_var} ON PARENT_SCOPE)
break()
endif()
endforeach()
endfunction(is_llvm_target_library)
function(is_llvm_target_specifier library return_var)
is_llvm_target_library(${library} ${return_var} ${ARGN})
string(TOUPPER "${library}" capitalized_lib)
if (NOT ${return_var})
if (
capitalized_lib STREQUAL "ALLTARGETSASMPARSERS"
OR capitalized_lib STREQUAL "ALLTARGETSDESCS"
OR capitalized_lib STREQUAL "ALLTARGETSDISASSEMBLERS"
OR capitalized_lib STREQUAL "ALLTARGETSINFOS"
OR capitalized_lib STREQUAL "NATIVE"
OR capitalized_lib STREQUAL "NATIVECODEGEN"
)
set(${return_var} ON PARENT_SCOPE)
endif()
endif()
endfunction()
macro(llvm_config executable)
cmake_parse_arguments(ARG "USE_SHARED" "" "" ${ARGN})
set(link_components ${ARG_UNPARSED_ARGUMENTS})
if (ARG_USE_SHARED)
# If USE_SHARED is specified, then we link against libLLVM,
# but also against the component libraries below. This is
# done in case libLLVM does not contain all of the components
# the target requires.
#
# Strip LLVM_DYLIB_COMPONENTS out of link_components.
# To do this, we need special handling for "all", since that
# may imply linking to libraries that are not included in
# libLLVM.
if (DEFINED link_components AND DEFINED LLVM_DYLIB_COMPONENTS)
if ("${LLVM_DYLIB_COMPONENTS}" STREQUAL "all")
set(link_components "")
else()
list(REMOVE_ITEM link_components ${LLVM_DYLIB_COMPONENTS})
endif()
endif()
target_link_libraries(${executable} PRIVATE LLVM)
endif()
explicit_llvm_config(${executable} ${link_components})
endmacro(llvm_config)
function(explicit_llvm_config executable)
set(link_components ${ARGN})
llvm_map_components_to_libnames(LIBRARIES ${link_components})
get_target_property(t ${executable} TYPE)
if (t STREQUAL "STATIC_LIBRARY")
target_link_libraries(${executable} INTERFACE ${LIBRARIES})
elseif (
t STREQUAL "EXECUTABLE"
OR t STREQUAL "SHARED_LIBRARY"
OR t STREQUAL "MODULE_LIBRARY"
)
target_link_libraries(${executable} PRIVATE ${LIBRARIES})
else()
# Use plain form for legacy user.
target_link_libraries(${executable} ${LIBRARIES})
endif()
endfunction(explicit_llvm_config)
# This is Deprecated
function(llvm_map_components_to_libraries OUT_VAR)
message(
AUTHOR_WARNING
"Using llvm_map_components_to_libraries() is deprecated. Use llvm_map_components_to_libnames() instead"
)
explicit_map_components_to_libraries(result ${ARGN})
set(${OUT_VAR} ${result} ${sys_result} PARENT_SCOPE)
endfunction(llvm_map_components_to_libraries)
# Expand pseudo-components into real components.
# Does not cover 'native', 'backend', or 'engine' as these require special
# handling. Also does not cover 'all' as we only have a list of the libnames
# available and not a list of the components.
function(llvm_expand_pseudo_components out_components)
set(link_components ${ARGN})
foreach (c ${link_components})
# add codegen, asmprinter, asmparser, disassembler
list(FIND LLVM_TARGETS_TO_BUILD ${c} idx)
if (NOT idx LESS 0)
if (TARGET LLVM${c}CodeGen)
list(APPEND expanded_components "${c}CodeGen")
else()
if (TARGET LLVM${c})
list(APPEND expanded_components "${c}")
else()
message(FATAL_ERROR "Target ${c} is not in the set of libraries.")
endif()
endif()
if (TARGET LLVM${c}AsmPrinter)
list(APPEND expanded_components "${c}AsmPrinter")
endif()
if (TARGET LLVM${c}AsmParser)
list(APPEND expanded_components "${c}AsmParser")
endif()
if (TARGET LLVM${c}Desc)
list(APPEND expanded_components "${c}Desc")
endif()
if (TARGET LLVM${c}Disassembler)
list(APPEND expanded_components "${c}Disassembler")
endif()
if (TARGET LLVM${c}Info)
list(APPEND expanded_components "${c}Info")
endif()
if (TARGET LLVM${c}Utils)
list(APPEND expanded_components "${c}Utils")
endif()
elseif (c STREQUAL "nativecodegen")
if (TARGET LLVM${LLVM_NATIVE_ARCH}CodeGen)
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}CodeGen")
endif()
if (TARGET LLVM${LLVM_NATIVE_ARCH}Desc)
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}Desc")
endif()
if (TARGET LLVM${LLVM_NATIVE_ARCH}Info)
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}Info")
endif()
elseif (c STREQUAL "AllTargetsCodeGens")
# Link all the codegens from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}CodeGen)
list(APPEND expanded_components "${t}CodeGen")
endif()
endforeach(t)
elseif (c STREQUAL "AllTargetsAsmParsers")
# Link all the asm parsers from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}AsmParser)
list(APPEND expanded_components "${t}AsmParser")
endif()
endforeach(t)
elseif (c STREQUAL "AllTargetsDescs")
# Link all the descs from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}Desc)
list(APPEND expanded_components "${t}Desc")
endif()
endforeach(t)
elseif (c STREQUAL "AllTargetsDisassemblers")
# Link all the disassemblers from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}Disassembler)
list(APPEND expanded_components "${t}Disassembler")
endif()
endforeach(t)
elseif (c STREQUAL "AllTargetsInfos")
# Link all the infos from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}Info)
list(APPEND expanded_components "${t}Info")
endif()
endforeach(t)
else()
list(APPEND expanded_components "${c}")
endif()
endforeach()
set(${out_components} ${expanded_components} PARENT_SCOPE)
endfunction(llvm_expand_pseudo_components out_components)
# This is a variant intended for the final user:
# Map LINK_COMPONENTS to actual libnames.
function(llvm_map_components_to_libnames out_libs)
set(link_components ${ARGN})
if (NOT LLVM_AVAILABLE_LIBS)
# Inside LLVM itself available libs are in a global property.
get_property(LLVM_AVAILABLE_LIBS GLOBAL PROPERTY LLVM_LIBS)
endif()
string(TOUPPER "${LLVM_AVAILABLE_LIBS}" capitalized_libs)
get_property(LLVM_TARGETS_CONFIGURED GLOBAL PROPERTY LLVM_TARGETS_CONFIGURED)
# Generally in our build system we avoid order-dependence. Unfortunately since
# not all targets create the same set of libraries we actually need to ensure
# that all build targets associated with a target are added before we can
# process target dependencies.
if (NOT LLVM_TARGETS_CONFIGURED)
foreach (c ${link_components})
is_llvm_target_specifier(${c} iltl_result ALL_TARGETS)
if (iltl_result)
message(
FATAL_ERROR
"Specified target library before target registration is complete."
)
endif()
endforeach()
endif()
# Expand some keywords:
list(FIND LLVM_TARGETS_TO_BUILD "${LLVM_NATIVE_ARCH}" have_native_backend)
list(FIND link_components "engine" engine_required)
if (NOT engine_required EQUAL -1)
list(FIND LLVM_TARGETS_WITH_JIT "${LLVM_NATIVE_ARCH}" have_jit)
if (NOT have_native_backend EQUAL -1 AND NOT have_jit EQUAL -1)
list(APPEND link_components "jit")
list(APPEND link_components "native")
else()
list(APPEND link_components "interpreter")
endif()
endif()
list(FIND link_components "native" native_required)
if (NOT native_required EQUAL -1)
if (NOT have_native_backend EQUAL -1)
list(APPEND link_components ${LLVM_NATIVE_ARCH})
endif()
endif()
# Translate symbolic component names to real libraries:
llvm_expand_pseudo_components(link_components ${link_components})
foreach (c ${link_components})
get_property(c_rename GLOBAL PROPERTY LLVM_COMPONENT_NAME_${c})
if (c_rename)
set(c ${c_rename})
endif()
if (c STREQUAL "native")
# already processed
elseif (c STREQUAL "backend")
# same case as in `native'.
elseif (c STREQUAL "engine")
# already processed
elseif (c STREQUAL "all")
get_property(all_components GLOBAL PROPERTY LLVM_COMPONENT_LIBS)
list(APPEND expanded_components ${all_components})
else()
# Canonize the component name:
string(TOUPPER "${c}" capitalized)
list(FIND capitalized_libs LLVM${capitalized} lib_idx)
if (lib_idx LESS 0)
# The component is unknown. Maybe is an omitted target?
is_llvm_target_library(${c} iltl_result OMITTED_TARGETS)
if (iltl_result)
# A missing library to a directly referenced omitted target would be bad.
message(
FATAL_ERROR
"Library '${c}' is a direct reference to a target library for an omitted target."
)
else()
# If it is not an omitted target we should assume it is a component
# that hasn't yet been processed by CMake. Missing components will
# cause errors later in the configuration, so we can safely assume
# that this is valid here.
list(APPEND expanded_components LLVM${c})
endif()
else(lib_idx LESS 0)
list(GET LLVM_AVAILABLE_LIBS ${lib_idx} canonical_lib)
list(APPEND expanded_components ${canonical_lib})
endif(lib_idx LESS 0)
endif(c STREQUAL "native")
endforeach(c)
set(${out_libs} ${expanded_components} PARENT_SCOPE)
endfunction()
# Perform a post-order traversal of the dependency graph.
# This duplicates the algorithm used by llvm-config, originally
# in tools/llvm-config/llvm-config.cpp, function ComputeLibsForComponents.
function(expand_topologically name required_libs visited_libs)
list(FIND visited_libs ${name} found)
if (found LESS 0)
list(APPEND visited_libs ${name})
set(visited_libs ${visited_libs} PARENT_SCOPE)
#
get_property(libname GLOBAL PROPERTY LLVM_COMPONENT_NAME_${name})
if (libname)
set(cname LLVM${libname})
elseif (TARGET ${name})
set(cname ${name})
elseif (TARGET LLVM${name})
set(cname LLVM${name})
else()
message(FATAL_ERROR "unknown component ${name}")
endif()
get_property(lib_deps TARGET ${cname} PROPERTY LLVM_LINK_COMPONENTS)
foreach (lib_dep ${lib_deps})
expand_topologically(${lib_dep} "${required_libs}" "${visited_libs}")
set(required_libs ${required_libs} PARENT_SCOPE)
set(visited_libs ${visited_libs} PARENT_SCOPE)
endforeach()
list(APPEND required_libs ${cname})
set(required_libs ${required_libs} PARENT_SCOPE)
endif()
endfunction()
# Expand dependencies while topologically sorting the list of libraries:
function(llvm_expand_dependencies out_libs)
set(expanded_components ${ARGN})
set(required_libs)
set(visited_libs)
foreach (lib ${expanded_components})
expand_topologically(${lib} "${required_libs}" "${visited_libs}")
endforeach()
if (required_libs)
list(REVERSE required_libs)
endif()
set(${out_libs} ${required_libs} PARENT_SCOPE)
endfunction()
function(explicit_map_components_to_libraries out_libs)
llvm_map_components_to_libnames(link_libs ${ARGN})
llvm_expand_dependencies(expanded_components ${link_libs})
# Return just the libraries included in this build:
set(result)
foreach (c ${expanded_components})
if (TARGET ${c})
set(result ${result} ${c})
endif()
endforeach(c)
set(${out_libs} ${result} PARENT_SCOPE)
endfunction(explicit_map_components_to_libraries)

View File

@@ -0,0 +1,129 @@
include(AddFileDependencies)
include(CMakeParseArguments)
function(llvm_replace_compiler_option var old new)
# Replaces a compiler option or switch `old' in `var' by `new'.
# If `old' is not in `var', appends `new' to `var'.
# Example: llvm_replace_compiler_option(CMAKE_CXX_FLAGS_RELEASE "-O3" "-O2")
# If the option already is on the variable, don't add it:
if ("${${var}}" MATCHES "(^| )${new}($| )")
set(n "")
else()
set(n "${new}")
endif()
if ("${${var}}" MATCHES "(^| )${old}($| )")
string(REGEX REPLACE "(^| )${old}($| )" " ${n} " ${var} "${${var}}")
else()
set(${var} "${${var}} ${n}")
endif()
set(${var} "${${var}}" PARENT_SCOPE)
endfunction(llvm_replace_compiler_option)
macro(add_td_sources srcs)
file(GLOB tds *.td)
if (tds)
source_group("TableGen descriptions" FILES ${tds})
set_source_files_properties(${tds} PROPERTIES HEADER_FILE_ONLY ON)
list(APPEND ${srcs} ${tds})
endif()
endmacro(add_td_sources)
function(add_header_files_for_glob hdrs_out glob)
file(GLOB hds ${glob})
set(filtered)
foreach (file ${hds})
# Explicit existence check is necessary to filter dangling symlinks
# out. See https://bugs.gentoo.org/674662.
if (EXISTS ${file})
list(APPEND filtered ${file})
endif()
endforeach()
set(${hdrs_out} ${filtered} PARENT_SCOPE)
endfunction(add_header_files_for_glob)
function(find_all_header_files hdrs_out additional_headerdirs)
add_header_files_for_glob(hds *.h)
list(APPEND all_headers ${hds})
foreach (additional_dir ${additional_headerdirs})
add_header_files_for_glob(hds "${additional_dir}/*.h")
list(APPEND all_headers ${hds})
add_header_files_for_glob(hds "${additional_dir}/*.inc")
list(APPEND all_headers ${hds})
endforeach(additional_dir)
set(${hdrs_out} ${all_headers} PARENT_SCOPE)
endfunction(find_all_header_files)
function(llvm_process_sources OUT_VAR)
cmake_parse_arguments(
ARG
"PARTIAL_SOURCES_INTENDED"
""
"ADDITIONAL_HEADERS;ADDITIONAL_HEADER_DIRS"
${ARGN}
)
set(sources ${ARG_UNPARSED_ARGUMENTS})
if (NOT ARG_PARTIAL_SOURCES_INTENDED)
llvm_check_source_file_list(${sources})
endif()
# This adds .td and .h files to the Visual Studio solution:
add_td_sources(sources)
find_all_header_files(hdrs "${ARG_ADDITIONAL_HEADER_DIRS}")
if (hdrs)
set_source_files_properties(${hdrs} PROPERTIES HEADER_FILE_ONLY ON)
endif()
set_source_files_properties(
${ARG_ADDITIONAL_HEADERS}
PROPERTIES HEADER_FILE_ONLY ON
)
list(APPEND sources ${ARG_ADDITIONAL_HEADERS} ${hdrs})
set(${OUT_VAR} ${sources} PARENT_SCOPE)
endfunction(llvm_process_sources)
function(llvm_check_source_file_list)
cmake_parse_arguments(ARG "" "SOURCE_DIR" "" ${ARGN})
foreach (l ${ARG_UNPARSED_ARGUMENTS})
get_filename_component(fp ${l} REALPATH)
list(APPEND listed ${fp})
endforeach()
if (ARG_SOURCE_DIR)
file(GLOB globbed "${ARG_SOURCE_DIR}/*.c" "${ARG_SOURCE_DIR}/*.cpp")
else()
file(GLOB globbed *.c *.cpp)
endif()
foreach (g ${globbed})
get_filename_component(fn ${g} NAME)
if (ARG_SOURCE_DIR)
set(entry "${g}")
else()
set(entry "${fn}")
endif()
get_filename_component(gp ${g} REALPATH)
# Don't reject hidden files. Some editors create backups in the
# same directory as the file.
if (NOT "${fn}" MATCHES "^\\.")
list(FIND LLVM_OPTIONAL_SOURCES ${entry} idx)
if (idx LESS 0)
list(FIND listed ${gp} idx)
if (idx LESS 0)
if (ARG_SOURCE_DIR)
set(fn_relative "${ARG_SOURCE_DIR}/${fn}")
else()
set(fn_relative "${fn}")
endif()
message(
SEND_ERROR
"Found unknown source file ${fn_relative}
Please update ${CMAKE_CURRENT_LIST_FILE}\n"
)
endif()
endif()
endif()
endforeach()
endfunction(llvm_check_source_file_list)

View File

@@ -0,0 +1,53 @@
# This file defines the `libcudacxx_build_compiler_targets()` function, which
# creates the following interface targets:
#
# libcudacxx.compiler_interface
# - Interface target linked into all targets in the libcudacxx developer build.
# Defines common warning flags, definitions, etc, including those defined in
# the global CCCL targets.
cccl_get_libcudacxx()
function(libcudacxx_build_compiler_targets)
set(cuda_compile_options)
set(cxx_compile_options)
set(cxx_compile_definitions)
# if (CCCL_USE_LIBCXX)
# list(APPEND cxx_compile_options "-stdlib=libc++")
# list(APPEND cxx_compile_definitions "_ALLOW_UNSUPPORTED_LIBCPP=1")
# endif()
# Set test specific flags
list(APPEND cxx_compile_definitions "CCCL_ENABLE_ASSERTIONS")
list(APPEND cxx_compile_definitions "CCCL_IGNORE_DEPRECATED_CPP_DIALECT")
list(
APPEND cxx_compile_definitions
"CCCL_IGNORE_DEPRECATED_DISCARD_MEMORY_HEADER"
)
list(
APPEND cxx_compile_definitions
"CCCL_IGNORE_DEPRECATED_STREAM_REF_HEADER"
)
if (CCCL_ENABLE_TILE)
list(APPEND cuda_compile_options "--enable-tile")
endif()
cccl_build_compiler_interface(
libcudacxx.compiler_flags
"${cuda_compile_options}"
"${cxx_compile_options}"
"${cxx_compile_definitions}"
)
add_library(libcudacxx.compiler_interface INTERFACE)
target_link_libraries(
libcudacxx.compiler_interface
INTERFACE
# order matters here, we need the libcudacxx options to override the cccl options.
cccl.compiler_interface
libcudacxx.compiler_flags
libcudacxx::libcudacxx
)
endfunction()

View File

@@ -0,0 +1,125 @@
# For every public header, build a translation unit containing `#include <header>`
# to let the compiler try to figure out warnings in that header if it is not otherwise
# included in tests, and also to verify if the headers are modular enough.
# .inl files are not globbed for, because they are not supposed to be used as public
# entrypoints.
cccl_get_cudatoolkit()
# Meta target for all configs' header builds:
add_custom_target(libcudacxx.test.internal_headers)
# Grep all internal headers
file(
GLOB_RECURSE internal_headers
RELATIVE "${libcudacxx_SOURCE_DIR}/include/"
CONFIGURE_DEPENDS
${libcudacxx_SOURCE_DIR}/include/cuda/__*/*.h
${libcudacxx_SOURCE_DIR}/include/cuda/std/__*/*.h
)
# Exclude <cuda/std/__cccl/(prologue|epilogue|visibility).h> from the test
list(
FILTER internal_headers
EXCLUDE
REGEX "__cccl/(prologue|epilogue|visibility)\.h"
)
# headers in `__cuda` are meant to come after the related "cuda" headers so they do not compile on their own
list(FILTER internal_headers EXCLUDE REGEX "__cuda/*")
# generated cuda::ptx headers are not standalone
list(FILTER internal_headers EXCLUDE REGEX "__ptx/instructions/generated")
# don't check nvtx3.h - it's not our header
list(FILTER internal_headers EXCLUDE REGEX ".*/__nvtx/nvtx3.h")
function(libcudacxx_add_internal_header_test_target target_name)
if (NOT ARGN)
return()
endif()
cccl_generate_header_tests(
${target_name}
libcudacxx/include
NO_METATARGETS
LANGUAGE CUDA
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
HEADERS ${ARGN}
)
target_compile_definitions(${target_name} PRIVATE _CCCL_HEADER_TEST)
target_link_libraries(
${target_name}
PUBLIC #
libcudacxx.compiler_interface
CUDA::cudart
)
add_dependencies(libcudacxx.test.internal_headers ${target_name})
endfunction()
libcudacxx_add_internal_header_test_target(
libcudacxx.test.internal_headers.base
${internal_headers}
)
# We have fallbacks for some type traits that we want to explicitly test so that they do not bitrot.
set(internal_headers_fallback)
set(internal_headers_fallback_per_header_defines)
foreach (header IN LISTS internal_headers)
# MSVC cannot handle some of the fallbacks.
if ("MSVC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
if (
"${header}" MATCHES "is_base_of"
OR "${header}" MATCHES "is_nothrow_destructible"
OR "${header}" MATCHES "is_polymorphic"
)
continue()
endif()
endif()
file(READ "${libcudacxx_SOURCE_DIR}/include/${header}" header_file)
string(REGEX MATCH "_LIBCUDACXX_[A-Z_]*_FALLBACK" fallback "${header_file}")
if (fallback)
list(APPEND internal_headers_fallback "${header}")
string(
REGEX REPLACE
"([][+.*^$()|?\\\\])"
"\\\\\\1"
header_regex
"${header}"
)
list(
APPEND internal_headers_fallback_per_header_defines
DEFINE
"${fallback}"
"^${header_regex}$"
)
endif()
endforeach()
if (internal_headers_fallback)
cccl_generate_header_tests(
libcudacxx.test.internal_headers.fallback
libcudacxx/include
NO_METATARGETS
LANGUAGE CUDA
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
HEADERS ${internal_headers_fallback}
PER_HEADER_DEFINES ${internal_headers_fallback_per_header_defines}
)
target_compile_definitions(
libcudacxx.test.internal_headers.fallback
PRIVATE _CCCL_HEADER_TEST
)
target_link_libraries(
libcudacxx.test.internal_headers.fallback
PUBLIC #
libcudacxx.compiler_interface
CUDA::cudart
)
add_dependencies(
libcudacxx.test.internal_headers
libcudacxx.test.internal_headers.fallback
)
endif()

View File

@@ -0,0 +1,47 @@
# For every public header, build a translation unit containing `#include <header>`
# to let the compiler try to figure out warnings in that header if it is not otherwise
# included in tests, and also to verify if the headers are modular enough.
# .inl files are not globbed for, because they are not supposed to be used as public
# entrypoints.
# Meta target for all configs' header builds:
add_custom_target(libcudacxx.test.public_headers)
# Grep all public headers
file(
GLOB public_headers
LIST_DIRECTORIES false
RELATIVE "${libcudacxx_SOURCE_DIR}/include"
CONFIGURE_DEPENDS
"${libcudacxx_SOURCE_DIR}/include/cuda/*"
"${libcudacxx_SOURCE_DIR}/include/cuda/std/*"
)
# annotated_ptr does not work with clang cuda due to __nv_associate_access_property
if ("Clang" STREQUAL "${CMAKE_CUDA_COMPILER_ID}")
list(REMOVE_ITEM public_headers "annotated_ptr")
endif()
function(libcudacxx_add_public_header_test_target target_name)
if (NOT ARGN)
return()
endif()
cccl_generate_header_tests(
${target_name}
libcudacxx/include
NO_METATARGETS
LANGUAGE CUDA
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
HEADERS ${ARGN}
)
target_compile_definitions(${target_name} PRIVATE _CCCL_HEADER_TEST)
target_link_libraries(${target_name} PUBLIC libcudacxx.compiler_interface)
add_dependencies(libcudacxx.test.public_headers ${target_name})
endfunction()
libcudacxx_add_public_header_test_target(
libcudacxx.test.public_headers.base
${public_headers}
)

View File

@@ -0,0 +1,75 @@
# For every public header, build a translation unit containing `#include <header>`
# to let the compiler try to figure out warnings in that header if it is not otherwise
# included in tests, and also to verify if the headers are modular enough.
# .inl files are not globbed for, because they are not supposed to be used as public
# entrypoints.
cccl_get_cudatoolkit()
# Meta target for all configs' header builds:
add_custom_target(libcudacxx.test.public_headers_host_only)
add_custom_target(libcudacxx.test.public_headers_host_only_with_ctk)
if (CCCL_ENABLE_TILE) # TODO(miscco): For now only test public headers with tile
return()
endif()
# Grep all public headers
file(
GLOB public_headers_host_only
LIST_DIRECTORIES false
RELATIVE "${libcudacxx_SOURCE_DIR}/include"
CONFIGURE_DEPENDS
"${libcudacxx_SOURCE_DIR}/include/cuda/*"
"${libcudacxx_SOURCE_DIR}/include/cuda/std/*"
)
set(public_host_header_cxx_compile_options)
set(public_host_header_cxx_compile_definitions)
# Specifically add libc++ testing if requested to the libcudacxx host suite
if (CCCL_USE_LIBCXX)
list(APPEND public_host_header_cxx_compile_options "-stdlib=libc++")
endif()
function(
libcudacxx_add_public_header_test_host_target
target_name
parent_target
with_ctk
)
cccl_generate_header_tests(
${target_name}
libcudacxx/include
NO_METATARGETS
LANGUAGE CXX
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
HEADERS ${public_headers_host_only}
)
target_compile_definitions(
${target_name}
PRIVATE #
${public_host_header_cxx_compile_definitions}
_CCCL_HEADER_TEST
)
target_compile_options(
${target_name}
PRIVATE ${public_host_header_cxx_compile_options}
)
target_link_libraries(${target_name} PUBLIC libcudacxx.compiler_interface)
if (with_ctk)
target_link_libraries(${target_name} PUBLIC CUDA::cudart)
endif()
add_dependencies(${parent_target} ${target_name})
endfunction()
libcudacxx_add_public_header_test_host_target(
libcudacxx.test.public_headers_host_only.base
libcudacxx.test.public_headers_host_only
OFF
)
libcudacxx_add_public_header_test_host_target(
libcudacxx.test.public_headers_host_only_with_ctk.base
libcudacxx.test.public_headers_host_only_with_ctk
ON
)

1569
cccl_upstream/libcudacxx/cmake/config.guess vendored Executable file

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,23 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// ignore deprecation warnings
#if defined(__clang__)
# pragma clang diagnostic ignored "-Wdeprecated"
# pragma clang diagnostic ignored "-Wdeprecated-declarations"
#elif defined(_MSC_VER)
# pragma warning (disable: 4996)
#else
# pragma GCC diagnostic ignored "-Wdeprecated"
# pragma GCC diagnostic ignored "-Wdeprecated-declarations"
#endif
// This file tests that the respective header is includable on its own with a cuda compiler
#include <@header@>

View File

@@ -0,0 +1 @@
cccl_add_subdir_helper(libcudacxx)

View File

@@ -0,0 +1 @@
build

View File

@@ -0,0 +1,51 @@
## Codegen adds the following build targets
# libcudacxx.atomics.codegen
# libcudacxx.atomics.codegen.install
## Test targets:
# libcudacxx.test.atomics.codegen.diff
add_executable(codegen EXCLUDE_FROM_ALL codegen.cpp)
target_compile_features(codegen PRIVATE cxx_std_20)
set(
atomic_generated_output
"${libcudacxx_BINARY_DIR}/codegen/cuda_ptx_generated.h"
)
set(
atomic_install_location
"${libcudacxx_SOURCE_DIR}/include/cuda/std/__atomic/functions"
)
add_custom_target(
libcudacxx.atomics.codegen
COMMAND codegen "${atomic_generated_output}"
BYPRODUCTS "${atomic_generated_output}"
)
add_custom_target(
libcudacxx.atomics.codegen.install
# gersemi: off
COMMAND
"${CMAKE_COMMAND}" -E copy
"${atomic_generated_output}"
"${atomic_install_location}/cuda_ptx_generated.h"
# gersemi: on
DEPENDS libcudacxx.atomics.codegen
BYPRODUCTS "${atomic_install_location}/cuda_ptx_generated.h"
)
add_test(
NAME libcudacxx.test.atomics.codegen.diff
# gersemi: off
COMMAND
"${CMAKE_COMMAND}" -E compare_files
"${atomic_install_location}/cuda_ptx_generated.h"
"${atomic_generated_output}"
# gersemi: on
)
set_tests_properties(
libcudacxx.test.atomics.codegen.diff
PROPERTIES REQUIRED_FILES "${atomic_generated_output}"
)

View File

@@ -0,0 +1,164 @@
#!/usr/bin/env python3
##===----------------------------------------------------------------------===##
##
## Part of libcu++, the C++ Standard Library for your entire system,
## under the Apache License v2.0 with LLVM Exceptions.
## See https://llvm.org/LICENSE.txt for license information.
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
##
##===----------------------------------------------------------------------===##
import argparse
import os
import cccl_paths
docs = os.path.join(cccl_paths.DOCS_LIBCUDACXX_DIR, "ptx", "instructions")
test = os.path.join(cccl_paths.LIBCUDACXX_TEST_DIR, "libcudacxx", "cuda", "ptx")
src = os.path.join(cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "__ptx", "instructions")
ptx_header = os.path.join(cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "ptx")
instr_docs = os.path.join(cccl_paths.DOCS_LIBCUDACXX_DIR, "ptx", "instructions.rst")
def add_docs(ptx_instr, url):
cpp_instr = ptx_instr.replace(".", "_")
underbar = "=" * len(ptx_instr)
(docs / f"{cpp_instr}.rst").write_text(
f""".. _libcudacxx-ptx-instructions-{ptx_instr.replace(".", "-")}:
{ptx_instr}
{underbar}
- PTX ISA:
`{ptx_instr} <{url}>`__
.. include:: generated/{cpp_instr}.rst
"""
)
def add_test(ptx_instr):
cpp_instr = ptx_instr.replace(".", "_")
dst = test / f"ptx.{ptx_instr}.compile.pass.cpp"
dst.write_text(
f"""//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: libcpp-has-no-threads
// <cuda/ptx>
#include <cuda/ptx>
#include <cuda/std/utility>
#include "generated/{cpp_instr}.h"
int main(int, char**)
{{
return 0;
}}
"""
)
def add_src(ptx_instr):
cpp_instr = ptx_instr.replace(".", "_")
(src / f"{cpp_instr}.h").write_text(
f"""// -*- C++ -*-
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA_PTX_{cpp_instr.upper()}_H_
#define _CUDA_PTX_{cpp_instr.upper()}_H_
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__ptx/ptx_dot_variants.h>
#include <cuda/__ptx/ptx_helper_functions.h>
#include <cuda/std/cstdint>
#include <nv/target> // __CUDA_MINIMUM_ARCH__ and friends
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_PTX
#include <cuda/__ptx/instructions/generated/{cpp_instr}.h>
_CCCL_END_NAMESPACE_CUDA_PTX
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA_PTX_{cpp_instr.upper()}_H_
"""
)
def add_ptx_header_include(ptx_instr):
cpp_instr = ptx_instr.replace(".", "_")
txt = ptx_header.read_text()
# just add as first new include. clang-format will sort it in
idx = txt.index("#include <cuda/__ptx/instructions")
txt = (
txt[:idx]
+ f"""#include <cuda/__ptx/instructions/{cpp_instr}.h>\n"""
+ txt[idx:]
)
ptx_header.write_text(txt)
def add_docs_include(ptx_instr):
cpp_instr = ptx_instr.replace(".", "_")
txt = instr_docs.read_text()
# just add as first new include
idx = txt.index(" instructions/")
txt = txt[:idx] + f" instructions/{cpp_instr}\n" + txt[idx:]
instr_docs.write_text(txt)
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument("ptx_instruction", type=str)
parser.add_argument("url", type=str)
args = parser.parse_args()
ptx_instr = args.ptx_instruction
url = args.url
# Enable using internal urls in the command-line, to be automatically converted to public URLs.
if url.startswith("index.html"):
url = url.replace(
"index.html",
"https://docs.nvidia.com/cuda/parallel-thread-execution/index.html",
)
add_test(ptx_instr)
add_docs(ptx_instr, url)
add_src(ptx_instr)
add_ptx_header_include(ptx_instr)
add_docs_include(ptx_instr)

View File

@@ -0,0 +1,21 @@
##===----------------------------------------------------------------------===##
##
## Part of libcu++, the C++ Standard Library for your entire system,
## under the Apache License v2.0 with LLVM Exceptions.
## See https://llvm.org/LICENSE.txt for license information.
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
##
##===----------------------------------------------------------------------===##
import os
LIBCUDACXX_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
LIBCUDACXX_CMAKE_DIR = os.path.join(LIBCUDACXX_DIR, "cmake")
LIBCUDACXX_CODEGEN_DIR = os.path.join(LIBCUDACXX_DIR, "codegen")
LIBCUDACXX_INCLUDE_DIR = os.path.join(LIBCUDACXX_DIR, "include")
LIBCUDACXX_TEST_DIR = os.path.join(LIBCUDACXX_DIR, "test")
DOCS_DIR = os.path.dirname(LIBCUDACXX_DIR)
DOCS_LIBCUDACXX_DIR = os.path.join(DOCS_DIR, "libcudacxx")

View File

@@ -0,0 +1,45 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <fstream>
#include <iostream>
#include <ostream>
#include "generators/compare_and_swap.h"
#include "generators/exchange.h"
#include "generators/fence.h"
#include "generators/fetch_ops.h"
#include "generators/header.h"
#include "generators/ld_st.h"
using namespace std::string_literals;
int main(int argc, char** argv)
{
std::fstream filestream;
if (argc == 2)
{
filestream.open(argv[1], filestream.out);
}
std::ostream& stream = filestream.is_open() ? filestream : std::cout;
FormatHeader(stream);
FormatFence(stream);
FormatLoad(stream);
FormatStore(stream);
FormatCompareAndSwap(stream);
FormatExchange(stream);
FormatFetchOps(stream);
FormatTail(stream);
return 0;
}

View File

@@ -0,0 +1,245 @@
#!/usr/bin/env python3
##===----------------------------------------------------------------------===##
##
## Part of libcu++, the C++ Standard Library for your entire system,
## under the Apache License v2.0 with LLVM Exceptions.
## See https://llvm.org/LICENSE.txt for license information.
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
##
##===----------------------------------------------------------------------===##
import datetime
import os
import cccl_paths
PROLOGUE_FILE = os.path.join(
cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "std", "__cccl", "prologue.h"
)
EPILOGUE_FILE = os.path.join(
cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "std", "__cccl", "epilogue.h"
)
HEADER = f"""\
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) {datetime.datetime.now().year} NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// !!! DO NOT EDIT THIS FILE !!! This file is generated by utils/generate_prologue_epilogue.py.
// NO include guards here (this file is included multiple times)"""
FOOTER = """\
// NO include guards here (this file is included multiple times)
"""
PUSH_POP_MACROS = {
"__declspec modifiers": [
"align",
"allocate",
"allocator",
"appdomain",
"code_seg",
"deprecated",
"dllimport",
"dllexport",
"empty_bases",
"hybrid_patchable",
"jitintrinsic",
"lifetimebound",
"naked",
"noalias",
"noinline",
"noreturn",
"nothrow",
"novtable",
"no_sanitize_address",
"process",
"property",
"restrict",
"safebuffers",
"selectany",
"spectre",
"thread",
"uuid",
],
"[[msvc::attribute]] attributes": [
"msvc",
"flatten",
"forceinline",
"forceinline_calls",
"intrinsic",
"noinline",
"noinline_calls",
"no_tls_guard",
],
"Windows nasty macros": ["min", "max", "interface"],
"sal.h on Windows": ["__valid", "__callback"],
"other macros": ["clang"],
"sys/sysmacros.h on linux": ["major", "minor", "makedev"],
}
def write_section(file, section):
file.write(section)
file.write("\n\n")
def make_prologue(file):
# Write common header.
write_section(file, HEADER)
# Add prologue/epilogue include logic check.
write_section(
file,
"""\
#if defined(_CCCL_PROLOGUE_INCLUDED)
# error \\
"cccl internal error: <cuda/std/__cccl/epilogue.h> must be included before next <cuda/std/__cccl/prologue.h> is reincluded"
#endif
#define _CCCL_PROLOGUE_INCLUDED() 1""",
)
# Add necessary includes.
write_section(
file,
"""\
#include <cuda/std/__cccl/compiler.h>
#include <cuda/std/__cccl/diagnostic.h>
#include <cuda/std/__cccl/dialect.h>""",
)
# Add push macros.
for group_name, macros in PUSH_POP_MACROS.items():
write_section(file, f"// {group_name}")
for macro in macros:
write_section(
file,
f"""\
#if defined({macro})
# pragma push_macro("{macro}")
# undef {macro}
# define _CCCL_POP_MACRO_{macro}
#endif // defined({macro})""",
)
# Add warnings suppressions.
write_section(
file,
'''\
_CCCL_DIAG_PUSH
_CCCL_NV_DIAG_PUSH()
// disable some msvc warnings
// https://github.com/microsoft/STL/blob/master/stl/inc/yvals_core.h#L353
// warning C4100: 'quack': unreferenced formal parameter
// warning C4127: conditional expression is constant
// warning C4180: qualifier applied to function type has no meaning; ignored
// warning C4197: 'purr': top-level volatile in cast is ignored
// warning C4324: 'roar': structure was padded due to alignment specifier
// warning C4455: literal suffix identifiers that do not start with an underscore are reserved
// warning C4503: 'hum': decorated name length exceeded, name was truncated
// warning C4522: 'woof' : multiple assignment operators specified
// warning C4668: 'meow' is not defined as a preprocessor macro, replacing with '0' for '#if/#elif'
// warning C4800: 'boo': forcing value to bool 'true' or 'false' (performance warning)
// warning C4996: 'meow': was declared deprecated
_CCCL_DIAG_SUPPRESS_MSVC(4100 4127 4180 4197 4296 4324 4455 4503 4522 4668 4800 4996)
// Suppress compiler warnings about C++ extensions.
#if _CCCL_COMPILER(GCC, >=, 12)
_CCCL_DIAG_SUPPRESS_GCC("-Wc++20-extensions")
_CCCL_DIAG_SUPPRESS_GCC("-Wc++23-extensions")
#endif // _CCCL_COMPILER(GCC, >=, 12)
#if _CCCL_COMPILER(GCC, >=, 14)
_CCCL_DIAG_SUPPRESS_GCC("-Wc++26-extensions")
#endif // _CCCL_COMPILER(GCC, >=, 14)
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++20-extensions")
#if _CCCL_COMPILER(CLANG, >=, 17)
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++23-extensions")
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++26-extensions")
#else // ^^^ _CCCL_COMPILER(CLANG, >=, 17) ^^^ / vvv _CCCL_COMPILER(CLANG, <, 17) vvv
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++2b-extensions")
#endif // ^^^ _CCCL_COMPILER(CLANG, <, 17) ^^^
// Suppress `if consteval`-related warnings.
_CCCL_DIAG_SUPPRESS_NVHPC(if_consteval_nonstandard)
_CCCL_DIAG_SUPPRESS_NVHPC(is_constant_evaluated_in_nonconstexpr_context)
_CCCL_DIAG_SUPPRESS_NVHPC(if_consteval_in_nonconstexpr_function)
_CCCL_DIAG_SUPPRESS_NVCC(3215) // "if consteval" and "if not consteval" are not standard in this mode
_CCCL_DIAG_SUPPRESS_NVCC(3206) // "if consteval" and "if not consteval" are meaningless in a non-constexpr function
_CCCL_DIAG_SUPPRESS_NVCC(3060) // call to __builtin_is_constant_evaluated appearing in a non-constexpr function always
// produces "false"''',
)
# Write the common footer.
file.write(FOOTER)
def make_epilogue(file):
# Write common header.
write_section(file, HEADER)
# Write includes.
write_section(
file,
"""\
#include <cuda/std/__cccl/compiler.h>
#include <cuda/std/__cccl/diagnostic.h>""",
)
# Add prologue/epilogue include logic check.
write_section(
file,
"""\
#if !defined(_CCCL_PROLOGUE_INCLUDED)
# error "cccl internal error: <cuda/std/__cccl/prologue.h> must be included before <cuda/std/__cccl/epilogue.h>"
#endif
#undef _CCCL_PROLOGUE_INCLUDED""",
)
# Pop warning suppressions.
write_section(
file,
"""\
_CCCL_NV_DIAG_POP()
_CCCL_DIAG_POP""",
)
# Add pop macros.
for group_name, macros in PUSH_POP_MACROS.items():
write_section(file, f"// {group_name}")
for macro in macros:
write_section(
file,
f"""\
#if defined({macro})
# error \\
"cccl internal error: macro `{macro}` was redefined between <cuda/std/__cccl/prologue.h> and <cuda/std/__cccl/epilogue.h>"
#elif defined(_CCCL_POP_MACRO_{macro})
# pragma pop_macro("{macro}")
# undef _CCCL_POP_MACRO_{macro}
#endif""",
)
# Write the common footer.
file.write(FOOTER)
if __name__ == "__main__":
with open(PROLOGUE_FILE, "w") as file:
make_prologue(file)
with open(EPILOGUE_FILE, "w") as file:
make_epilogue(file)

View File

@@ -0,0 +1,201 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef COMPARED_AND_SWAP_H
#define COMPARED_AND_SWAP_H
#include <format>
#include <string>
#include "definitions.h"
inline void FormatCompareAndSwap(std::ostream& out)
{
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE bool __cuda_atomic_compare_swap_memory_order_dispatch(_Fn& __cuda_cas, int __success_memorder, int __failure_memorder, _Sco) {
bool __res = false;
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__stronger_order_cuda(__success_memorder, __failure_memorder)) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __res = __cuda_cas(__atomic_cuda_acquire{}); break;
case __ATOMIC_ACQ_REL: __res = __cuda_cas(__atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELEASE: __res = __cuda_cas(__atomic_cuda_release{}); break;
case __ATOMIC_RELAXED: __res = __cuda_cas(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__stronger_order_cuda(__success_memorder, __failure_memorder)) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __res = __cuda_cas(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __res = __cuda_cas(__atomic_cuda_volatile{}); break;
case __ATOMIC_RELAXED: __res = __cuda_cas(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
return __res;
}
)XXX";
// Argument ID Reference
// 0 - Operand Type
// 1 - Operand Size
// 2 - Type Constraint
// 3 - Memory Order
// 4 - Memory Order function tag
// 5 - Scope Constraint
// 6 - Scope function tag
constexpr auto asm_intrinsic_format_128 = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE bool __cuda_atomic_compare_exchange(
_Type* __ptr, _Type& __dst, _Type __cmp, _Type __op, {4}, __atomic_cuda_operand_{0}{1}, {6})
{{
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b CAS is not supported until PTX ISA version 840");
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_90, (),
NV_ANY_TARGET, (__atomic_cas_128b_unsupported_before_SM_90();)
)
asm volatile(R"YYY(
{{
.reg .b128 _d;
.reg .b128 _v;
mov.b128 _d, {{%3, %4}};
mov.b128 _v, {{%5, %6}};
atom.cas{3}{5}.b128 _d,[%2],_d,_v;
mov.b128 {{%0, %1}}, _d;
}}
)YYY" : "=l"(__dst.__x),"=l"(__dst.__y) : "l"(__ptr), "l"(__cmp.__x),"l"(__cmp.__y), "l"(__op.__x),"l"(__op.__y) : "memory"); return __dst.__x == __cmp.__x && __dst.__y == __cmp.__y; }})XXX";
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE bool __cuda_atomic_compare_exchange(
_Type* __ptr, _Type& __dst, _Type __cmp, _Type __op, {4}, __atomic_cuda_operand_{0}{1}, {6})
{{ asm volatile("atom.cas{3}{5}.{0}{1} %0,[%1],%2,%3;" : "={2}"(__dst) : "l"(__ptr), "{2}"(__cmp), "{2}"(__op) : "memory"); return __dst == __cmp; }})XXX";
constexpr Operand supported_types[] = {
Operand::Bit,
};
constexpr size_t supported_sizes[] = {
32,
64,
128,
};
constexpr Semantic supported_semantics[] = {
Semantic::Acquire,
Semantic::Relaxed,
Semantic::Release,
Semantic::Acq_Rel,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
for (auto size : supported_sizes)
{
for (auto type : supported_types)
{
for (auto sem : supported_semantics)
{
for (auto sco : supported_scopes)
{
if (size == 2 && type != Operand::Bit)
{
continue;
}
if (size == 128 && type != Operand::Bit)
{
continue;
}
if (size == 128)
{
out << std::format(
asm_intrinsic_format_128,
operand(type),
size,
constraints(type, size),
semantic(sem),
semantic_tag(sem),
scope(sco),
scope_tag(sco));
}
else
{
out << std::format(
asm_intrinsic_format,
operand(type),
size,
constraints(type, size),
semantic(sem),
semantic_tag(sem),
scope(sco),
scope_tag(sco));
}
}
}
}
}
out << "\n"
<< R"XXX(
template <typename _Type, typename _Tag, typename _Sco>
struct __cuda_atomic_bind_compare_exchange {
_Type* __ptr;
_Type* __exp;
_Type* __des;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE bool operator()(_Atomic_Memorder) {
return __cuda_atomic_compare_exchange(__ptr, *__exp, *__exp, *__des, _Atomic_Memorder{}, _Tag{}, _Sco{});
}
};
template <class _Type, class _Sco>
static inline _CCCL_DEVICE bool __atomic_compare_exchange_cuda(_Type* __ptr, _Type* __exp, _Type __des, bool, int __success_memorder, int __failure_memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
__proxy_t* __exp_proxy = reinterpret_cast<__proxy_t*>(__exp);
__proxy_t* __des_proxy = reinterpret_cast<__proxy_t*>(&__des);
bool __res = false;
if (__cuda_compare_exchange_weak_if_local(__ptr_proxy, __exp_proxy, __des_proxy, &__res)) {return __res;}
__cuda_atomic_bind_compare_exchange<__proxy_t, __proxy_tag, _Sco> __bound_compare_swap{__ptr_proxy, __exp_proxy, __des_proxy};
return __cuda_atomic_compare_swap_memory_order_dispatch(__bound_compare_swap, __success_memorder, __failure_memorder, _Sco{});
}
template <class _Type, class _Sco>
static inline _CCCL_DEVICE bool __atomic_compare_exchange_cuda(_Type volatile* __ptr, _Type* __exp, _Type __des, bool, int __success_memorder, int __failure_memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __exp_proxy = reinterpret_cast<__proxy_t*>(__exp);
__proxy_t* __des_proxy = reinterpret_cast<__proxy_t*>(&__des);
bool __res = false;
if (__cuda_compare_exchange_weak_if_local(__ptr_proxy, __exp_proxy, __des_proxy, &__res)) {return __res;}
__cuda_atomic_bind_compare_exchange<__proxy_t, __proxy_tag, _Sco> __bound_compare_swap{__ptr_proxy, __exp_proxy, __des_proxy};
return __cuda_atomic_compare_swap_memory_order_dispatch(__bound_compare_swap, __success_memorder, __failure_memorder, _Sco{});
}
)XXX";
}
#endif // COMPARED_AND_SWAP_H

View File

@@ -0,0 +1,192 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef DEFINITIONS_H
#define DEFINITIONS_H
#include <format>
#include <map>
#include <string>
#include <type_traits>
#include <vector>
enum class Mmio
{
Disabled,
Enabled,
};
inline std::string mmio(Mmio m)
{
static const char* mmio_map[]{
"",
".mmio",
};
return mmio_map[std::underlying_type_t<Mmio>(m)];
}
inline std::string mmio_tag(Mmio m)
{
static const char* mmio_map[]{
"__atomic_cuda_mmio_disable",
"__atomic_cuda_mmio_enable",
};
return mmio_map[std::underlying_type_t<Mmio>(m)];
}
enum class Operand
{
Floating,
Unsigned,
Signed,
Bit,
};
inline std::string operand(Operand op)
{
static std::map op_map = {
std::pair{Operand::Floating, "f"},
std::pair{Operand::Unsigned, "u"},
std::pair{Operand::Signed, "s"},
std::pair{Operand::Bit, "b"},
};
return op_map[op];
}
inline std::string operand_proxy_type(Operand op, size_t sz)
{
if (op == Operand::Floating)
{
if (sz == 32)
{
return {"float"};
}
else
{
return {"double"};
}
}
else if (op == Operand::Signed)
{
return std::format("int{}_t", sz);
}
// Binary and unsigned can be the same proxy_type
return std::format("uint{}_t", sz);
}
inline std::string constraints(Operand op, size_t sz)
{
static std::map constraint_map = {
std::pair{32,
std::map{
std::pair{Operand::Bit, "r"},
std::pair{Operand::Unsigned, "r"},
std::pair{Operand::Signed, "r"},
std::pair{Operand::Floating, "f"},
}},
std::pair{64,
std::map{
std::pair{Operand::Bit, "l"},
std::pair{Operand::Unsigned, "l"},
std::pair{Operand::Signed, "l"},
std::pair{Operand::Floating, "d"},
}},
std::pair{128,
std::map{
std::pair{Operand::Bit, "l"},
std::pair{Operand::Unsigned, "l"},
std::pair{Operand::Signed, "l"},
std::pair{Operand::Floating, "d"},
}},
};
if (sz == 16)
{
return {"h"};
}
else
{
return constraint_map[sz][op];
}
}
enum class Semantic
{
Relaxed,
Release,
Acquire,
Acq_Rel,
Seq_Cst,
Volatile,
};
inline std::string semantic(Semantic sem)
{
static std::map sem_map = {
std::pair{Semantic::Relaxed, ".relaxed"},
std::pair{Semantic::Release, ".release"},
std::pair{Semantic::Acquire, ".acquire"},
std::pair{Semantic::Acq_Rel, ".acq_rel"},
std::pair{Semantic::Seq_Cst, ".sc"},
std::pair{Semantic::Volatile, ""},
};
return sem_map[sem];
}
inline std::string semantic_tag(Semantic sem)
{
static std::map sem_map = {
std::pair{Semantic::Relaxed, "__atomic_cuda_relaxed"},
std::pair{Semantic::Release, "__atomic_cuda_release"},
std::pair{Semantic::Acquire, "__atomic_cuda_acquire"},
std::pair{Semantic::Acq_Rel, "__atomic_cuda_acq_rel"},
std::pair{Semantic::Seq_Cst, "__atomic_cuda_seq_cst"},
std::pair{Semantic::Volatile, "__atomic_cuda_volatile"},
};
return sem_map[sem];
}
enum class Scope
{
Thread,
Warp,
CTA,
Cluster,
GPU,
System,
};
inline std::string scope(Scope sco)
{
static std::map sco_map = {
std::pair{Scope::Thread, ""},
std::pair{Scope::Warp, ""},
std::pair{Scope::CTA, ".cta"},
std::pair{Scope::Cluster, ".cluster"},
std::pair{Scope::GPU, ".gpu"},
std::pair{Scope::System, ".sys"},
};
return sco_map[sco];
}
inline std::string scope_tag(Scope sco)
{
static std::map sco_map = {
std::pair{Scope::Thread, "__thread_scope_thread_tag"},
std::pair{Scope::Warp, ""},
std::pair{Scope::CTA, "__thread_scope_block_tag"},
std::pair{Scope::Cluster, "__thread_scope_cluster_tag"},
std::pair{Scope::GPU, "__thread_scope_device_tag"},
std::pair{Scope::System, "__thread_scope_system_tag"},
};
return sco_map[sco];
}
#endif // DEFINITIONS_H

View File

@@ -0,0 +1,197 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef EXCHANGE_H
#define EXCHANGE_H
#include <format>
#include <string>
#include "definitions.h"
inline void FormatExchange(std::ostream& out)
{
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE void __cuda_atomic_exchange_memory_order_dispatch(_Fn& __cuda_exch, int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_exch(__atomic_cuda_acquire{}); break;
case __ATOMIC_ACQ_REL: __cuda_exch(__atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELEASE: __cuda_exch(__atomic_cuda_release{}); break;
case __ATOMIC_RELAXED: __cuda_exch(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_exch(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __cuda_exch(__atomic_cuda_volatile{}); break;
case __ATOMIC_RELAXED: __cuda_exch(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
// Argument ID Reference
// 0 - Operand Type
// 1 - Operand Size
// 2 - Type Constraint
// 3 - Memory Order
// 4 - Memory Order function tag
// 5 - Scope Constraint
// 6 - Scope function tag
constexpr auto asm_intrinsic_format_128 = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_exchange(
_Type* __ptr, _Type& __old, _Type __new, {4}, __atomic_cuda_operand_{0}{1}, {6})
{{
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b exchange is not supported until PTX ISA version 840");
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_90, (),
NV_ANY_TARGET, (__atomic_exchange_128b_unsupported_before_SM_90();)
)
asm volatile(R"YYY(
{{
.reg .b128 _d;
.reg .b128 _v;
mov.b128 _v, {{%3, %4}};
atom.exch{3}{5}.b128 _d,[%2],_v;
mov.b128 {{%0, %1}}, _d;
}}
)YYY" : "=l"(__old.__x),"=l"(__old.__y) : "l"(__ptr), "l"(__new.__x),"l"(__new.__y) : "memory");
}})XXX";
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_exchange(
_Type* __ptr, _Type& __old, _Type __new, {4}, __atomic_cuda_operand_{0}{1}, {6})
{{ asm volatile("atom.exch{3}{5}.{0}{1} %0,[%1],%2;" : "={2}"(__old) : "l"(__ptr), "{2}"(__new) : "memory"); }})XXX";
constexpr Operand supported_types[] = {
Operand::Bit,
};
constexpr size_t supported_sizes[] = {
32,
64,
128,
};
constexpr Semantic supported_semantics[] = {
Semantic::Acquire,
Semantic::Relaxed,
Semantic::Release,
Semantic::Acq_Rel,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
for (auto size : supported_sizes)
{
for (auto type : supported_types)
{
for (auto sem : supported_semantics)
{
for (auto sco : supported_scopes)
{
if (size == 2 && type != Operand::Bit)
{
continue;
}
if (size == 128 && type != Operand::Bit)
{
continue;
}
if (size == 128)
{
out << std::format(
asm_intrinsic_format_128,
operand(type),
size,
constraints(type, size),
semantic(sem),
semantic_tag(sem),
scope(sco),
scope_tag(sco));
}
else
{
out << std::format(
asm_intrinsic_format,
operand(type),
size,
constraints(type, size),
semantic(sem),
semantic_tag(sem),
scope(sco),
scope_tag(sco));
}
}
}
}
}
out << "\n"
<< R"XXX(
template <typename _Type, typename _Tag, typename _Sco>
struct __cuda_atomic_bind_exchange {
_Type* __ptr;
_Type* __old;
_Type* __new;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
__cuda_atomic_exchange(__ptr, *__old, *__new, _Atomic_Memorder{}, _Tag{}, _Sco{});
}
};
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_exchange_cuda(_Type* __ptr, _Type& __old, _Type __new, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
__proxy_t* __old_proxy = reinterpret_cast<__proxy_t*>(&__old);
__proxy_t* __new_proxy = reinterpret_cast<__proxy_t*>(&__new);
if(__cuda_exchange_weak_if_local(__ptr_proxy, __new_proxy, __old_proxy)) {{return;}}
__cuda_atomic_bind_exchange<__proxy_t, __proxy_tag, _Sco> __bound_swap{__ptr_proxy, __old_proxy, __new_proxy};
__cuda_atomic_exchange_memory_order_dispatch(__bound_swap, __memorder, _Sco{});
}
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_exchange_cuda(_Type volatile* __ptr, _Type& __old, _Type __new, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __old_proxy = reinterpret_cast<__proxy_t*>(&__old);
__proxy_t* __new_proxy = reinterpret_cast<__proxy_t*>(&__new);
if(__cuda_exchange_weak_if_local(__ptr_proxy, __new_proxy, __old_proxy)) {{return;}}
__cuda_atomic_bind_exchange<__proxy_t, __proxy_tag, _Sco> __bound_swap{__ptr_proxy, __old_proxy, __new_proxy};
__cuda_atomic_exchange_memory_order_dispatch(__bound_swap, __memorder, _Sco{});
}
)XXX";
}
#endif // EXCHANGE_H

View File

@@ -0,0 +1,110 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef FENCE_H
#define FENCE_H
#include <format>
#include <string>
#include "definitions.h"
inline std::string membar_scope(Scope sco)
{
static std::map scope_map{
std::pair{Scope::GPU, ".gl"},
std::pair{Scope::System, ".sys"},
std::pair{Scope::CTA, ".cta"},
};
return scope_map[sco];
}
inline void FormatFence(std::ostream& out)
{
// Argument ID Reference
// 0 - Membar scope tag
// 1 - Membar scope
constexpr auto intrinsic_membar = R"XXX(
static inline _CCCL_DEVICE void __cuda_atomic_membar({0})
{{ asm volatile("membar{1};" ::: "memory"); }})XXX";
const std::map membar_scopes{
std::pair{Scope::GPU, ".gl"},
std::pair{Scope::System, ".sys"},
std::pair{Scope::CTA, ".cta"},
};
for (const auto& sco : membar_scopes)
{
out << std::format(intrinsic_membar, scope_tag(sco.first), sco.second);
}
// Argument ID Reference
// 0 - Fence scope tag
// 1 - Fence scope
// 2 - Fence order tag
// 3 - Fence order
constexpr auto intrinsic_fence = R"XXX(
static inline _CCCL_DEVICE void __cuda_atomic_fence({0}, {2})
{{ asm volatile("fence{1}{3};" ::: "memory"); }})XXX";
const Scope fence_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
const Semantic fence_semantics[] = {
Semantic::Acq_Rel,
Semantic::Seq_Cst,
};
for (const auto& sco : fence_scopes)
{
for (const auto& sem : fence_semantics)
{
out << std::format(intrinsic_fence, scope_tag(sco), semantic(sem), semantic_tag(sem), scope(sco));
}
}
out << "\n"
<< R"XXX(
template <typename _Sco>
static inline _CCCL_DEVICE void __atomic_thread_fence_cuda(int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); break;
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: [[fallthrough]];
case __ATOMIC_ACQ_REL: [[fallthrough]];
case __ATOMIC_RELEASE: __cuda_atomic_fence(_Sco{}, __atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELAXED: break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: [[fallthrough]];
case __ATOMIC_ACQ_REL: [[fallthrough]];
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELAXED: break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
}
#endif // FENCE_H

View File

@@ -0,0 +1,219 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef FETCH_OPS_H
#define FETCH_OPS_H
#include <array>
#include <format>
#include <string>
#include "definitions.h"
inline std::string fetch_op_skip_v(std::string fetch_op)
{
if (fetch_op == "add")
{
return "constexpr auto __skip_v = __atomic_ptr_skip_t<_Type>::__skip;";
}
return "constexpr auto __skip_v = 1;";
}
inline void FormatFetchOps(std::ostream& out)
{
const std::vector arithmetic_types = {
Operand::Floating,
Operand::Unsigned,
Operand::Signed,
};
const std::vector minmax_types = {
Operand::Unsigned,
Operand::Signed,
};
const std::vector bitwise_types = {Operand::Bit};
const std::map op_support_map{
std::pair{std::string{"add"}, std::pair{arithmetic_types, std::string{"arithmetic"}}},
std::pair{std::string{"min"}, std::pair{minmax_types, std::string{"minmax"}}},
std::pair{std::string{"max"}, std::pair{minmax_types, std::string{"minmax"}}},
std::pair{std::string{"or"}, std::pair{bitwise_types, std::string{"bitwise"}}},
std::pair{std::string{"xor"}, std::pair{bitwise_types, std::string{"bitwise"}}},
std::pair{std::string{"and"}, std::pair{bitwise_types, std::string{"bitwise"}}},
};
// Memory order dispatcher
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE void __cuda_atomic_fetch_memory_order_dispatch(_Fn& __cuda_fetch, int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_fetch(__atomic_cuda_acquire{}); break;
case __ATOMIC_ACQ_REL: __cuda_fetch(__atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELEASE: __cuda_fetch(__atomic_cuda_release{}); break;
case __ATOMIC_RELAXED: __cuda_fetch(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_fetch(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __cuda_fetch(__atomic_cuda_volatile{}); break;
case __ATOMIC_RELAXED: __cuda_fetch(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
// Argument ID Reference
// 0 - Atomic Operation
// 1 - Operand Type
// 2 - Operand Size
// 3 - Type Constraint
// 4 - Memory Order
// 5 - Memory Order function tag
// 6 - Scope Constraint
// 7 - Scope function tag
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_fetch_{0}(
_Type* __ptr, _Type& __dst, _Type __op, {5}, __atomic_cuda_operand_{1}{2}, {7})
{{ asm volatile("atom.{0}{4}{6}.{1}{2} %0,[%1],%2;" : "={3}"(__dst) : "l"(__ptr), "{3}"(__op) : "memory"); }})XXX";
// 0 - Atomic Operation
// 1 - Operand type constraint
// 2 - Pointer op skip_v
constexpr auto fetch_bind_invoke = R"XXX(
template <typename _Type, typename _Tag, typename _Sco>
struct __cuda_atomic_bind_fetch_{0} {{
_Type* __ptr;
_Type* __dst;
_Type* __op;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {{
__cuda_atomic_fetch_{0}(__ptr, *__dst, *__op, _Atomic_Memorder{{}}, _Tag{{}}, _Sco{{}});
}}
}};
template <class _Type, class _Up, class _Sco, __atomic_enable_if_native_{1}<_Type> = 0>
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_{0}_cuda(_Type* __ptr, _Up __op, int __memorder, _Sco)
{{
{2}
__op = __op * __skip_v;
using __proxy_t = typename __atomic_cuda_deduce_{1}<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_{1}<_Type>::__tag;
_Type __dst{{}};
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
__proxy_t* __op_proxy = reinterpret_cast<__proxy_t*>(&__op);
if (__cuda_fetch_{0}_weak_if_local(__ptr_proxy, *__op_proxy, __dst_proxy)) {{return __dst;}}
__cuda_atomic_bind_fetch_{0}<__proxy_t, __proxy_tag, _Sco> __bound_{0}{{__ptr_proxy, __dst_proxy, __op_proxy}};
__cuda_atomic_fetch_memory_order_dispatch(__bound_{0}, __memorder, _Sco{{}});
return __dst;
}}
template <class _Type, class _Up, class _Sco, __atomic_enable_if_native_{1}<_Type> = 0>
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_{0}_cuda(_Type volatile* __ptr, _Up __op, int __memorder, _Sco)
{{
{2}
__op = __op * __skip_v;
using __proxy_t = typename __atomic_cuda_deduce_{1}<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_{1}<_Type>::__tag;
_Type __dst{{}};
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
__proxy_t* __op_proxy = reinterpret_cast<__proxy_t*>(&__op);
if (__cuda_fetch_{0}_weak_if_local(__ptr_proxy, *__op_proxy, __dst_proxy)) {{return __dst;}}
__cuda_atomic_bind_fetch_{0}<__proxy_t, __proxy_tag, _Sco> __bound_{0}{{__ptr_proxy, __dst_proxy, __op_proxy}};
__cuda_atomic_fetch_memory_order_dispatch(__bound_{0}, __memorder, _Sco{{}});
return __dst;
}}
)XXX";
constexpr size_t supported_sizes[] = {
32,
64,
};
constexpr Semantic supported_semantics[] = {
Semantic::Acquire,
Semantic::Relaxed,
Semantic::Release,
Semantic::Acq_Rel,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
for (auto& op_kp : op_support_map)
{
const auto& op_name = op_kp.first;
const auto& op_type_kp = op_kp.second;
const auto& type_list = op_type_kp.first;
const auto& deduction = op_type_kp.second;
for (auto type : type_list)
{
for (auto size : supported_sizes)
{
const std::string proxy_type = operand_proxy_type(type, size);
for (auto sco : supported_scopes)
{
for (auto sem : supported_semantics)
{
// There is no atom.add.s64
if (op_name == "add" && type == Operand::Signed && size == 64)
{
continue;
}
out << std::format(
asm_intrinsic_format,
/* 0 */ op_name,
/* 1 */ operand(type),
/* 2 */ size,
/* 3 */ constraints(type, size),
/* 4 */ semantic(sem),
/* 5 */ semantic_tag(sem),
/* 6 */ scope(sco),
/* 7 */ scope_tag(sco));
}
}
}
}
out << "\n" << std::format(fetch_bind_invoke, op_name, deduction, fetch_op_skip_v(op_name));
}
out << R"XXX(
template <class _Type, class _Up, class _Sco>
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_sub_cuda(_Type* __ptr, _Up __op, int __memorder, _Sco)
{
return __atomic_fetch_add_cuda(__ptr, -__op, __memorder, _Sco{});
}
template <class _Type, class _Up, class _Sco>
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_sub_cuda(_Type volatile* __ptr, _Up __op, int __memorder, _Sco)
{
return __atomic_fetch_add_cuda(__ptr, -__op, __memorder, _Sco{});
}
)XXX";
}
#endif // FETCH_OPS_H

View File

@@ -0,0 +1,88 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef HEADER_H
#define HEADER_H
#include <string>
inline void FormatHeader(std::ostream& out)
{
constexpr auto header = R"XXX(//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// This is an autogenerated file, we want to ensure that it contains exactly the contents we want to generate
// clang-format off
#ifndef _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
#define _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/__type_traits/enable_if.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/is_unsigned.h>
#include <cuda/std/__atomic/scopes.h>
#include <cuda/std/__atomic/order.h>
#include <cuda/std/__atomic/functions/common.h>
#include <cuda/std/__atomic/functions/cuda_ptx_generated_helper.h>
#include <cuda/std/__atomic/functions/cuda_local.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_STD
#if _CCCL_CUDA_COMPILATION()
extern "C" _CCCL_DEVICE void __atomic_cas_128b_unsupported_before_SM_90();
extern "C" _CCCL_DEVICE void __atomic_exchange_128b_unsupported_before_SM_90();
extern "C" _CCCL_DEVICE void __atomic_ldst_128b_unsupported_before_SM_70();
)XXX";
out << header;
}
inline void FormatTail(std::ostream& out)
{
constexpr auto tail = R"XXX(
#endif // _CCCL_CUDA_COMPILATION()
_CCCL_END_NAMESPACE_CUDA_STD
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
// clang-format on
)XXX";
out << tail;
}
#endif // HEADER_H

View File

@@ -0,0 +1,407 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef LD_ST_H
#define LD_ST_H
#include <format>
#include <string>
#include "definitions.h"
inline std::string semantic_ld_st(Semantic sem)
{
static std::map sem_map = {
std::pair{Semantic::Relaxed, ".relaxed"},
std::pair{Semantic::Release, ".release"},
std::pair{Semantic::Acquire, ".acquire"},
std::pair{Semantic::Volatile, ".volatile"},
};
return sem_map[sem];
}
inline std::string scope_ld_st(Semantic sem, Scope sco)
{
if (sem == Semantic::Volatile)
{
return "";
}
return scope(sco);
}
inline void FormatLoad(std::ostream& out)
{
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE void __cuda_atomic_load_memory_order_dispatch(_Fn &__cuda_load, int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_load(__atomic_cuda_acquire{}); break;
case __ATOMIC_RELAXED: __cuda_load(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_load(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELAXED: __cuda_load(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
// Argument ID Reference
// 0 - Operand Type
// 1 - Operand Size
// 2 - Constraint
// 3 - Memory order
// 4 - Memory order semantic
// 5 - Scope tag
// 6 - Scope semantic
// 7 - Mmio tag
// 8 - Mmio semantic
constexpr auto asm_intrinsic_format_128 = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_load(
const _Type* __ptr, _Type& __dst, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
{{
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b ld/st is not supported until PTX ISA version 840");
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (),
NV_ANY_TARGET, (__atomic_ldst_128b_unsupported_before_SM_70();)
)
asm volatile(R"YYY(
{{
.reg .b128 _d;
ld{8}{4}{6}.b128 _d,[%2];
mov.b128 {{%0, %1}}, _d;
}}
)YYY" : "=l"(__dst.__x),"=l"(__dst.__y) : "l"(__ptr) : "memory");
}})XXX";
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_load(
const _Type* __ptr, _Type& __dst, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
{{ asm volatile("ld{8}{4}{6}.{0}{1} %0,[%1];" : "={2}"(__dst) : "l"(__ptr) : "memory"); }})XXX";
constexpr size_t supported_sizes[] = {
16,
32,
64,
128,
};
constexpr Operand supported_types[] = {
Operand::Bit,
Operand::Floating,
Operand::Unsigned,
Operand::Signed,
};
constexpr Semantic supported_semantics[] = {
Semantic::Acquire,
Semantic::Relaxed,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
constexpr Mmio mmio_states[] = {
Mmio::Disabled,
Mmio::Enabled,
};
for (auto size : supported_sizes)
{
for (auto type : supported_types)
{
for (auto sem : supported_semantics)
{
for (auto sco : supported_scopes)
{
for (auto mm : mmio_states)
{
if (size == 16 && type == Operand::Floating)
{
continue;
}
if (size == 128 && type != Operand::Bit)
{
continue;
}
if ((mm == Mmio::Enabled) && ((sco != Scope::System) || (sem != Semantic::Relaxed)))
{
continue;
}
if (size == 128)
{
out << std::format(
asm_intrinsic_format_128,
/* 0 */ operand(type),
/* 1 */ size,
/* 2 */ constraints(type, size),
/* 3 */ semantic_tag(sem),
/* 4 */ semantic_ld_st(sem),
/* 5 */ scope_tag(sco),
/* 6 */ scope_ld_st(sem, sco),
/* 7 */ mmio_tag(mm),
/* 8 */ mmio(mm));
}
else
{
out << std::format(
asm_intrinsic_format,
/* 0 */ operand(type),
/* 1 */ size,
/* 2 */ constraints(type, size),
/* 3 */ semantic_tag(sem),
/* 4 */ semantic_ld_st(sem),
/* 5 */ scope_tag(sco),
/* 6 */ scope_ld_st(sem, sco),
/* 7 */ mmio_tag(mm),
/* 8 */ mmio(mm));
}
}
}
}
}
}
out << "\n"
<< R"XXX(
template <typename _Type, typename _Tag, typename _Sco, typename _Mmio>
struct __cuda_atomic_bind_load {
const _Type* __ptr;
_Type* __dst;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
__cuda_atomic_load(__ptr, *__dst, _Atomic_Memorder{}, _Tag{}, _Sco{}, _Mmio{});
}
};
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_load_cuda(const _Type* __ptr, _Type& __dst, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
const __proxy_t* __ptr_proxy = reinterpret_cast<const __proxy_t*>(__ptr);
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
if (__cuda_load_weak_if_local(__ptr_proxy, __dst_proxy, sizeof(__proxy_t))) {{return;}}
__cuda_atomic_bind_load<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_load{__ptr_proxy, __dst_proxy};
__cuda_atomic_load_memory_order_dispatch(__bound_load, __memorder, _Sco{});
}
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_load_cuda(const _Type volatile* __ptr, _Type& __dst, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
const __proxy_t* __ptr_proxy = reinterpret_cast<const __proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
if (__cuda_load_weak_if_local(__ptr_proxy, __dst_proxy, sizeof(__proxy_t))) {{return;}}
__cuda_atomic_bind_load<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_load{__ptr_proxy, __dst_proxy};
__cuda_atomic_load_memory_order_dispatch(__bound_load, __memorder, _Sco{});
}
)XXX";
}
inline void FormatStore(std::ostream& out)
{
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE void __cuda_atomic_store_memory_order_dispatch(_Fn &__cuda_store, int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_RELEASE: __cuda_store(__atomic_cuda_release{}); break;
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_RELAXED: __cuda_store(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_RELEASE: [[fallthrough]];
case __ATOMIC_SEQ_CST: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_RELAXED: __cuda_store(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
// Argument ID Reference
// 0 - Operand Type
// 1 - Operand Size
// 2 - Constraint
// 3 - Memory order
// 4 - Memory order semantic
// 5 - Scope tag
// 6 - Scope semantic
// 7 - Mmio tag
// 8 - Mmio semantic
constexpr auto asm_intrinsic_format_128 = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_store(
_Type* __ptr, _Type& __val, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
{{
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b ld/st is not supported until PTX ISA version 840");
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (),
NV_ANY_TARGET, (__atomic_ldst_128b_unsupported_before_SM_70();)
)
asm volatile(R"YYY(
{{
.reg .b128 _v;
mov.b128 _v, {{%1, %2}};
st{8}{4}{6}.b128 [%0],_v;
}}
)YYY" :: "l"(__ptr), "l"(__val.__x),"l"(__val.__y) : "memory");
}})XXX";
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_store(
_Type* __ptr, _Type& __val, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
{{ asm volatile("st{8}{4}{6}.{0}{1} [%0],%1;" :: "l"(__ptr), "{2}"(__val) : "memory"); }})XXX";
constexpr size_t supported_sizes[] = {
16,
32,
64,
128,
};
constexpr Operand supported_types[] = {
Operand::Bit,
};
constexpr Semantic supported_semantics[] = {
Semantic::Release,
Semantic::Relaxed,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
constexpr Mmio mmio_states[] = {
Mmio::Disabled,
Mmio::Enabled,
};
for (auto size : supported_sizes)
{
for (auto type : supported_types)
{
for (auto sem : supported_semantics)
{
for (auto sco : supported_scopes)
{
for (auto mm : mmio_states)
{
if (size == 16 && type == Operand::Floating)
{
continue;
}
if (size == 128 && type != Operand::Bit)
{
continue;
}
if ((mm == Mmio::Enabled) && ((sco != Scope::System) || (sem != Semantic::Relaxed)))
{
continue;
}
if (size == 128)
{
out << std::format(
asm_intrinsic_format_128,
/* 0 */ operand(type),
/* 1 */ size,
/* 2 */ constraints(type, size),
/* 3 */ semantic_tag(sem),
/* 4 */ semantic_ld_st(sem),
/* 5 */ scope_tag(sco),
/* 6 */ scope_ld_st(sem, sco),
/* 7 */ mmio_tag(mm),
/* 8 */ mmio(mm));
}
else
{
out << std::format(
asm_intrinsic_format,
/* 0 */ operand(type),
/* 1 */ size,
/* 2 */ constraints(type, size),
/* 3 */ semantic_tag(sem),
/* 4 */ semantic_ld_st(sem),
/* 5 */ scope_tag(sco),
/* 6 */ scope_ld_st(sem, sco),
/* 7 */ mmio_tag(mm),
/* 8 */ mmio(mm));
}
}
}
}
}
}
out << "\n"
<< R"XXX(
template <typename _Type, typename _Tag, typename _Sco, typename _Mmio>
struct __cuda_atomic_bind_store {
_Type* __ptr;
_Type* __val;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
__cuda_atomic_store(__ptr, *__val, _Atomic_Memorder{}, _Tag{}, _Sco{}, _Mmio{});
}
};
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_store_cuda(_Type* __ptr, _Type& __val, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
__proxy_t* __val_proxy = reinterpret_cast<__proxy_t*>(&__val);
if (__cuda_store_weak_if_local(__ptr_proxy, __val_proxy, sizeof(__proxy_t))) {{return;}}
__cuda_atomic_bind_store<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_store{__ptr_proxy, __val_proxy};
__cuda_atomic_store_memory_order_dispatch(__bound_store, __memorder, _Sco{});
}
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_store_cuda(volatile _Type* __ptr, _Type& __val, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __val_proxy = reinterpret_cast<__proxy_t*>(&__val);
if (__cuda_store_weak_if_local(__ptr_proxy, __val_proxy, sizeof(__proxy_t))) {{return;}}
__cuda_atomic_bind_store<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_store{__ptr_proxy, __val_proxy};
__cuda_atomic_store_memory_order_dispatch(__bound_store, __memorder, _Sco{});
}
)XXX";
}
#endif // LD_ST_H

View File

@@ -0,0 +1,68 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef __CUDA___ALGORITHM_COMMON
#define __CUDA___ALGORITHM_COMMON
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__concepts/concept_macros.h>
#include <cuda/std/__concepts/convertible_to.h>
#include <cuda/std/__ranges/concepts.h>
#include <cuda/std/__type_traits/remove_reference.h>
#include <cuda/std/mdspan>
#include <cuda/std/span>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <typename _Tp>
using __as_span_t = ::cuda::std::span<::cuda::std::remove_reference_t<::cuda::std::ranges::range_reference_t<_Tp>>>;
//! @brief A concept that checks if the type can be converted to a `cuda::std::span`.
//! The type must be a contiguous range.
template <typename _Tp>
_CCCL_CONCEPT __spannable = _CCCL_REQUIRES_EXPR((_Tp))( //
requires(::cuda::std::ranges::contiguous_range<_Tp>), //
requires(::cuda::std::convertible_to<_Tp, __as_span_t<_Tp>>));
template <typename _Tp>
using __as_mdspan_t =
::cuda::std::mdspan<typename ::cuda::std::decay_t<_Tp>::value_type,
typename ::cuda::std::decay_t<_Tp>::extents_type,
typename ::cuda::std::decay_t<_Tp>::layout_type,
typename ::cuda::std::decay_t<_Tp>::accessor_type>;
//! @brief A concept that checks if the type can be converted to a `cuda::std::mdspan`.
//! The type must have a conversion to `__as_mdspan_t<_Tp>`.
template <typename _Tp>
_CCCL_CONCEPT __mdspannable =
_CCCL_REQUIRES_EXPR((_Tp))(requires(::cuda::std::convertible_to<_Tp, __as_mdspan_t<_Tp>>));
template <typename _Tp>
[[nodiscard]] _CCCL_HOST_API constexpr auto __as_mdspan(_Tp&& __value) noexcept -> __as_mdspan_t<_Tp>
{
return ::cuda::std::forward<_Tp>(__value);
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif //__CUDA___ALGORITHM_COMMON

View File

@@ -0,0 +1,187 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef __CUDA___ALGORITHM_COPY_H
#define __CUDA___ALGORITHM_COPY_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
# include <cuda/__algorithm/common.h>
# include <cuda/__stream/launch_transform.h>
# include <cuda/__stream/stream_ref.h>
# include <cuda/__type_traits/is_trivially_copyable.h>
# include <cuda/std/__concepts/concept_macros.h>
# include <cuda/std/__exception/exception_macros.h>
# include <cuda/std/__host_stdlib/stdexcept>
# include <cuda/std/mdspan>
# include <cuda/std/span>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Source access order for copy_bytes
enum class source_access_order
{
# if _CCCL_CTK_AT_LEAST(13, 0)
//! @brief Access source in stream order
stream = ::cudaMemcpySrcAccessOrderStream,
//! @brief Access source during the copy call, source can be destroyed after the API returns
during_api_call = ::cudaMemcpySrcAccessOrderDuringApiCall,
//! @brief Access source in any order, the order can change across CUDA releases
any = ::cudaMemcpySrcAccessOrderAny,
# else
any = 0x3,
# endif // _CCCL_CTK_BELOW(13, 0)
};
//! @brief Configuration for copy_bytes
struct copy_configuration
{
//! @brief Source memory location hint for copy_bytes, used only for managed memory
memory_location src_location_hint = {};
//! @brief Destination memory location hint for copy_bytes, used only for managed memory
memory_location dst_location_hint = {};
//! @brief Source access order for copy_bytes
source_access_order src_access_order = source_access_order::any;
};
namespace __detail
{
template <typename _SrcTy, typename _DstTy>
_CCCL_HOST_API void __copy_bytes_impl(
stream_ref __stream,
::cuda::std::span<_SrcTy> __src,
::cuda::std::span<_DstTy> __dst,
[[maybe_unused]] copy_configuration __config)
{
static_assert(!::cuda::std::is_const_v<_DstTy>, "Copy destination can't be const");
static_assert(::cuda::is_trivially_copyable_v<_SrcTy> && ::cuda::is_trivially_copyable_v<_DstTy>);
if (__src.size_bytes() > __dst.size_bytes())
{
_CCCL_THROW(::std::invalid_argument, "Copy destination is too small to fit the source data");
}
if (__src.size_bytes() == 0)
{
return;
}
# if _CCCL_CTK_AT_LEAST(13, 0)
CUmemcpyAttributes __attributes = {};
__attributes.srcAccessOrder = static_cast<::CUmemcpySrcAccessOrder>(__config.src_access_order);
__attributes.srcLocHint.id = __config.src_location_hint.id;
__attributes.srcLocHint.type = static_cast<::CUmemLocationType>(__config.src_location_hint.type);
__attributes.dstLocHint.id = __config.dst_location_hint.id;
__attributes.dstLocHint.type = static_cast<::CUmemLocationType>(__config.dst_location_hint.type);
::cuda::__ensure_current_context guard(__stream);
::cuda::__driver::__memcpyAsyncWithAttributes(
__dst.data(), __src.data(), __src.size_bytes(), __stream.get(), __attributes);
# else
::cuda::__driver::__memcpyAsync(__dst.data(), __src.data(), __src.size_bytes(), __stream.get());
# endif // _CCCL_CTK_BELOW(13, 0)
}
template <typename _SrcElem,
typename _SrcExtents,
typename _SrcLayout,
typename _SrcAccessor,
typename _DstElem,
typename _DstExtents,
typename _DstLayout,
typename _DstAccessor>
_CCCL_HOST_API void __copy_bytes_impl(
stream_ref __stream,
::cuda::std::mdspan<_SrcElem, _SrcExtents, _SrcLayout, _SrcAccessor> __src,
::cuda::std::mdspan<_DstElem, _DstExtents, _DstLayout, _DstAccessor> __dst,
copy_configuration __config)
{
static_assert(::cuda::std::is_constructible_v<_DstExtents, _SrcExtents>,
"Multidimensional copy requires both source and destination extents to be compatible");
static_assert(::cuda::std::is_same_v<_SrcLayout, _DstLayout>,
"Multidimensional copy requires both source and destination layouts to match");
// Check only destination, because the layout of destination is the same as source
if (!__dst.is_exhaustive())
{
_CCCL_THROW(::std::invalid_argument, "copy_bytes supports only exhaustive mdspans");
}
if (__src.extents() != __dst.extents())
{
_CCCL_THROW(::std::invalid_argument, "Copy destination size differs from the source");
}
::cuda::__detail::__copy_bytes_impl(
__stream,
::cuda::std::span(__src.data_handle(), __src.mapping().required_span_size()),
::cuda::std::span(__dst.data_handle(), __dst.mapping().required_span_size()),
__config);
}
} // namespace __detail
//! @brief Launches a bytewise memory copy from source to destination into the provided
//! stream.
//!
//! Both source and destination needs to be a `contiguous_range` and convert to
//! `cuda::std::span`. The element types of both the source and destination range is
//! required to be trivially copyable.
//!
//! This call might be synchronous if either source or destination is pagable host memory.
//! It will be synchronous if both destination and copy is located in host memory.
//!
//! @param __stream Stream that the copy should be inserted into
//! @param __src Source to copy from
//! @param __dst Destination to copy into
//! @param __config Configuration for the copy
_CCCL_TEMPLATE(typename _SrcTy, typename _DstTy)
_CCCL_REQUIRES( // NOLINT(modernize-type-traits)
__spannable<transformed_device_argument_t<_SrcTy>> _CCCL_AND __spannable<transformed_device_argument_t<_DstTy>>)
_CCCL_HOST_API void copy_bytes(stream_ref __stream, _SrcTy&& __src, _DstTy&& __dst, copy_configuration __config = {})
{
::cuda::__detail::__copy_bytes_impl(
__stream,
::cuda::std::span(launch_transform(__stream, ::cuda::std::forward<_SrcTy>(__src))),
::cuda::std::span(launch_transform(__stream, ::cuda::std::forward<_DstTy>(__dst))),
__config);
}
//! @overload
//! @note This overload accepts mdspan-compatible types.
_CCCL_TEMPLATE(typename _SrcTy, typename _DstTy)
_CCCL_REQUIRES( // NOLINT(modernize-type-traits)
__mdspannable<transformed_device_argument_t<_SrcTy>> _CCCL_AND __mdspannable<transformed_device_argument_t<_DstTy>>)
_CCCL_HOST_API void copy_bytes(stream_ref __stream, _SrcTy&& __src, _DstTy&& __dst, copy_configuration __config = {})
{
::cuda::__detail::__copy_bytes_impl(
__stream,
::cuda::__as_mdspan(launch_transform(__stream, ::cuda::std::forward<_SrcTy>(__src))),
::cuda::__as_mdspan(launch_transform(__stream, ::cuda::std::forward<_DstTy>(__dst))),
__config);
}
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
#endif // __CUDA___ALGORITHM_COPY_H

View File

@@ -0,0 +1,101 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef __CUDA___ALGORITHM_FILL
#define __CUDA___ALGORITHM_FILL
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
# include <cuda/__algorithm/common.h>
# include <cuda/__stream/launch_transform.h>
# include <cuda/__stream/stream_ref.h>
# include <cuda/__type_traits/is_trivially_copyable.h>
# include <cuda/std/__concepts/concept_macros.h>
# include <cuda/std/__exception/exception_macros.h>
# include <cuda/std/__host_stdlib/stdexcept>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
namespace __detail
{
template <typename _DstTy, ::cuda::std::size_t _DstSize>
_CCCL_HOST_API void
__fill_bytes_impl(stream_ref __stream, ::cuda::std::span<_DstTy, _DstSize> __dst, ::cuda::std::uint8_t __value)
{
static_assert(!::cuda::std::is_const_v<_DstTy>, "Fill destination can't be const");
static_assert(::cuda::is_trivially_copyable_v<_DstTy>);
// TODO do a host callback if not device accessible?
::cuda::__driver::__memsetAsync(__dst.data(), __value, __dst.size_bytes(), __stream.get());
}
template <typename _DstElem, typename _DstExtents, typename _DstLayout, typename _DstAccessor>
_CCCL_HOST_API void __fill_bytes_impl(stream_ref __stream,
::cuda::std::mdspan<_DstElem, _DstExtents, _DstLayout, _DstAccessor> __dst,
::cuda::std::uint8_t __value)
{
// Check if the mdspan is exhaustive
if (!__dst.is_exhaustive())
{
_CCCL_THROW(::std::invalid_argument, "fill_bytes supports only exhaustive mdspans");
}
::cuda::__detail::__fill_bytes_impl(
__stream, ::cuda::std::span(__dst.data_handle(), __dst.mapping().required_span_size()), __value);
}
} // namespace __detail
//! @brief Launches an operation to bytewise fill the memory into the provided stream.
//!
//! The destination needs to be or launch_transform to a `contiguous_range` and convert to `cuda::std::span`.
//! The element type of the destination is required to be trivially copyable.
//!
//! The destination cannot reside in pagable host memory.
//!
//! @param __stream Stream that the copy should be inserted into
//! @param __dst Destination memory to fill
//! @param __value Value to fill into every byte in the destination
_CCCL_TEMPLATE(typename _DstTy)
_CCCL_REQUIRES(__spannable<transformed_device_argument_t<_DstTy>>)
_CCCL_HOST_API void fill_bytes(stream_ref __stream, _DstTy&& __dst, ::cuda::std::uint8_t __value)
{
::cuda::__detail::__fill_bytes_impl(
__stream, ::cuda::std::span(launch_transform(__stream, ::cuda::std::forward<_DstTy>(__dst))), __value);
}
//! @overload
//! @note This overload accepts mdspan-compatible types.
_CCCL_TEMPLATE(typename _DstTy)
_CCCL_REQUIRES(__mdspannable<transformed_device_argument_t<_DstTy>>)
_CCCL_HOST_API void fill_bytes(stream_ref __stream, _DstTy&& __dst, ::cuda::std::uint8_t __value)
{
::cuda::__detail::__fill_bytes_impl(
__stream, __as_mdspan(launch_transform(__stream, ::cuda::std::forward<_DstTy>(__dst))), __value);
}
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
#endif // __CUDA___ALGORITHM_FILL

View File

@@ -0,0 +1,170 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_H
#define _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__annotated_ptr/access_property_encoding.h>
#include <cuda/std/cstddef>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <typename>
class __annotated_ptr_base; // forward declaration
class access_property
{
private:
uint64_t __descriptor = __l2_interleave_normal;
friend class __annotated_ptr_base<access_property>;
// needed by __annotated_ptr_base
_CCCL_HOST_DEVICE_API constexpr access_property(uint64_t __descriptor1) noexcept
: __descriptor{__descriptor1}
{}
public:
struct shared
{};
struct global
{};
struct persisting
{
#if _CCCL_HAS_CTK()
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr operator ::cudaAccessProperty() const noexcept
{
return ::cudaAccessProperty::cudaAccessPropertyPersisting;
}
#endif // _CCCL_HAS_CTK()
};
struct streaming
{
#if _CCCL_HAS_CTK()
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr operator ::cudaAccessProperty() const noexcept
{
return ::cudaAccessProperty::cudaAccessPropertyStreaming;
}
#endif // _CCCL_HAS_CTK()
};
struct normal
{
#if _CCCL_HAS_CTK()
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr operator ::cudaAccessProperty() const noexcept
{
return ::cudaAccessProperty::cudaAccessPropertyNormal;
}
#endif // _CCCL_HAS_CTK()
};
_CCCL_HIDE_FROM_ABI access_property() noexcept = default;
_CCCL_HOST_DEVICE_API constexpr access_property(normal, float __fraction) noexcept
: __descriptor{
::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_Unchanged, __fraction)}
{}
_CCCL_HOST_DEVICE_API constexpr access_property(streaming, float __fraction) noexcept
: __descriptor{
::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_Unchanged, __fraction)}
{}
_CCCL_HOST_DEVICE_API constexpr access_property(persisting, float __fraction) noexcept
: __descriptor{
::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_Unchanged, __fraction)}
{}
_CCCL_HOST_DEVICE_API constexpr access_property(normal, float __fraction, streaming) noexcept
: __descriptor{
::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_First, __fraction)}
{}
_CCCL_HOST_DEVICE_API constexpr access_property(persisting, float __fraction, streaming) noexcept
: __descriptor{::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_First, __fraction)}
{}
_CCCL_HOST_DEVICE_API constexpr access_property(global) noexcept {}
_CCCL_HOST_DEVICE_API constexpr access_property(normal) noexcept
: access_property{normal{}, 1.0f}
{}
_CCCL_HOST_DEVICE_API constexpr access_property(streaming) noexcept
: access_property{streaming{}, 1.0f}
{}
_CCCL_HOST_DEVICE_API constexpr access_property(persisting) noexcept
: access_property{persisting{}, 1.0f}
{}
_CCCL_HOST_DEVICE_API inline access_property(
void* __ptr, size_t __primary_bytes, size_t __total_bytes, normal) noexcept
: __descriptor{::cuda::__block_encoding(
__l2_evict_t::_L2_Evict_Normal_Demote,
__l2_evict_t::_L2_Evict_Unchanged,
__ptr,
__primary_bytes,
__total_bytes)}
{}
_CCCL_HOST_DEVICE_API inline access_property(
void* __ptr, size_t __primary_bytes, size_t __total_bytes, streaming) noexcept
: __descriptor{::cuda::__block_encoding(
__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_Unchanged, __ptr, __primary_bytes, __total_bytes)}
{}
_CCCL_HOST_DEVICE_API inline access_property(
void* __ptr, size_t __primary_bytes, size_t __total_bytes, persisting) noexcept
: __descriptor{::cuda::__block_encoding(
__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_Unchanged, __ptr, __primary_bytes, __total_bytes)}
{}
_CCCL_HOST_DEVICE_API inline access_property(
void* __ptr, size_t __primary_bytes, size_t __total_bytes, global, streaming) noexcept
: __descriptor{::cuda::__block_encoding(
__l2_evict_t::_L2_Evict_Unchanged, __l2_evict_t::_L2_Evict_First, __ptr, __primary_bytes, __total_bytes)}
{}
_CCCL_HOST_DEVICE_API inline access_property(
void* __ptr, size_t __primary_bytes, size_t __total_bytes, normal, streaming) noexcept
: __descriptor{::cuda::__block_encoding(
__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_First, __ptr, __primary_bytes, __total_bytes)}
{}
_CCCL_HOST_DEVICE_API inline access_property(
void* __ptr, size_t __primary_bytes, size_t __total_bytes, streaming, streaming) noexcept
: __descriptor{::cuda::__block_encoding(
__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_First, __ptr, __primary_bytes, __total_bytes)}
{}
_CCCL_HOST_DEVICE_API inline access_property(
void* __ptr, size_t __primary_bytes, size_t __total_bytes, persisting, streaming) noexcept
: __descriptor{::cuda::__block_encoding(
__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_First, __ptr, __primary_bytes, __total_bytes)}
{}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr explicit operator uint64_t() const noexcept
{
return __descriptor;
}
};
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_H

View File

@@ -0,0 +1,171 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_ENCODING_H
#define _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_ENCODING_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__annotated_ptr/createpolicy.h>
#include <cuda/__cmath/ilog.h>
#include <cuda/std/__algorithm/clamp.h>
#include <cuda/std/__algorithm/max.h>
#include <cuda/std/__bit/bit_cast.h>
#include <cuda/std/__numeric/saturating_sub.h>
#include <cuda/std/__utility/to_underlying.h>
#include <cuda/std/cstddef>
#include <cuda/std/cstdint>
#include <cuda/std/limits>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
enum class __l2_descriptor_mode_t : uint32_t
{
_Desc_Implicit = 0,
_Desc_Interleaved = 2,
_Desc_Block_Type = 3
};
/***********************************************************************************************************************
* Range Block Descriptor
**********************************************************************************************************************/
// MemoryDescriptor:blockDesc_t reference
//
// struct __block_desc_t // 64 bits
// {
// uint64_t __reserved1 : 37;
// uint32_t __block_count : 7;
// uint32_t __block_start : 7;
// uint32_t __reserved2 : 1;
// uint32_t __block_size_enum : 4; // 56 bits
//
// uint32_t __l2_cop_off : 1;
// uint32_t __l2_cop_on : 2;
// uint32_t __l2_descriptor_mode : 2;
// uint32_t __l1_inv_dont_allocate : 1;
// uint32_t __l2_sector_promote_256B : 1;
// uint32_t __reserved3 : 1;
// };
#if !_CCCL_CUDA_COMPILER(NVRTC)
[[nodiscard]] _CCCL_HOST_API inline uint64_t __block_encoding_host(
__l2_evict_t __primary, __l2_evict_t __secondary, const void* __ptr, uint32_t __primary_bytes, uint32_t __total_bytes)
{
_CCCL_ASSERT(__primary_bytes > 0, "primary_size must be greater than 0");
_CCCL_ASSERT(__primary_bytes <= __total_bytes, "primary_size must be less than or equal to total_size");
_CCCL_ASSERT(__secondary == __l2_evict_t::_L2_Evict_First || __secondary == __l2_evict_t::_L2_Evict_Unchanged,
"secondary policy must be evict_first or evict_unchanged");
auto __raw_ptr = ::cuda::std::bit_cast<uintptr_t>(__ptr);
auto __log2_total_size = ::cuda::ceil_ilog2(__total_bytes);
auto __block_size_enum = ::cuda::std::saturating_sub<uint32_t>(__log2_total_size, 19); // min block size = 4K
auto __log2_block_size = 12u + __block_size_enum;
auto __block_size = 1u << __log2_block_size;
auto __block_start = static_cast<uint32_t>(__raw_ptr >> __log2_block_size); // ptr / block_size
// vvvv block_end = ceil_div(ptr + primary_size, block_size)
auto __block_end = static_cast<uint32_t>((__raw_ptr + __primary_bytes + __block_size - 1) >> __log2_block_size);
_CCCL_ASSERT(__block_end >= __block_start, "block_end < block_start");
// NOTE: there is a bug in PTX createpolicy when __block_size_enum == 13. The *incorrect* behavior matches the
// following code:
// auto __block_count = (__block_size_enum == 13)
// ? ((__block_end - __block_start <= 127u) ? (__block_end - __block_start) : 1)
// : ::cuda::std::clamp(__block_end - __block_start, 1u, 127u);
auto __block_count = ::cuda::std::clamp(__block_end - __block_start, 1u, 127u);
auto __l2_cop_off = ::cuda::std::to_underlying(__secondary);
auto __l2_cop_on = ::cuda::std::to_underlying(__primary);
auto __l2_descriptor_mode = ::cuda::std::to_underlying(__l2_descriptor_mode_t::_Desc_Block_Type);
return static_cast<uint64_t>(__block_count) << 37 //
| static_cast<uint64_t>(__block_start) << 44 //
| static_cast<uint64_t>(__block_size_enum) << 52 //
| static_cast<uint64_t>(__l2_cop_off) << 56 //
| static_cast<uint64_t>(__l2_cop_on) << 57 //
| static_cast<uint64_t>(__l2_descriptor_mode) << 59;
}
#endif // !_CCCL_CUDA_COMPILER(NVRTC)
[[nodiscard]] _CCCL_HOST_DEVICE_API inline uint64_t __block_encoding(
__l2_evict_t __primary, __l2_evict_t __secondary, const void* __ptr, size_t __primary_bytes, size_t __total_bytes)
{
_CCCL_ASSERT(__primary_bytes <= size_t{0xFFFFFFFF}, "primary size must be less than 4GB");
_CCCL_ASSERT(__total_bytes <= size_t{0xFFFFFFFF}, "total size must be less than 4GB");
auto __primary_bytes1 = static_cast<uint32_t>(__primary_bytes);
auto __total_bytes1 = static_cast<uint32_t>(__total_bytes);
NV_IF_ELSE_TARGET(
NV_IS_HOST,
(return ::cuda::__block_encoding_host(__primary, __secondary, __ptr, __primary_bytes1, __total_bytes1);),
(return ::cuda::__createpolicy_range(__primary, __secondary, __ptr, __primary_bytes1, __total_bytes1);))
}
/***********************************************************************************************************************
* Interleaved Descriptor
**********************************************************************************************************************/
// MemoryDescriptor:interleaveDesc_t reference
//
// struct __interleaved_desc_t // 64 bits
// {
// uint64_t : 52;
// uint32_t __fraction : 4; // 56 bits
//
// uint32_t __l2_cop_off : 1;
// uint32_t __l2_cop_on : 2;
// uint32_t __l2_descriptor_mode : 2;
// uint32_t __l1_inv_dont_allocate : 1;
// uint32_t __l2_sector_promote_256B : 1;
// uint32_t : 1;
// };
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr uint64_t
__l2_interleave(__l2_evict_t __primary, __l2_evict_t __secondary, float __fraction)
{
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
{
NV_IF_ELSE_TARGET(
NV_PROVIDES_SM_80, (return ::cuda::__createpolicy_fraction(__primary, __secondary, __fraction);), (return 0;))
}
_CCCL_ASSERT(__fraction > 0.0f && __fraction <= 1.0f, "fraction must be between 0.0f and 1.0f");
_CCCL_ASSERT(__secondary == __l2_evict_t::_L2_Evict_First || __secondary == __l2_evict_t::_L2_Evict_Unchanged,
"secondary policy must be evict_first or evict_unchanged");
constexpr auto __epsilon = ::cuda::std::numeric_limits<float>::epsilon();
auto __num = static_cast<uint32_t>((__fraction - __epsilon) * 16.0f); // fraction = num / 16
auto __l2_cop_off = ::cuda::std::to_underlying(__secondary);
auto __l2_cop_on = ::cuda::std::to_underlying(__primary);
auto __l2_descriptor_mode = ::cuda::std::to_underlying(__l2_descriptor_mode_t::_Desc_Interleaved);
return static_cast<uint64_t>(__num) << 52 //
| static_cast<uint64_t>(__l2_cop_off) << 56 //
| static_cast<uint64_t>(__l2_cop_on) << 57 //
| static_cast<uint64_t>(__l2_descriptor_mode) << 59;
}
inline constexpr auto __l2_interleave_normal = uint64_t{0x10F0000000000000};
inline constexpr auto __l2_interleave_streaming = uint64_t{0x12F0000000000000};
inline constexpr auto __l2_interleave_persisting = uint64_t{0x14F0000000000000};
inline constexpr auto __l2_interleave_normal_demote = uint64_t{0x16F0000000000000};
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_ENCODING_H

View File

@@ -0,0 +1,216 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_H
#define _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__annotated_ptr/access_property.h>
#include <cuda/__annotated_ptr/annotated_ptr_base.h>
#include <cuda/__memcpy_async/memcpy_async.h>
#include <cuda/__memory/address_space.h>
#include <cuda/std/cstddef>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <typename _Tp, typename _Property>
class annotated_ptr : private ::cuda::__annotated_ptr_base<_Property>
{
public:
using value_type = _Tp;
using size_type = size_t;
using reference = value_type&;
using pointer = value_type*;
using const_pointer = const value_type*;
using difference_type = ptrdiff_t;
private:
static_assert(__is_access_property_v<_Property>);
static constexpr bool __is_smem = ::cuda::std::is_same_v<_Property, access_property::shared>;
// Converting from a 64-bit to 32-bit shared pointer and maybe back just for storage might or might not be profitable.
pointer __repr = nullptr;
[[nodiscard]] _CCCL_HOST_DEVICE_API inline pointer __get(difference_type __n = 0) const noexcept
{
NV_IF_TARGET(NV_IS_DEVICE,
(auto __repr1 = const_cast<void*>(static_cast<const volatile void*>(__repr + __n));
return static_cast<pointer>(this->__apply_prop(__repr1));))
return __repr + __n;
}
[[nodiscard]] _CCCL_HOST_DEVICE_API inline pointer __offset(difference_type __n) const noexcept
{
return __get(__n);
}
public:
_CCCL_HIDE_FROM_ABI annotated_ptr() noexcept = default;
_CCCL_HOST_DEVICE_API explicit constexpr annotated_ptr(pointer __p) noexcept
: __repr{__p}
{
NV_IF_TARGET(NV_IS_HOST, (_CCCL_ASSERT(!__is_smem, "shared memory pointer is not supported on the host");))
if constexpr (__is_smem)
{
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
{
NV_IF_TARGET(NV_IS_DEVICE,
(_CCCL_ASSERT(::cuda::device::is_address_from(__p, ::cuda::device::address_space::shared),
"__p must be shared");))
}
}
else
{
_CCCL_ASSERT(__p != nullptr, "__p must not be null");
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
{
NV_IF_TARGET(NV_IS_DEVICE,
(_CCCL_ASSERT(::cuda::device::is_address_from(__p, ::cuda::device::address_space::global),
"__p must be global");))
}
}
}
template <typename _RuntimeProperty>
_CCCL_HOST_DEVICE_API inline annotated_ptr(pointer __p, _RuntimeProperty __prop) noexcept
: ::cuda::__annotated_ptr_base<_Property>{access_property{__prop}}
, __repr{__p}
{
static_assert(::cuda::std::is_same_v<_Property, access_property>,
"This method requires annotated_ptr<T, cuda::access_property>");
static_assert(__is_global_access_property_v<_RuntimeProperty>,
"This method requires RuntimeProperty=global|normal|streaming|persisting|access_property");
_CCCL_ASSERT(__p != nullptr, "__p must not be null");
NV_IF_TARGET(NV_IS_DEVICE,
(_CCCL_ASSERT(::cuda::device::is_address_from(__p, ::cuda::device::address_space::global),
"__p must be global");))
}
// cannot be constexpr because of get()
template <typename _OtherType, class _OtherProperty>
_CCCL_HOST_DEVICE_API inline annotated_ptr(const annotated_ptr<_OtherType, _OtherProperty>& __other) noexcept
: ::cuda::__annotated_ptr_base<_Property>{__other.__property()}
, __repr{__other.get()}
{
using namespace ::cuda::std;
static_assert(is_assignable_v<pointer&, _OtherType*>, "pointer must be assignable from other pointer");
static_assert(is_same_v<_Property, _OtherProperty>
|| (is_same_v<_Property, access_property> && !is_same_v<_OtherProperty, access_property::shared>),
"Both properties must have same address space, or current property is access_property and "
"OtherProperty is not shared");
}
// cannot be constexpr because is_constant_evaluated is not supported by clang-14, gcc-8.
// when the method is called in these platforms, it needs to be called at run-time.
[[nodiscard]] _CCCL_HOST_DEVICE_API inline pointer operator->() const noexcept
{
return __get();
}
[[nodiscard]] _CCCL_HOST_DEVICE_API inline reference operator*() const noexcept
{
_CCCL_ASSERT(__get() != nullptr, "dereference of null annotated_ptr");
return *__get();
}
[[nodiscard]] _CCCL_HOST_DEVICE_API inline reference operator[](difference_type __n) const noexcept
{
_CCCL_ASSERT(__offset(__n) != nullptr, "dereference of null annotated_ptr");
return *__offset(__n);
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr difference_type operator-(annotated_ptr __other) const noexcept
{
_CCCL_ASSERT(__repr >= __other.__repr, "underflow");
return __repr - __other.__repr;
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr explicit operator bool() const noexcept
{
return (__repr != nullptr);
}
// cannot be constexpr because of operator->()
[[nodiscard]] _CCCL_HOST_DEVICE_API inline pointer get() const noexcept
{
return (__is_smem || __repr == nullptr)
? __repr
: annotated_ptr<value_type, access_property::global>{__repr}.operator->();
}
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr _Property __property() const noexcept
{
return this->__get_property();
}
};
//----------------------------------------------------------------------------------------------------------------------
// memcpy_async
template <typename _Dst, typename _Src, typename _SrcProperty, typename _Shape, typename _Sync>
_CCCL_HOST_DEVICE_API inline void
memcpy_async(_Dst* __dst, annotated_ptr<_Src, _SrcProperty> __src, _Shape __shape, _Sync& __sync) noexcept
{
::cuda::memcpy_async(__dst, __src.operator->(), __shape, __sync);
}
template <typename _Dst, typename _DstProperty, typename _Src, typename _SrcProperty, typename _Shape, typename _Sync>
_CCCL_HOST_DEVICE_API inline void memcpy_async(
annotated_ptr<_Dst, _DstProperty> __dst,
annotated_ptr<_Src, _SrcProperty> __src,
_Shape __shape,
_Sync& __sync) noexcept
{
::cuda::memcpy_async(__dst.operator->(), __src.operator->(), __shape, __sync);
}
template <typename _Group, typename _Dst, typename _Src, typename _SrcProperty, typename _Shape, typename _Sync>
_CCCL_HOST_DEVICE_API inline void memcpy_async(
const _Group& __group, _Dst* __dst, annotated_ptr<_Src, _SrcProperty> __src, _Shape __shape, _Sync& __sync) noexcept
{
::cuda::memcpy_async(__group, __dst, __src.operator->(), __shape, __sync);
}
template <typename _Group,
typename _Dst,
typename _DstProperty,
typename _Src,
typename _SrcProperty,
typename _Shape,
typename _Sync>
_CCCL_HOST_DEVICE_API inline void memcpy_async(
const _Group& __group,
annotated_ptr<_Dst, _DstProperty> __dst,
annotated_ptr<_Src, _SrcProperty> __src,
_Shape __shape,
_Sync& __sync) noexcept
{
::cuda::memcpy_async(__group, __dst.operator->(), __src.operator->(), __shape, __sync);
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_H

View File

@@ -0,0 +1,100 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_BASE_H
#define _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_BASE_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__annotated_ptr/access_property.h>
#include <cuda/__annotated_ptr/associate_access_property.h>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <typename _AccessProperty>
class __annotated_ptr_base
{
protected:
_CCCL_HOST_DEVICE_API static constexpr uint64_t __default_property() noexcept
{
return ::cuda::std::is_same_v<_AccessProperty, access_property::global> ? __l2_interleave_normal
: ::cuda::std::is_same_v<_AccessProperty, access_property::normal> ? __l2_interleave_normal_demote
: ::cuda::std::is_same_v<_AccessProperty, access_property::persisting> ? __l2_interleave_persisting
: ::cuda::std::is_same_v<_AccessProperty, access_property::streaming>
? __l2_interleave_streaming
: 0; // access_property::shared;
}
static constexpr uint64_t __prop = __default_property();
_CCCL_HIDE_FROM_ABI __annotated_ptr_base() noexcept = default;
_CCCL_HOST_DEVICE_API constexpr __annotated_ptr_base(_AccessProperty) noexcept {}
#if _CCCL_CUDA_COMPILATION()
[[nodiscard]] _CCCL_DEVICE_API void* __apply_prop(void* __p) const
{
return ::cuda::__associate(__p, _AccessProperty{});
}
#endif // _CCCL_CUDA_COMPILATION()
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr _AccessProperty __get_property() const noexcept
{
return _AccessProperty{};
}
};
//----------------------------------------------------------------------------------------------------------------------
// Specialization for dynamic access property
template <>
class __annotated_ptr_base<access_property>
{
protected:
uint64_t __prop = static_cast<uint64_t>(access_property{});
_CCCL_HOST_DEVICE_API constexpr __annotated_ptr_base(access_property __property) noexcept
: __prop{static_cast<uint64_t>(__property)}
{}
_CCCL_HIDE_FROM_ABI __annotated_ptr_base() noexcept = default;
#if _CCCL_CUDA_COMPILATION()
[[nodiscard]] _CCCL_DEVICE_API void* __apply_prop(void* __p) const
{
return ::cuda::__associate_raw_descriptor(__p, __prop);
}
#endif // _CCCL_CUDA_COMPILATION()
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr access_property __get_property() const noexcept
{
return access_property{__prop};
}
};
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_BASE_H

View File

@@ -0,0 +1,83 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___ANNOTATED_PTR_APPLY_ACCESS_PROPERTY_H
#define _CUDA___ANNOTATED_PTR_APPLY_ACCESS_PROPERTY_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__annotated_ptr/access_property.h>
#include <cuda/__memory/address_space.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <typename _Shape>
_CCCL_HOST_DEVICE_API inline void apply_access_property(
[[maybe_unused]] const volatile void* __ptr,
[[maybe_unused]] _Shape __shape,
[[maybe_unused]] access_property::persisting __prop) noexcept
{
// clang-format off
NV_IF_TARGET(
NV_PROVIDES_SM_80,
(_CCCL_ASSERT(__ptr != nullptr, "null pointer");
if (!::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::global))
{
return;
}
constexpr size_t __line_size = 128;
auto __p = reinterpret_cast<uint8_t*>(const_cast<void*>(__ptr));
auto __nbytes = static_cast<size_t>(__shape);
// Apply to all 128 bytes aligned cache lines inclusive of __p
for (size_t __i = 0; __i < __nbytes; __i += __line_size) {
asm volatile("prefetch.global.L2::evict_last [%0];" ::"l"(__p + __i) :);
}))
// clang-format on
}
template <typename _Shape>
_CCCL_HOST_DEVICE_API inline void apply_access_property(
[[maybe_unused]] const volatile void* __ptr,
[[maybe_unused]] _Shape __shape,
[[maybe_unused]] access_property::normal __prop) noexcept
{
// clang-format off
NV_IF_TARGET(
NV_PROVIDES_SM_80,
(_CCCL_ASSERT(__ptr != nullptr, "null pointer");
if (!::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::global))
{
return;
}
constexpr size_t __line_size = 128;
auto __p = reinterpret_cast<uint8_t*>(const_cast<void*>(__ptr));
auto __nbytes = static_cast<size_t>(__shape);
// Apply to all 128 bytes aligned cache lines inclusive of __p
for (size_t __i = 0; __i < __nbytes; __i += __line_size) {
asm volatile("prefetch.global.L2::evict_normal [%0];" ::"l"(__p + __i) :);
}))
// clang-format on
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___ANNOTATED_PTR_APPLY_ACCESS_PROPERTY_H

View File

@@ -0,0 +1,127 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___ANNOTATED_PTR_ASSOCIATE_ACCESS_PROPERTY_H
#define _CUDA___ANNOTATED_PTR_ASSOCIATE_ACCESS_PROPERTY_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__annotated_ptr/access_property.h>
#include <cuda/__memory/address_space.h>
#include <cuda/std/__type_traits/always_false.h>
#include <cuda/std/__type_traits/is_one_of.h>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//----------------------------------------------------------------------------------------------------------------------
// Private access property methods
template <typename _Property>
inline constexpr bool __is_access_property_v =
::cuda::std::__is_one_of_v<_Property,
access_property::shared,
access_property::global,
access_property::normal,
access_property::persisting,
access_property::streaming,
access_property>;
template <typename _Property>
inline constexpr bool __is_global_access_property_v =
::cuda::std::__is_one_of_v<_Property,
access_property::global,
access_property::normal,
access_property::persisting,
access_property::streaming,
access_property>;
#if _CCCL_CUDA_COMPILATION()
template <typename _Property>
[[nodiscard]] _CCCL_DEVICE_API void* __associate_address_space(void* __ptr, [[maybe_unused]] _Property __prop)
{
if constexpr (::cuda::std::is_same_v<_Property, access_property::shared>)
{
[[maybe_unused]] bool __b = ::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::shared);
_CCCL_ASSERT(__b, "");
_CCCL_ASSUME(__b);
}
else if constexpr (__is_global_access_property_v<_Property>)
{
[[maybe_unused]] bool __b = ::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::global);
_CCCL_ASSERT(__b, "");
_CCCL_ASSUME(__b);
}
else
{
static_assert(::cuda::std::__always_false_v<_Property>, "invalid access_property");
}
return __ptr;
}
_CCCL_DEVICE_API inline void* __associate_raw_descriptor(void* __ptr, [[maybe_unused]] uint64_t __prop)
{
NV_IF_TARGET(NV_PROVIDES_SM_80, (return ::__nv_associate_access_property(__ptr, __prop);))
return __ptr;
}
template <typename _Property>
[[nodiscard]] _CCCL_DEVICE_API void* __associate_descriptor(void* __ptr, _Property __prop)
{
static_assert(__is_access_property_v<_Property>, "invalid cuda::access_property");
if constexpr (!::cuda::std::is_same_v<_Property, access_property::shared>)
{
[[maybe_unused]] auto __raw_prop = static_cast<uint64_t>(access_property{__prop});
return ::cuda::__associate_raw_descriptor(__ptr, __raw_prop);
}
return __ptr;
}
#endif // _CCCL_CUDA_COMPILATION()
template <typename _Type, typename _Property>
[[nodiscard]] _CCCL_HOST_DEVICE_API inline _Type* __associate(_Type* __ptr, [[maybe_unused]] _Property __prop) noexcept
{
static_assert(__is_access_property_v<_Property>, "invalid cuda::access_property");
NV_IF_ELSE_TARGET(
NV_IS_DEVICE,
(auto __void_ptr = const_cast<void*>(static_cast<const void*>(__ptr));
auto __associated_ptr = ::cuda::__associate_address_space(__void_ptr, __prop);
return static_cast<_Type*>(::cuda::__associate_descriptor(__associated_ptr, __prop));),
(return __ptr;))
}
//----------------------------------------------------------------------------------------------------------------------
// Public access property methods
template <typename _Tp, typename _Property>
[[nodiscard]] _CCCL_HOST_DEVICE_API inline _Tp* associate_access_property(_Tp* __ptr, _Property __prop) noexcept
{
static_assert(__is_access_property_v<_Property>, "invalid cuda::access_property");
return ::cuda::__associate(__ptr, __prop);
}
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___ANNOTATED_PTR_ASSOCIATE_ACCESS_PROPERTY_H

View File

@@ -0,0 +1,210 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___ANNOTATED_PTR_CREATEPOLICY_H
#define _CUDA___ANNOTATED_PTR_CREATEPOLICY_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__memory/address_space.h>
#include <cuda/std/cstddef>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
enum class __l2_evict_t : uint32_t
{
_L2_Evict_Unchanged = 0, // called "_L2_Evict_Normal" at lower level
_L2_Evict_First = 1,
_L2_Evict_Last = 2,
_L2_Evict_Normal_Demote = 3
};
/***********************************************************************************************************************
* PTX MAPPING
**********************************************************************************************************************/
#if _CCCL_CUDA_COMPILATION()
template <typename = void>
[[nodiscard]] _CCCL_CONST _CCCL_DEVICE_API uint64_t __createpolicy_range_ptx(
__l2_evict_t __primary, __l2_evict_t __secondary, size_t __gmem_ptr, uint32_t __primary_size, uint32_t __total_size)
{
uint64_t __policy;
if (__secondary == __l2_evict_t::_L2_Evict_Unchanged)
{
if (__primary == __l2_evict_t::_L2_Evict_Last)
{
asm("createpolicy.range.global.L2::evict_last.b64 %0, [%1], %2, %3;"
: "=l"(__policy)
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
}
else if (__primary == __l2_evict_t::_L2_Evict_Normal_Demote)
{
asm("createpolicy.range.global.L2::evict_normal.b64 %0, [%1], %2, %3;"
: "=l"(__policy)
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
}
else if (__primary == __l2_evict_t::_L2_Evict_First)
{
asm("createpolicy.range.global.L2::evict_first.b64 %0, [%1], %2, %3;"
: "=l"(__policy)
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
}
else if (__primary == __l2_evict_t::_L2_Evict_Unchanged)
{
asm("createpolicy.range.global.L2::evict_unchanged.b64 %0, [%1], %2, %3;"
: "=l"(__policy)
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
}
else
{
_CCCL_UNREACHABLE();
}
}
else // __secondary == _L2_Evict_First
{
if (__primary == __l2_evict_t::_L2_Evict_Last)
{
asm("createpolicy.range.global.L2::evict_last.L2::evict_first.b64 %0, [%1], %2, %3;"
: "=l"(__policy)
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
}
else if (__primary == __l2_evict_t::_L2_Evict_Normal_Demote)
{
asm("createpolicy.range.global.L2::evict_normal.L2::evict_first.b64 %0, [%1], %2, %3;"
: "=l"(__policy)
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
}
else if (__primary == __l2_evict_t::_L2_Evict_First)
{
asm("createpolicy.range.global.L2::evict_first.L2::evict_first.b64 %0, [%1], %2, %3;"
: "=l"(__policy)
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
}
else if (__primary == __l2_evict_t::_L2_Evict_Unchanged)
{
asm("createpolicy.range.global.L2::evict_unchanged.L2::evict_first.b64 %0, [%1], %2, %3;"
: "=l"(__policy)
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
}
else
{
_CCCL_UNREACHABLE();
}
}
return __policy;
}
template <typename = void>
[[nodiscard]] _CCCL_CONST _CCCL_DEVICE_API uint64_t
__createpolicy_fraction_ptx(__l2_evict_t __primary, __l2_evict_t __secondary, float __fraction)
{
uint64_t __policy;
if (__secondary == __l2_evict_t::_L2_Evict_Unchanged)
{
if (__primary == __l2_evict_t::_L2_Evict_Last)
{
asm("createpolicy.fractional.L2::evict_last.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
}
else if (__primary == __l2_evict_t::_L2_Evict_Normal_Demote)
{
asm("createpolicy.fractional.L2::evict_normal.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
}
else if (__primary == __l2_evict_t::_L2_Evict_First)
{
asm("createpolicy.fractional.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
}
else if (__primary == __l2_evict_t::_L2_Evict_Unchanged)
{
asm("createpolicy.fractional.L2::evict_unchanged.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
}
else
{
_CCCL_UNREACHABLE();
}
}
else // __secondary == _L2_Evict_First
{
if (__primary == __l2_evict_t::_L2_Evict_Last)
{
asm("createpolicy.fractional.L2::evict_last.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
}
else if (__primary == __l2_evict_t::_L2_Evict_Normal_Demote)
{
asm("createpolicy.fractional.L2::evict_normal.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
}
else if (__primary == __l2_evict_t::_L2_Evict_First)
{
asm("createpolicy.fractional.L2::evict_first.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
}
else if (__primary == __l2_evict_t::_L2_Evict_Unchanged)
{
asm("createpolicy.fractional.L2::evict_unchanged.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
}
else
{
_CCCL_UNREACHABLE();
}
}
return __policy;
}
/***********************************************************************************************************************
* C++ API
**********************************************************************************************************************/
extern "C" _CCCL_DEVICE void __createpolicy_is_not_supported_before_SM_80();
template <typename T = void>
[[nodiscard]] _CCCL_CONST _CCCL_DEVICE_API uint64_t __createpolicy_range(
__l2_evict_t __primary, __l2_evict_t __secondary, const void* __ptr, uint32_t __primary_size, uint32_t __total_size)
{
_CCCL_ASSERT(::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::global), "ptr must be global");
_CCCL_ASSERT(__primary_size > 0, "primary_size must be greater than zero");
_CCCL_ASSERT(__primary_size <= __total_size, "primary_size must be less than or equal to total_size");
_CCCL_ASSERT(__secondary == __l2_evict_t::_L2_Evict_First || __secondary == __l2_evict_t::_L2_Evict_Unchanged,
"secondary policy must be evict_first or evict_unchanged");
[[maybe_unused]] auto __gmem_ptr = ::__cvta_generic_to_global(__ptr);
NV_IF_ELSE_TARGET(
NV_PROVIDES_SM_80,
(return ::cuda::__createpolicy_range_ptx(__primary, __secondary, __gmem_ptr, __primary_size, __total_size);),
(::cuda::__createpolicy_is_not_supported_before_SM_80(); return 0;))
}
template <typename T = void>
[[nodiscard]] _CCCL_CONST _CCCL_DEVICE_API uint64_t
__createpolicy_fraction(__l2_evict_t __primary, __l2_evict_t __secondary, float __fraction = 1.0f)
{
_CCCL_ASSERT(__fraction > 0.0f && __fraction <= 1.0f, "fraction must be between 0.0f and 1.0f");
_CCCL_ASSERT(__secondary == __l2_evict_t::_L2_Evict_First || __secondary == __l2_evict_t::_L2_Evict_Unchanged,
"secondary policy must be evict_first or evict_unchanged");
NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
(return ::cuda::__createpolicy_fraction_ptx(__primary, __secondary, __fraction);),
(::cuda::__createpolicy_is_not_supported_before_SM_80(); return 0;))
}
#endif // _CCCL_CUDA_COMPILATION()
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___ANNOTATED_PTR_CREATEPOLICY_H

View File

@@ -0,0 +1,997 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___ARGUMENT_ARGUMENT_H
#define _CUDA___ARGUMENT_ARGUMENT_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__argument/argument_bounds.h>
#include <cuda/std/__algorithm/max_element.h>
#include <cuda/std/__algorithm/min_element.h>
#include <cuda/std/__cccl/assert.h>
#include <cuda/std/__iterator/iterator_traits.h>
#include <cuda/std/__iterator/readable_traits.h>
#include <cuda/std/__ranges/concepts.h>
#include <cuda/std/__type_traits/is_arithmetic.h>
#include <cuda/std/__type_traits/is_array.h>
#include <cuda/std/__type_traits/is_integer.h>
#include <cuda/std/__type_traits/is_integral.h>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/std/__type_traits/remove_cv.h>
#include <cuda/std/__type_traits/remove_cvref.h>
#include <cuda/std/__type_traits/void_t.h>
#include <cuda/std/__utility/cmp.h>
#include <cuda/std/__utility/declval.h>
#include <cuda/std/__utility/forward.h>
#include <cuda/std/__utility/move.h>
#include <cuda/std/cstddef>
#include <cuda/std/limits>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_ARGUMENT
struct __access;
// =====================================================================
// __element_type_of
// =====================================================================
template <class _Tp, class = void>
struct __element_type_from_member_iterator
{
using type = _Tp;
};
template <class _Tp>
struct __element_type_from_member_iterator<_Tp, ::cuda::std::void_t<::cuda::std::iter_value_t<typename _Tp::iterator>>>
{
using type = ::cuda::std::iter_value_t<typename _Tp::iterator>;
};
// Fallback: element type is the type itself.
template <class _Tp, class = void>
struct __element_type_of : __element_type_from_member_iterator<_Tp>
{};
template <class _Tp>
struct __element_type_of<_Tp, ::cuda::std::void_t<::cuda::std::iter_value_t<_Tp>>>
{
using type = ::cuda::std::iter_value_t<_Tp>;
};
template <class _Tp>
using __element_type_of_t = typename __element_type_of<::cuda::std::remove_cvref_t<_Tp>>::type;
// =====================================================================
// __is_sequence_v
// =====================================================================
template <class _Tp>
inline constexpr bool __is_sequence_v =
(::cuda::std::is_array_v<::cuda::std::remove_cvref_t<_Tp>> || ::cuda::std::ranges::range<_Tp>)
|| ::cuda::std::__has_random_access_traversal<_Tp>;
// =====================================================================
// constant
// =====================================================================
// Non-sequence wrappers intentionally do not reject types with a distinct element type.
// A pointer or iterator can represent either a single value or a sequence; the wrapper
// spelling carries that intent.
//! @brief Wraps a compile-time constant argument value.
template <auto _Value, class _Tp = ::cuda::std::remove_cvref_t<decltype(_Value)>>
class constant
{
public:
using value_type = ::cuda::std::remove_cvref_t<_Tp>;
using __element_type = value_type;
[[nodiscard]] _CCCL_API static constexpr value_type __get_value() noexcept
{
return static_cast<value_type>(_Value);
}
};
//! @brief Wraps a compile-time constant argument sequence.
template <auto _Value>
class __constant_sequence
{
public:
using value_type = ::cuda::std::remove_cvref_t<decltype(_Value)>;
using __element_type = __element_type_of_t<value_type>;
static_assert(__is_sequence_v<value_type>, "The value type of __constant_sequence must be a sequence");
};
// __assert_in_range
// =====================================================================
template <class _To, class _From>
_CCCL_API constexpr void __assert_in_range([[maybe_unused]] _From __val) noexcept
{
if constexpr (::cuda::std::__cccl_is_cv_integer_v<_To> && ::cuda::std::__cccl_is_cv_integer_v<_From>)
{
_CCCL_ASSERT(::cuda::std::in_range<::cuda::std::remove_cv_t<_To>>(__val),
"runtime bound value overflows the element type");
}
}
template <class _To, class _From>
[[nodiscard]] _CCCL_API constexpr _To __runtime_bound_cast(_From __val) noexcept
{
__assert_in_range<_To>(__val);
return static_cast<_To>(__val);
}
template <class _To, auto _Value>
_CCCL_API constexpr bool __static_bound_in_range() noexcept
{
using _RawTo = ::cuda::std::remove_cv_t<_To>;
using _RawFrom = ::cuda::std::remove_cv_t<decltype(_Value)>;
if constexpr (::cuda::std::__cccl_is_integer_v<_RawTo> && ::cuda::std::__cccl_is_integer_v<_RawFrom>)
{
return ::cuda::std::in_range<_RawTo>(_Value);
}
else if constexpr (::cuda::std::is_arithmetic_v<_RawTo> && ::cuda::std::is_arithmetic_v<_RawFrom>)
{
return static_cast<_RawFrom>(static_cast<_RawTo>(_Value)) == _Value;
}
else
{
return true;
}
}
template <class _ElementType, class _StaticBounds>
inline constexpr bool __valid_static_bounds_v = false;
template <class _ElementType>
inline constexpr bool __valid_static_bounds_v<_ElementType, no_bounds> = true;
template <class _ElementType, auto _Lowest, auto _Highest>
inline constexpr bool __valid_static_bounds_v<_ElementType, static_bounds<_Lowest, _Highest>> =
__static_bound_in_range<_ElementType, _Lowest>() && __static_bound_in_range<_ElementType, _Highest>();
template <class _ElementType, class _StaticBounds>
_CCCL_API constexpr _ElementType __wrapper_static_lowest() noexcept
{
if constexpr (::cuda::std::is_same_v<_StaticBounds, no_bounds>)
{
return __type_lowest<_ElementType>();
}
else
{
return static_cast<_ElementType>(_StaticBounds::lower());
}
}
template <class _ElementType, class _StaticBounds>
_CCCL_API constexpr _ElementType __wrapper_static_highest() noexcept
{
if constexpr (::cuda::std::is_same_v<_StaticBounds, no_bounds>)
{
return __type_highest<_ElementType>();
}
else
{
return static_cast<_ElementType>(_StaticBounds::upper());
}
}
template <class _ElementType, class _StaticBounds>
_CCCL_API constexpr _ElementType __effective_lowest(runtime_bounds<_ElementType> __runtime_bounds) noexcept
{
const auto __static_lowest = __wrapper_static_lowest<_ElementType, _StaticBounds>();
return __static_lowest < __runtime_bounds.lower() ? __runtime_bounds.lower() : __static_lowest;
}
template <class _ElementType, class _StaticBounds>
_CCCL_API constexpr _ElementType __effective_highest(runtime_bounds<_ElementType> __runtime_bounds) noexcept
{
const auto __static_highest = __wrapper_static_highest<_ElementType, _StaticBounds>();
return __static_highest < __runtime_bounds.upper() ? __static_highest : __runtime_bounds.upper();
}
template <class _ElementType, class _StaticBounds>
_CCCL_API constexpr bool __has_bounds_intersection(runtime_bounds<_ElementType> __runtime_bounds) noexcept
{
return __bounds_less_equal(__effective_lowest<_ElementType, _StaticBounds>(__runtime_bounds),
__effective_highest<_ElementType, _StaticBounds>(__runtime_bounds));
}
template <class _ElementType, class _StaticBounds>
_CCCL_API constexpr void __validate_bounds_intersection(runtime_bounds<_ElementType> __runtime_bounds) noexcept
{
static_assert(__valid_static_bounds_v<_ElementType, _StaticBounds>,
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
"values representable by the element type");
_CCCL_VERIFY((__has_bounds_intersection<_ElementType, _StaticBounds>(__runtime_bounds)),
"static and runtime argument bounds do not intersect");
}
template <class _ElementType, class _StaticBounds>
_CCCL_API constexpr void __validate_static_element_bounds([[maybe_unused]] const _ElementType& __val) noexcept
{
if constexpr (!::cuda::std::is_same_v<_StaticBounds, no_bounds>)
{
_CCCL_ASSERT((__bounds_greater_equal(__val, __wrapper_static_lowest<_ElementType, _StaticBounds>())),
"immediate argument value is below static lowest bound");
_CCCL_ASSERT((__bounds_less_equal(__val, __wrapper_static_highest<_ElementType, _StaticBounds>())),
"immediate argument value is above static highest bound");
}
}
template <class _ElementType>
_CCCL_API constexpr void __validate_runtime_element_bounds(
[[maybe_unused]] const _ElementType& __val, [[maybe_unused]] runtime_bounds<_ElementType> __runtime_bounds) noexcept
{
_CCCL_ASSERT((__bounds_greater_equal(__val, __runtime_bounds.lower())),
"immediate argument value is below runtime lower bound");
_CCCL_ASSERT((__bounds_less_equal(__val, __runtime_bounds.upper())),
"immediate argument value is above runtime upper bound");
}
// =====================================================================
// immediate
// =====================================================================
//! @brief Wraps a runtime argument value with optional bounds.
//!
//! The value is host-accessible at API call time.
template <class _Arg, class _StaticBounds = no_bounds>
class immediate
{
public:
using __element_type = __element_type_of_t<_Arg>;
static_assert(__valid_static_bounds_v<__element_type, _StaticBounds>,
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
"values representable by the element type");
private:
friend struct __access;
_Arg __arg_;
_CCCL_API constexpr void __validate_value() const noexcept
{
if constexpr (::cuda::std::is_same_v<::cuda::std::remove_cvref_t<_Arg>, __element_type>
&& ::cuda::std::is_arithmetic_v<__element_type>)
{
__validate_static_element_bounds<__element_type, _StaticBounds>(__arg_);
}
}
public:
_CCCL_API constexpr immediate(_Arg __arg) noexcept
: __arg_{::cuda::std::move(__arg)}
{
__validate_value();
}
_CCCL_API constexpr immediate(_Arg __arg, _StaticBounds) noexcept
: __arg_{::cuda::std::move(__arg)}
{
__validate_value();
}
};
#ifndef _CCCL_DOXYGEN_INVOKED
template <class _Arg, auto _Lowest, auto _Highest>
_CCCL_HOST_DEVICE immediate(_Arg, static_bounds<_Lowest, _Highest>)
-> immediate<_Arg, static_bounds<_Lowest, _Highest>>;
#endif // _CCCL_DOXYGEN_INVOKED
// =====================================================================
// __immediate_sequence
// =====================================================================
//! @brief Wraps a runtime argument sequence with optional bounds.
template <class _Arg, class _StaticBounds = no_bounds>
class __immediate_sequence
{
public:
using __element_type = __element_type_of_t<_Arg>;
static_assert(__is_sequence_v<_Arg>, "immediate sequence arguments must have a distinct element type");
static_assert(__valid_static_bounds_v<__element_type, _StaticBounds>,
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
"values representable by the element type");
private:
friend struct __access;
_Arg __arg_;
runtime_bounds<__element_type> __runtime_bounds_{};
_CCCL_API constexpr void __validate_bounds() const noexcept
{
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
}
_CCCL_API constexpr void __validate_element(const __element_type& __val) const noexcept
{
__validate_static_element_bounds<__element_type, _StaticBounds>(__val);
__validate_runtime_element_bounds(__val, __runtime_bounds_);
}
_CCCL_API constexpr void __validate_value() const noexcept
{
if constexpr (::cuda::std::__has_random_access_traversal<_Arg>)
{ // FIXME: (miscco) This is broken. we do not know the size of the sequence
}
else if constexpr (__is_sequence_v<_Arg> && !::cuda::std::__has_random_access_traversal<_Arg>
&& ::cuda::std::is_arithmetic_v<__element_type>)
{
for (const auto& __a : __arg_)
{
__validate_element(__a);
}
}
}
public:
_CCCL_API constexpr __immediate_sequence(_Arg __arg) noexcept
: __arg_{::cuda::std::move(__arg)}
{
__validate_bounds();
__validate_value();
}
_CCCL_API constexpr __immediate_sequence(_Arg __arg, _StaticBounds) noexcept
: __arg_{::cuda::std::move(__arg)}
{
__validate_bounds();
__validate_value();
}
template <class _BoundsTp>
_CCCL_API constexpr __immediate_sequence(_Arg __arg, runtime_bounds<_BoundsTp> __rb) noexcept
: __arg_{::cuda::std::move(__arg)}
, __runtime_bounds_{__runtime_bound_cast<__element_type>(__rb.lower()),
__runtime_bound_cast<__element_type>(__rb.upper())}
{
__validate_bounds();
__validate_value();
}
template <class _BoundsTp>
_CCCL_API constexpr __immediate_sequence(_Arg __arg, _StaticBounds, runtime_bounds<_BoundsTp> __rb) noexcept
: __arg_{::cuda::std::move(__arg)}
, __runtime_bounds_{__runtime_bound_cast<__element_type>(__rb.lower()),
__runtime_bound_cast<__element_type>(__rb.upper())}
{
__validate_bounds();
__validate_value();
}
template <class _BoundsTp>
_CCCL_API constexpr __immediate_sequence(_Arg __arg, runtime_bounds<_BoundsTp> __rb, _StaticBounds __sb) noexcept
: __immediate_sequence(::cuda::std::move(__arg), __sb, __rb)
{}
};
#ifndef _CCCL_DOXYGEN_INVOKED
template <class _Arg, auto _Lowest, auto _Highest>
_CCCL_HOST_DEVICE __immediate_sequence(_Arg, static_bounds<_Lowest, _Highest>)
-> __immediate_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
template <class _Arg, auto _Lowest, auto _Highest, class _Tp>
_CCCL_HOST_DEVICE __immediate_sequence(_Arg, static_bounds<_Lowest, _Highest>, runtime_bounds<_Tp>)
-> __immediate_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
template <class _Arg, class _Tp, auto _Lowest, auto _Highest>
_CCCL_HOST_DEVICE __immediate_sequence(_Arg, runtime_bounds<_Tp>, static_bounds<_Lowest, _Highest>)
-> __immediate_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
#endif // _CCCL_DOXYGEN_INVOKED
// =====================================================================
// __deferred_base / deferred / deferred_sequence
// =====================================================================
//! @brief Common base for deferred argument wrappers.
template <class _Arg, class _StaticBounds = no_bounds>
class __deferred_base
{
public:
using __element_type = __element_type_of_t<_Arg>;
static_assert(__valid_static_bounds_v<__element_type, _StaticBounds>,
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
"values representable by the element type");
private:
friend struct __access;
_Arg __arg_;
runtime_bounds<__element_type> __runtime_bounds_{};
public:
_CCCL_API constexpr __deferred_base(_Arg __arg) noexcept
: __arg_{::cuda::std::move(__arg)}
{
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
}
_CCCL_API constexpr __deferred_base(_Arg __arg, _StaticBounds) noexcept
: __arg_{::cuda::std::move(__arg)}
{
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
}
template <class _BoundsTp>
_CCCL_API constexpr __deferred_base(_Arg __arg, runtime_bounds<_BoundsTp> __rb) noexcept
: __arg_{::cuda::std::move(__arg)}
, __runtime_bounds_{__runtime_bound_cast<__element_type>(__rb.lower()),
__runtime_bound_cast<__element_type>(__rb.upper())}
{
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
}
template <class _BoundsTp>
_CCCL_API constexpr __deferred_base(_Arg __arg, _StaticBounds, runtime_bounds<_BoundsTp> __rb) noexcept
: __arg_{::cuda::std::move(__arg)}
, __runtime_bounds_{__runtime_bound_cast<__element_type>(__rb.lower()),
__runtime_bound_cast<__element_type>(__rb.upper())}
{
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
}
template <class _BoundsTp>
_CCCL_API constexpr __deferred_base(_Arg __arg, runtime_bounds<_BoundsTp> __rb, _StaticBounds __sb) noexcept
: __deferred_base(::cuda::std::move(__arg), __sb, __rb)
{}
};
//! @brief Wraps a reference to a single value that is potentially not available at API call time but will be available
//! by the time the argument is consumed in stream order.
template <class _Arg, class _StaticBounds = no_bounds>
class deferred : public __deferred_base<_Arg, _StaticBounds>
{
public:
using __deferred_base<_Arg, _StaticBounds>::__deferred_base;
};
#ifndef _CCCL_DOXYGEN_INVOKED
template <class _Arg>
_CCCL_HOST_DEVICE deferred(_Arg) -> deferred<_Arg>;
template <class _Arg, auto _Lowest, auto _Highest>
_CCCL_HOST_DEVICE deferred(_Arg, static_bounds<_Lowest, _Highest>) -> deferred<_Arg, static_bounds<_Lowest, _Highest>>;
template <class _Arg, class _Tp>
_CCCL_HOST_DEVICE deferred(_Arg, runtime_bounds<_Tp>) -> deferred<_Arg>;
template <class _Arg, auto _Lowest, auto _Highest, class _Tp>
_CCCL_HOST_DEVICE deferred(_Arg, static_bounds<_Lowest, _Highest>, runtime_bounds<_Tp>)
-> deferred<_Arg, static_bounds<_Lowest, _Highest>>;
template <class _Arg, class _Tp, auto _Lowest, auto _Highest>
_CCCL_HOST_DEVICE deferred(_Arg, runtime_bounds<_Tp>, static_bounds<_Lowest, _Highest>)
-> deferred<_Arg, static_bounds<_Lowest, _Highest>>;
#endif // _CCCL_DOXYGEN_INVOKED
//! @brief Wraps a reference to a sequence of values that is potentially not available at API call time but will be
//! available by the time the argument is consumed in stream order.
template <class _Arg, class _StaticBounds = no_bounds>
class deferred_sequence : public __deferred_base<_Arg, _StaticBounds>
{
public:
static_assert(__is_sequence_v<_Arg>, "deferred sequence arguments must have a distinct element type");
using __deferred_base<_Arg, _StaticBounds>::__deferred_base;
};
#ifndef _CCCL_DOXYGEN_INVOKED
template <class _Arg>
_CCCL_HOST_DEVICE deferred_sequence(_Arg) -> deferred_sequence<_Arg>;
template <class _Arg, auto _Lowest, auto _Highest>
_CCCL_HOST_DEVICE deferred_sequence(_Arg, static_bounds<_Lowest, _Highest>)
-> deferred_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
template <class _Arg, class _Tp>
_CCCL_HOST_DEVICE deferred_sequence(_Arg, runtime_bounds<_Tp>) -> deferred_sequence<_Arg>;
template <class _Arg, auto _Lowest, auto _Highest, class _Tp>
_CCCL_HOST_DEVICE deferred_sequence(_Arg, static_bounds<_Lowest, _Highest>, runtime_bounds<_Tp>)
-> deferred_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
template <class _Arg, class _Tp, auto _Lowest, auto _Highest>
_CCCL_HOST_DEVICE deferred_sequence(_Arg, runtime_bounds<_Tp>, static_bounds<_Lowest, _Highest>)
-> deferred_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
#endif // _CCCL_DOXYGEN_INVOKED
// =====================================================================
// __access
// =====================================================================
struct __access
{
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr _Arg& __arg(immediate<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__arg_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr const _Arg& __arg(const immediate<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__arg_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr _Arg&& __arg(immediate<_Arg, _StaticBounds>&& __wrapper) noexcept
{
return ::cuda::std::move(__wrapper.__arg_);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr _Arg& __arg(__immediate_sequence<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__arg_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr const _Arg&
__arg(const __immediate_sequence<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__arg_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr _Arg&& __arg(__immediate_sequence<_Arg, _StaticBounds>&& __wrapper) noexcept
{
return ::cuda::std::move(__wrapper.__arg_);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr _Arg& __arg(__deferred_base<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__arg_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr const _Arg&
__arg(const __deferred_base<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__arg_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr _Arg&& __arg(__deferred_base<_Arg, _StaticBounds>&& __wrapper) noexcept
{
return ::cuda::std::move(__wrapper.__arg_);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr runtime_bounds<__element_type_of_t<_Arg>>&
__runtime_bounds(__immediate_sequence<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__runtime_bounds_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr const runtime_bounds<__element_type_of_t<_Arg>>&
__runtime_bounds(const __immediate_sequence<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__runtime_bounds_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr runtime_bounds<__element_type_of_t<_Arg>>&
__runtime_bounds(__deferred_base<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__runtime_bounds_;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API static constexpr const runtime_bounds<__element_type_of_t<_Arg>>&
__runtime_bounds(const __deferred_base<_Arg, _StaticBounds>& __wrapper) noexcept
{
return __wrapper.__runtime_bounds_;
}
};
// =====================================================================
// __unwrap
// =====================================================================
template <class _Tp>
inline constexpr bool __is_wrapper_v = false;
template <class _Arg, class _StaticBounds>
inline constexpr bool __is_wrapper_v<immediate<_Arg, _StaticBounds>> = true;
template <auto _Value, class _Tp>
inline constexpr bool __is_wrapper_v<constant<_Value, _Tp>> = true;
template <auto _Value>
inline constexpr bool __is_wrapper_v<__constant_sequence<_Value>> = true;
template <class _Arg, class _StaticBounds>
inline constexpr bool __is_wrapper_v<__immediate_sequence<_Arg, _StaticBounds>> = true;
template <class _Arg, class _StaticBounds>
inline constexpr bool __is_wrapper_v<deferred<_Arg, _StaticBounds>> = true;
template <class _Arg, class _StaticBounds>
inline constexpr bool __is_wrapper_v<deferred_sequence<_Arg, _StaticBounds>> = true;
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES((!__is_wrapper_v<::cuda::std::remove_cvref_t<_Tp>>) )
[[nodiscard]] _CCCL_API constexpr _Tp&& __unwrap(_Tp&& __arg) noexcept
{
return ::cuda::std::forward<_Tp>(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr _Arg& __unwrap(immediate<_Arg, _StaticBounds>& __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr const _Arg& __unwrap(const immediate<_Arg, _StaticBounds>& __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr _Arg __unwrap(immediate<_Arg, _StaticBounds>&& __arg) noexcept
{
return __access::__arg(::cuda::std::move(__arg));
}
template <auto _Value, class _Tp>
[[nodiscard]] _CCCL_API constexpr typename constant<_Value, _Tp>::value_type
__unwrap(const constant<_Value, _Tp>&) noexcept
{
return constant<_Value, _Tp>::__get_value();
}
template <auto _Value>
[[nodiscard]] _CCCL_API constexpr ::cuda::std::remove_cvref_t<decltype(_Value)>
__unwrap(const __constant_sequence<_Value>&) noexcept
{
return _Value;
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr _Arg& __unwrap(__immediate_sequence<_Arg, _StaticBounds>& __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr const _Arg& __unwrap(const __immediate_sequence<_Arg, _StaticBounds>& __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr _Arg __unwrap(__immediate_sequence<_Arg, _StaticBounds>&& __arg) noexcept
{
return __access::__arg(::cuda::std::move(__arg));
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr _Arg& __unwrap(deferred<_Arg, _StaticBounds>& __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr const _Arg& __unwrap(const deferred<_Arg, _StaticBounds>& __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr _Arg __unwrap(deferred<_Arg, _StaticBounds>&& __arg) noexcept
{
return __access::__arg(::cuda::std::move(__arg));
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr _Arg& __unwrap(deferred_sequence<_Arg, _StaticBounds>& __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr const _Arg& __unwrap(const deferred_sequence<_Arg, _StaticBounds>& __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr _Arg __unwrap(deferred_sequence<_Arg, _StaticBounds>&& __arg) noexcept
{
return __access::__arg(::cuda::std::move(__arg));
}
template <auto _Value, class _Tp>
_CCCL_API constexpr auto __constant_compute_lowest() noexcept
{
return constant<_Value, _Tp>::__get_value();
}
template <auto _Value, class _Tp>
_CCCL_API constexpr auto __constant_compute_highest() noexcept
{
return constant<_Value, _Tp>::__get_value();
}
template <auto _Value>
_CCCL_API constexpr auto __constant_sequence_compute_lowest() noexcept
{
using _ElementType = __element_type_of_t<::cuda::std::remove_cvref_t<decltype(_Value)>>;
auto __first = _Value.begin();
auto __last = _Value.end();
if (__first == __last)
{
return __type_lowest<_ElementType>();
}
return static_cast<_ElementType>(*::cuda::std::min_element(__first, __last));
}
template <auto _Value>
_CCCL_API constexpr auto __constant_sequence_compute_highest() noexcept
{
using _ElementType = __element_type_of_t<::cuda::std::remove_cvref_t<decltype(_Value)>>;
auto __first = _Value.begin();
auto __last = _Value.end();
if (__first == __last)
{
return __type_highest<_ElementType>();
}
return static_cast<_ElementType>(*::cuda::std::max_element(__first, __last));
}
// =====================================================================
// __traits
// =====================================================================
//! @brief Traits for argument wrappers and plain argument values.
//!
//! Models @c numeric_limits for bounds: @c lowest is the lower bound, @c highest is the upper bound.
//! Use in @c if @c constexpr for compile-time dispatch based on bounds.
template <class _Tp>
struct __traits_impl
{
using value_type = _Tp;
using element_type = __element_type_of_t<_Tp>;
static constexpr bool is_constant = false;
static constexpr bool is_deferred = false;
static constexpr bool is_single_value = true;
static constexpr element_type lowest = __type_lowest<element_type>();
static constexpr element_type highest = __type_highest<element_type>();
};
template <auto _Value, class _Tp>
struct __traits_impl<constant<_Value, _Tp>>
{
using value_type = typename constant<_Value, _Tp>::value_type;
using element_type = value_type;
static constexpr bool is_constant = true;
static constexpr bool is_deferred = false;
static constexpr bool is_single_value = true;
static constexpr element_type lowest = __constant_compute_lowest<_Value, _Tp>();
static constexpr element_type highest = __constant_compute_highest<_Value, _Tp>();
};
template <class _Arg, class _StaticBounds>
struct __traits_impl<immediate<_Arg, _StaticBounds>>
{
using value_type = _Arg;
using element_type = __element_type_of_t<_Arg>;
static_assert(__valid_static_bounds_v<element_type, _StaticBounds>,
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
"values representable by the element type");
static constexpr bool is_constant = false;
static constexpr bool is_deferred = false;
static constexpr bool is_single_value = true;
static constexpr element_type lowest = __wrapper_static_lowest<element_type, _StaticBounds>();
static constexpr element_type highest = __wrapper_static_highest<element_type, _StaticBounds>();
};
template <auto _Value>
struct __traits_impl<__constant_sequence<_Value>>
{
using value_type = ::cuda::std::remove_cvref_t<decltype(_Value)>;
using element_type = __element_type_of_t<value_type>;
static_assert(__is_sequence_v<value_type>, "The value type of __constant_sequence must be a sequence");
static constexpr bool is_constant = true;
static constexpr bool is_deferred = false;
static constexpr bool is_single_value = false;
static constexpr element_type lowest = __constant_sequence_compute_lowest<_Value>();
static constexpr element_type highest = __constant_sequence_compute_highest<_Value>();
};
template <class _Arg, class _StaticBounds>
struct __traits_impl<__immediate_sequence<_Arg, _StaticBounds>>
{
using value_type = _Arg;
using element_type = __element_type_of_t<_Arg>;
static_assert(__is_sequence_v<value_type>, "immediate sequence arguments must have a distinct element type");
static_assert(__valid_static_bounds_v<element_type, _StaticBounds>,
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
"values representable by the element type");
static constexpr bool is_constant = false;
static constexpr bool is_deferred = false;
static constexpr bool is_single_value = false;
static constexpr element_type lowest = __wrapper_static_lowest<element_type, _StaticBounds>();
static constexpr element_type highest = __wrapper_static_highest<element_type, _StaticBounds>();
};
template <class _Arg, class _StaticBounds>
struct __traits_impl<deferred<_Arg, _StaticBounds>>
{
using value_type = _Arg;
using element_type = __element_type_of_t<_Arg>;
static_assert(__valid_static_bounds_v<element_type, _StaticBounds>,
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
"values representable by the element type");
static constexpr bool is_constant = false;
static constexpr bool is_deferred = true;
static constexpr bool is_single_value = true;
static constexpr element_type lowest = __wrapper_static_lowest<element_type, _StaticBounds>();
static constexpr element_type highest = __wrapper_static_highest<element_type, _StaticBounds>();
};
template <class _Arg, class _StaticBounds>
struct __traits_impl<deferred_sequence<_Arg, _StaticBounds>>
{
using value_type = _Arg;
using element_type = __element_type_of_t<_Arg>;
static_assert(__is_sequence_v<value_type>, "deferred sequence arguments must have a distinct element type");
static_assert(__valid_static_bounds_v<element_type, _StaticBounds>,
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
"values representable by the element type");
static constexpr bool is_constant = false;
static constexpr bool is_deferred = true;
static constexpr bool is_single_value = false;
static constexpr element_type lowest = __wrapper_static_lowest<element_type, _StaticBounds>();
static constexpr element_type highest = __wrapper_static_highest<element_type, _StaticBounds>();
};
template <class _Tp>
struct __traits : __traits_impl<::cuda::std::remove_cvref_t<_Tp>>
{};
// =====================================================================
// __lowest_ / __highest_ — free functions
// =====================================================================
//! @brief Returns the effective lowest bound, combining static and runtime bounds.
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES((!__is_wrapper_v<::cuda::std::remove_cv_t<_Tp>>) )
[[nodiscard]] _CCCL_API constexpr auto __lowest_(_Tp) noexcept
{
return __type_lowest<__element_type_of_t<_Tp>>();
}
template <auto _Value, class _Tp>
[[nodiscard]] _CCCL_API constexpr auto __lowest_(constant<_Value, _Tp>) noexcept
{
return __constant_compute_lowest<_Value, _Tp>();
}
template <auto _Value>
[[nodiscard]] _CCCL_API constexpr auto __lowest_(__constant_sequence<_Value>) noexcept
{
return __constant_sequence_compute_lowest<_Value>();
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr auto __lowest_(immediate<_Arg, _StaticBounds> __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr auto __lowest_(__immediate_sequence<_Arg, _StaticBounds> __arg) noexcept
{
using _ET = __element_type_of_t<_Arg>;
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
return __effective_lowest<_ET, _StaticBounds>(__runtime_bounds);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr auto __lowest_(deferred<_Arg, _StaticBounds> __arg) noexcept
{
using _ET = __element_type_of_t<_Arg>;
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
return __effective_lowest<_ET, _StaticBounds>(__runtime_bounds);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr auto __lowest_(deferred_sequence<_Arg, _StaticBounds> __arg) noexcept
{
using _ET = __element_type_of_t<_Arg>;
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
return __effective_lowest<_ET, _StaticBounds>(__runtime_bounds);
}
//! @brief Returns the effective highest bound, combining static and runtime bounds.
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES((!__is_wrapper_v<::cuda::std::remove_cv_t<_Tp>>) )
[[nodiscard]] _CCCL_API constexpr auto __highest_(_Tp) noexcept
{
return __type_highest<__element_type_of_t<_Tp>>();
}
template <auto _Value, class _Tp>
[[nodiscard]] _CCCL_API constexpr auto __highest_(constant<_Value, _Tp>) noexcept
{
return __constant_compute_highest<_Value, _Tp>();
}
template <auto _Value>
[[nodiscard]] _CCCL_API constexpr auto __highest_(__constant_sequence<_Value>) noexcept
{
return __constant_sequence_compute_highest<_Value>();
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr auto __highest_(immediate<_Arg, _StaticBounds> __arg) noexcept
{
return __access::__arg(__arg);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr auto __highest_(__immediate_sequence<_Arg, _StaticBounds> __arg) noexcept
{
using _ET = __element_type_of_t<_Arg>;
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
return __effective_highest<_ET, _StaticBounds>(__runtime_bounds);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr auto __highest_(deferred<_Arg, _StaticBounds> __arg) noexcept
{
using _ET = __element_type_of_t<_Arg>;
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
return __effective_highest<_ET, _StaticBounds>(__runtime_bounds);
}
template <class _Arg, class _StaticBounds>
[[nodiscard]] _CCCL_API constexpr auto __highest_(deferred_sequence<_Arg, _StaticBounds> __arg) noexcept
{
using _ET = __element_type_of_t<_Arg>;
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
return __effective_highest<_ET, _StaticBounds>(__runtime_bounds);
}
_CCCL_END_NAMESPACE_CUDA_ARGUMENT
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___ARGUMENT_ARGUMENT_H

Some files were not shown because too many files have changed in this diff Show More