[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
49
cccl_upstream/libcudacxx/CMakeLists.txt
Normal file
49
cccl_upstream/libcudacxx/CMakeLists.txt
Normal file
@@ -0,0 +1,49 @@
|
||||
if (NOT CCCL_ENABLE_LIBCUDACXX)
|
||||
include(cmake/libcudacxxAddSubdir.cmake)
|
||||
return()
|
||||
endif()
|
||||
|
||||
cmake_minimum_required(VERSION 3.21)
|
||||
set(PACKAGE_NAME libcudacxx)
|
||||
set(PACKAGE_VERSION 11.0)
|
||||
set(PACKAGE_STRING "${PACKAGE_NAME} ${PACKAGE_VERSION}")
|
||||
project(libcudacxx LANGUAGES CXX)
|
||||
|
||||
# Add codegen module
|
||||
option(
|
||||
libcudacxx_ENABLE_CODEGEN
|
||||
"Enable libcudacxx's atomics backend codegen and tests."
|
||||
OFF
|
||||
)
|
||||
if (libcudacxx_ENABLE_CODEGEN)
|
||||
add_subdirectory(codegen)
|
||||
endif()
|
||||
|
||||
list(PREPEND CMAKE_MODULE_PATH "${libcudacxx_SOURCE_DIR}/cmake")
|
||||
set(LLVM_PATH "${libcudacxx_SOURCE_DIR}" CACHE STRING "" FORCE)
|
||||
|
||||
# Configuration options.
|
||||
option(LIBCUDACXX_ENABLE_CUDA "Enable the CUDA language support." ON)
|
||||
if (LIBCUDACXX_ENABLE_CUDA)
|
||||
enable_language(CUDA)
|
||||
endif()
|
||||
|
||||
option(LIBCUDACXX_ENABLE_LIBCUDACXX_TESTS "Enable libcu++ tests." ON)
|
||||
if (LIBCUDACXX_ENABLE_LIBCUDACXX_TESTS)
|
||||
enable_testing()
|
||||
|
||||
# Create the compiler targets with the common compile flags
|
||||
include(cmake/LibcudacxxBuildCompilerTargets.cmake)
|
||||
libcudacxx_build_compiler_targets()
|
||||
|
||||
# Test all public and internal headers
|
||||
include(cmake/LibcudacxxInternalHeaderTesting.cmake)
|
||||
include(cmake/LibcudacxxPublicHeaderTesting.cmake)
|
||||
include(cmake/LibcudacxxPublicHeaderTestingHost.cmake)
|
||||
|
||||
add_subdirectory(test)
|
||||
endif()
|
||||
|
||||
if (CCCL_ENABLE_BENCHMARKS)
|
||||
add_subdirectory(benchmarks)
|
||||
endif()
|
||||
311
cccl_upstream/libcudacxx/LICENSE.TXT
Normal file
311
cccl_upstream/libcudacxx/LICENSE.TXT
Normal file
@@ -0,0 +1,311 @@
|
||||
==============================================================================
|
||||
libcu++ is under the Apache License v2.0 with LLVM Exceptions:
|
||||
==============================================================================
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
|
||||
|
||||
---- LLVM Exceptions to the Apache 2.0 License ----
|
||||
|
||||
As an exception, if, as a result of your compiling your source code, portions
|
||||
of this Software are embedded into an Object form of such source code, you
|
||||
may redistribute such embedded portions in such Object form without complying
|
||||
with the conditions of Sections 4(a), 4(b) and 4(d) of the License.
|
||||
|
||||
In addition, if you combine or link compiled forms of this Software with
|
||||
software that is licensed under the GPLv2 ("Combined Software") and if a
|
||||
court of competent jurisdiction determines that the patent provision (Section
|
||||
3), the indemnity provision (Section 9) or other Section of the License
|
||||
conflicts with the conditions of the GPLv2, you may retroactively and
|
||||
prospectively choose to deem waived or otherwise exclude such Section(s) of
|
||||
the License, but only in their entirety and only with respect to the Combined
|
||||
Software.
|
||||
|
||||
==============================================================================
|
||||
Software from third parties included in the LLVM Project:
|
||||
==============================================================================
|
||||
The LLVM Project contains third party software which is under different license
|
||||
terms. All such code will be identified clearly using at least one of two
|
||||
mechanisms:
|
||||
1) It will be in a separate directory tree with its own `LICENSE.txt` or
|
||||
`LICENSE` file at the top containing the specific license and restrictions
|
||||
which apply to that software, or
|
||||
2) It will contain specific license and restriction terms at the top of every
|
||||
file.
|
||||
|
||||
==============================================================================
|
||||
Legacy LLVM License (https://llvm.org/docs/DeveloperPolicy.html#legacy):
|
||||
==============================================================================
|
||||
|
||||
The libc++ library is dual licensed under both the University of Illinois
|
||||
"BSD-Like" license and the MIT license. As a user of this code you may choose
|
||||
to use it under either license. As a contributor, you agree to allow your code
|
||||
to be used under both.
|
||||
|
||||
Full text of the relevant licenses is included below.
|
||||
|
||||
==============================================================================
|
||||
|
||||
University of Illinois/NCSA
|
||||
Open Source License
|
||||
|
||||
Copyright (c) 2009-2019 by the contributors listed in CREDITS.TXT
|
||||
|
||||
All rights reserved.
|
||||
|
||||
Developed by:
|
||||
|
||||
LLVM Team
|
||||
|
||||
University of Illinois at Urbana-Champaign
|
||||
|
||||
http://llvm.org
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal with
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
of the Software, and to permit persons to whom the Software is furnished to do
|
||||
so, subject to the following conditions:
|
||||
|
||||
* Redistributions of source code must retain the above copyright notice,
|
||||
this list of conditions and the following disclaimers.
|
||||
|
||||
* Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimers in the
|
||||
documentation and/or other materials provided with the distribution.
|
||||
|
||||
* Neither the names of the LLVM Team, University of Illinois at
|
||||
Urbana-Champaign, nor the names of its contributors may be used to
|
||||
endorse or promote products derived from this Software without specific
|
||||
prior written permission.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS WITH THE
|
||||
SOFTWARE.
|
||||
|
||||
==============================================================================
|
||||
|
||||
Copyright (c) 2009-2014 by the contributors listed in CREDITS.TXT
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
92
cccl_upstream/libcudacxx/benchmarks/CMakeLists.txt
Normal file
92
cccl_upstream/libcudacxx/benchmarks/CMakeLists.txt
Normal file
@@ -0,0 +1,92 @@
|
||||
include(${CMAKE_SOURCE_DIR}/benchmarks/cmake/CCCLBenchmarkRegistry.cmake)
|
||||
|
||||
cccl_get_nvbench_helper()
|
||||
|
||||
set(benches_root "${CMAKE_CURRENT_LIST_DIR}")
|
||||
|
||||
if (NOT CMAKE_BUILD_TYPE STREQUAL "Release")
|
||||
set(message_type FATAL_ERROR)
|
||||
if (CCCL_ENABLE_CLANG_TIDY)
|
||||
# We are here because CI has force-enabled clang-tidy. We must use a debug build for
|
||||
# this because certain clang-tidy checks (such as out of bounds or clang static
|
||||
# analyzer) work better when they see assert()'s. In this case we don't actually
|
||||
# intend to run any of the benchmarks, we just need them to be compilable, so a simple
|
||||
# warning is enough.
|
||||
#
|
||||
# We don't ignore this outright (by making it say, DEBUG or VERBOSE), because it's
|
||||
# possible that a user may accidentally stumble into enabling the option.
|
||||
set(message_type WARNING)
|
||||
endif()
|
||||
message(${message_type} "libcu++ benchmarks must be built in release mode.")
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED CMAKE_CUDA_ARCHITECTURES)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"CMAKE_CUDA_ARCHITECTURES must be set to build libcu++ benchmarks."
|
||||
)
|
||||
endif()
|
||||
|
||||
set(benches_meta_target libcudacxx.all.benches)
|
||||
add_custom_target(${benches_meta_target})
|
||||
|
||||
function(get_recursive_subdirs subdirs)
|
||||
set(dirs)
|
||||
file(
|
||||
GLOB_RECURSE contents
|
||||
CONFIGURE_DEPENDS
|
||||
LIST_DIRECTORIES ON
|
||||
"${CMAKE_CURRENT_LIST_DIR}/bench/*"
|
||||
)
|
||||
|
||||
foreach (test_dir IN LISTS contents)
|
||||
if (IS_DIRECTORY "${test_dir}")
|
||||
list(APPEND dirs "${test_dir}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
set(${subdirs} "${dirs}" PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
create_benchmark_registry()
|
||||
|
||||
function(add_bench target_name bench_name bench_src)
|
||||
set(bench_target ${bench_name})
|
||||
set(${target_name} ${bench_target} PARENT_SCOPE)
|
||||
|
||||
cccl_add_executable(${bench_target} SOURCES "${bench_src}")
|
||||
target_link_libraries(
|
||||
${bench_target}
|
||||
PRIVATE libcudacxx::libcudacxx cccl.nvbench_helper nvbench::main
|
||||
)
|
||||
endfunction()
|
||||
|
||||
function(add_bench_dir bench_dir)
|
||||
file(GLOB bench_srcs CONFIGURE_DEPENDS "${bench_dir}/*.cu")
|
||||
file(RELATIVE_PATH bench_prefix "${benches_root}" "${bench_dir}")
|
||||
file(TO_CMAKE_PATH "${bench_prefix}" bench_prefix)
|
||||
string(REPLACE "/" "." bench_prefix "${bench_prefix}")
|
||||
|
||||
foreach (bench_src IN LISTS bench_srcs)
|
||||
# base tuning
|
||||
get_filename_component(bench_name "${bench_src}" NAME_WLE)
|
||||
string(PREPEND bench_name "libcudacxx.${bench_prefix}.")
|
||||
|
||||
set(base_bench_name "${bench_name}.base")
|
||||
add_bench(base_bench_target ${base_bench_name} "${bench_src}")
|
||||
add_dependencies(${benches_meta_target} ${base_bench_target})
|
||||
target_compile_definitions(${base_bench_target} PRIVATE TUNE_BASE=1)
|
||||
target_compile_options(
|
||||
${base_bench_target}
|
||||
PRIVATE "$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--extended-lambda>"
|
||||
)
|
||||
# benchmarking
|
||||
register_cccl_benchmark("${bench_name}" "")
|
||||
endforeach()
|
||||
endfunction()
|
||||
|
||||
get_recursive_subdirs(subdirs)
|
||||
|
||||
foreach (subdir IN LISTS subdirs)
|
||||
add_bench_dir("${subdir}")
|
||||
endforeach()
|
||||
@@ -0,0 +1,67 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/adjacent_difference.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_difference(cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_difference(
|
||||
cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin(), ::cuda::std::greater<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,77 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
|
||||
|
||||
thrust::device_vector<T> in(elements, thrust::no_init);
|
||||
thrust::sequence(in.begin(), in.end(), 0);
|
||||
in[mismatch_point] = in[mismatch_point + 1];
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(mismatch_point);
|
||||
state.add_global_memory_writes<T>(0);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
|
||||
|
||||
thrust::device_vector<T> in(elements, thrust::no_init);
|
||||
thrust::sequence(in.begin(), in.end(), 0);
|
||||
in[mismatch_point] = in[mismatch_point + 1];
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(mismatch_point);
|
||||
state.add_global_memory_writes<T>(0);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend(), ::cuda::std::greater<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/all_of/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/all_of/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::all_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/any_of/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/any_of/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::any_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
70
cccl_upstream/libcudacxx/benchmarks/bench/copy/basic.cu
Normal file
70
cccl_upstream/libcudacxx/benchmarks/bench/copy/basic.cu
Normal file
@@ -0,0 +1,70 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("contiguous")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void random_access(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::counting_iterator<std::size_t>{0},
|
||||
cuda::counting_iterator{elements},
|
||||
out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
|
||||
.set_name("random_access")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
51
cccl_upstream/libcudacxx/benchmarks/bench/copy_if/basic.cu
Normal file
51
cccl_upstream/libcudacxx/benchmarks/bench/copy_if/basic.cu
Normal file
@@ -0,0 +1,51 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
67
cccl_upstream/libcudacxx/benchmarks/bench/copy_n/basic.cu
Normal file
67
cccl_upstream/libcudacxx/benchmarks/bench/copy_n/basic.cu
Normal file
@@ -0,0 +1,67 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy_n(cuda_policy(alloc, launch), in.begin(), elements, out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("contiguous")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void random_access(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::copy_n(cuda_policy(alloc, launch), cuda::counting_iterator<std::size_t>{0}, elements, out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
|
||||
.set_name("random_access")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
41
cccl_upstream/libcudacxx/benchmarks/bench/count/basic.cu
Normal file
41
cccl_upstream/libcudacxx/benchmarks/bench/count/basic.cu
Normal file
@@ -0,0 +1,41 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::count(cuda_policy(alloc, launch), in.begin(), in.end(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
50
cccl_upstream/libcudacxx/benchmarks/bench/count_if/basic.cu
Normal file
50
cccl_upstream/libcudacxx/benchmarks/bench/count_if/basic.cu
Normal file
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::count_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
82
cccl_upstream/libcudacxx/benchmarks/bench/equal/basic.cu
Normal file
82
cccl_upstream/libcudacxx/benchmarks/bench/equal/basic.cu
Normal file
@@ -0,0 +1,82 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::equal(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_iter")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void range_range(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::equal(
|
||||
cuda_policy(alloc, launch),
|
||||
dinput.begin(),
|
||||
dinput.end(),
|
||||
cuda::constant_iterator<T>{0},
|
||||
cuda::constant_iterator<T>{0, elements}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_range")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,68 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::exclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, ::cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,43 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, max_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
40
cccl_upstream/libcudacxx/benchmarks/bench/fill/basic.cu
Normal file
40
cccl_upstream/libcudacxx/benchmarks/bench/fill/basic.cu
Normal file
@@ -0,0 +1,40 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::fill(cuda_policy(alloc, launch), output.begin(), output.end(), T{42});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
40
cccl_upstream/libcudacxx/benchmarks/bench/fill_n/basic.cu
Normal file
40
cccl_upstream/libcudacxx/benchmarks/bench/fill_n/basic.cu
Normal file
@@ -0,0 +1,40 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::fill_n(cuda_policy(alloc, launch), output.begin(), elements, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
46
cccl_upstream/libcudacxx/benchmarks/bench/find/basic.cu
Normal file
46
cccl_upstream/libcudacxx/benchmarks/bench/find/basic.cu
Normal file
@@ -0,0 +1,46 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::find(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), val));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/find_if/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/find_if/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::find_if(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::find_if_not(
|
||||
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::not_fn(cuda::equal_to_value{val})));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
52
cccl_upstream/libcudacxx/benchmarks/bench/for_each/basic.cu
Normal file
52
cccl_upstream/libcudacxx/benchmarks/bench/for_each/basic.cu
Normal file
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct square_t
|
||||
{
|
||||
__device__ void operator()(T& x) const
|
||||
{
|
||||
x = x * x;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in(elements, T{1});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
square_t<T> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::for_each(cuda_policy(alloc, launch), in.begin(), in.end(), op);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct square_t
|
||||
{
|
||||
__device__ void operator()(T& x) const
|
||||
{
|
||||
x = x * x;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in(elements, T{1});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
square_t<T> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::for_each_n(cuda_policy(alloc, launch), in.begin(), elements, op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
42
cccl_upstream/libcudacxx/benchmarks/bench/generate/basic.cu
Normal file
42
cccl_upstream/libcudacxx/benchmarks/bench/generate/basic.cu
Normal file
@@ -0,0 +1,42 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
struct generator
|
||||
{
|
||||
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
|
||||
{
|
||||
return 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::generate(cuda_policy(alloc, launch), output.begin(), output.end(), generator<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,42 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
struct generator
|
||||
{
|
||||
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
|
||||
{
|
||||
return 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::generate_n(cuda_policy(alloc, launch), output.begin(), elements, generator<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,94 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,69 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
85
cccl_upstream/libcudacxx/benchmarks/bench/is_heap/basic.cu
Normal file
85
cccl_upstream/libcudacxx/benchmarks/bench/is_heap/basic.cu
Normal file
@@ -0,0 +1,85 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/fill.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// All-zero is a valid heap; setting one element to 1 forces a violation at
|
||||
// that child index since its parent is still 0.
|
||||
template <typename T>
|
||||
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
|
||||
{
|
||||
thrust::fill(d.begin(), d.end(), T{0});
|
||||
if (violation_point >= 1 && violation_point < d.size())
|
||||
{
|
||||
d[violation_point] = T{1};
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,85 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/fill.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// All-zero is a valid heap; setting one element to 1 forces a violation at
|
||||
// that child index since its parent is still 0.
|
||||
template <typename T>
|
||||
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
|
||||
{
|
||||
thrust::fill(d.begin(), d.end(), T{0});
|
||||
if (violation_point >= 1 && violation_point < d.size())
|
||||
{
|
||||
d[violation_point] = T{1};
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
#include <thrust/sequence.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_partitioned(
|
||||
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), select_op_t{static_cast<T>(mismatch_point)}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
78
cccl_upstream/libcudacxx/benchmarks/bench/is_sorted/basic.cu
Normal file
78
cccl_upstream/libcudacxx/benchmarks/bench/is_sorted/basic.cu
Normal file
@@ -0,0 +1,78 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::greater<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,78 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,66 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/extrema.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
95
cccl_upstream/libcudacxx/benchmarks/bench/merge/basic.cu
Normal file
95
cccl_upstream/libcudacxx/benchmarks/bench/merge/basic.cu
Normal file
@@ -0,0 +1,95 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
#include <thrust/merge.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
|
||||
const auto entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
thrust::sort(in.begin(), in.begin() + elements_in_lhs);
|
||||
thrust::sort(in.begin() + elements_in_lhs, in.end());
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::merge(
|
||||
cuda_policy(alloc, launch),
|
||||
in.cbegin(),
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cend(),
|
||||
out.begin());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"})
|
||||
.add_int64_axis("InputSizeRatio", {25, 50, 75});
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
|
||||
const auto entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
thrust::sort(in.begin(), in.begin() + elements_in_lhs, ::cuda::std::greater<T>{});
|
||||
thrust::sort(in.begin() + elements_in_lhs, in.end(), ::cuda::std::greater<T>{});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::merge(
|
||||
cuda_policy(alloc, launch),
|
||||
in.cbegin(),
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cend(),
|
||||
out.begin(),
|
||||
::cuda::std::greater<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"})
|
||||
.add_int64_axis("InputSizeRatio", {25, 50, 75});
|
||||
@@ -0,0 +1,66 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/extrema.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
82
cccl_upstream/libcudacxx/benchmarks/bench/mismatch/basic.cu
Normal file
82
cccl_upstream/libcudacxx/benchmarks/bench/mismatch/basic.cu
Normal file
@@ -0,0 +1,82 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::mismatch(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_iter")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void range_range(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::mismatch(
|
||||
cuda_policy(alloc, launch),
|
||||
dinput.begin(),
|
||||
dinput.end(),
|
||||
cuda::constant_iterator<T>{0},
|
||||
cuda::constant_iterator<T>{0, elements}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_range")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/none_of/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/none_of/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::none_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/partition/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/partition/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
@@ -0,0 +1,55 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::partition_copy(
|
||||
cuda_policy(alloc, launch),
|
||||
input.begin(),
|
||||
input.end(),
|
||||
output.begin(),
|
||||
cuda::std::make_reverse_iterator(output.begin() + elements),
|
||||
select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
41
cccl_upstream/libcudacxx/benchmarks/bench/reduce/basic.cu
Normal file
41
cccl_upstream/libcudacxx/benchmarks/bench/reduce/basic.cu
Normal file
@@ -0,0 +1,41 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::reduce(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
43
cccl_upstream/libcudacxx/benchmarks/bench/remove/basic.cu
Normal file
43
cccl_upstream/libcudacxx/benchmarks/bench/remove/basic.cu
Normal file
@@ -0,0 +1,43 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/complex>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements - count);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::remove(cuda_policy(alloc, launch), in.begin(), in.end(), T{42});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,44 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/complex>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements - count);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::remove_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::remove_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
50
cccl_upstream/libcudacxx/benchmarks/bench/remove_if/basic.cu
Normal file
50
cccl_upstream/libcudacxx/benchmarks/bench/remove_if/basic.cu
Normal file
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::remove_if(cuda_policy(alloc, launch), in.begin(), in.end(), is_even{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
41
cccl_upstream/libcudacxx/benchmarks/bench/replace/basic.cu
Normal file
41
cccl_upstream/libcudacxx/benchmarks/bench/replace/basic.cu
Normal file
@@ -0,0 +1,41 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::replace(cuda_policy(alloc, launch), in.begin(), in.end(), 42, 1337);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::replace_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), 42, 1337));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == static_cast<T>(42);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::replace_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), equal_to_42{}, 1337));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == static_cast<T>(42);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::replace_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}, 1337);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
42
cccl_upstream/libcudacxx/benchmarks/bench/reverse/basic.cu
Normal file
42
cccl_upstream/libcudacxx/benchmarks/bench/reverse/basic.cu
Normal file
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/reverse.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::reverse(cuda_policy(alloc, launch), in.begin(), in.end());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,43 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/reverse.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::reverse_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
44
cccl_upstream/libcudacxx/benchmarks/bench/rotate/basic.cu
Normal file
44
cccl_upstream/libcudacxx/benchmarks/bench/rotate/basic.cu
Normal file
@@ -0,0 +1,44 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("MidpointAt");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::rotate(cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -0,0 +1,45 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("MidpointAt");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::rotate_copy(
|
||||
cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -0,0 +1,43 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("ShiftedTo");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements - midpoint);
|
||||
state.add_global_memory_writes<T>(elements - midpoint);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::shift_left(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("ShiftedTo");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements - midpoint);
|
||||
state.add_global_memory_writes<T>(elements - midpoint);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::shift_right(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.6, 0.45, 0.01});
|
||||
81
cccl_upstream/libcudacxx/benchmarks/bench/sort/basic.cu
Normal file
81
cccl_upstream/libcudacxx/benchmarks/bench/sort/basic.cu
Normal file
@@ -0,0 +1,81 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"});
|
||||
|
||||
struct fake_less
|
||||
{
|
||||
template <class T, class U>
|
||||
[[nodiscard]] _CCCL_API constexpr bool operator()(const T& t, const U& u) const
|
||||
{
|
||||
// complex is not less than comparable, so just compare the first element
|
||||
if constexpr (cuda::std::__is_cpp17_less_than_comparable_v<T, U>)
|
||||
{
|
||||
return t < u;
|
||||
}
|
||||
else
|
||||
{
|
||||
return cuda::std::get<0>(t) < cuda::std::get<0>(u);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end(), fake_less{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(all_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"});
|
||||
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of CUDA Experimental in CUDA C++ Core Libraries,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::stable_partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
@@ -0,0 +1,72 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/swap.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in1 = generate(elements);
|
||||
thrust::device_vector<T> in2 = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<T>(2 * elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::swap_ranges(cuda_policy(alloc, launch), in1.begin(), in1.end(), in2.begin());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_iter_swap(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in1 = generate(elements);
|
||||
thrust::device_vector<T> in2 = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<T>(2 * elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::swap_ranges(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::std::reverse_iterator{in1.end()},
|
||||
cuda::std::reverse_iterator{in1.begin()},
|
||||
cuda::std::reverse_iterator{in2.end()});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_iter_swap, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_iter_swap")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,156 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
#include <thrust/iterator/zip_iterator.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
// The benchmarks are inspired by the BabelStream thrust version:
|
||||
// https://github.com/UoB-HPC/BabelStream/blob/main/src/thrust/ThrustStream.cu
|
||||
|
||||
// Modified from BabelStream to also work for integers
|
||||
constexpr auto startA = 1; // BabelStream: 0.1
|
||||
constexpr auto startB = 2; // BabelStream: 0.2
|
||||
constexpr auto startC = 3; // BabelStream: 0.1
|
||||
constexpr auto startScalar = 4; // BabelStream: 0.4
|
||||
|
||||
using element_types = nvbench::type_list<std::int8_t, std::int16_t, float, double, __int128>;
|
||||
// Different benchmarks use a different number of buffers. H200/B200 can fit 2^31 elements for all benchmarks and types.
|
||||
// Upstream BabelStream uses 2^25. Allocation failure just skips the benchmark
|
||||
auto array_size_powers = std::vector<std::int64_t>{25, 31};
|
||||
|
||||
template <typename T>
|
||||
static void mul(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch), c.begin(), c.end(), b.begin(), [=] _CCCL_HOST_DEVICE(const T& ci) {
|
||||
return ci * scalar;
|
||||
}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(mul, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("mul")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void add(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(2 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch), a.begin(), a.end(), b.begin(), c.begin(), cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(add, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("add")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void triad(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(2 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch),
|
||||
b.begin(),
|
||||
b.end(),
|
||||
c.begin(),
|
||||
a.begin(),
|
||||
[=] _CCCL_HOST_DEVICE(const T& bi, const T& ci) {
|
||||
return bi + scalar * ci;
|
||||
}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(triad, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("triad")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void nstream(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(3 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::make_zip_iterator(a.begin(), b.begin(), c.begin()),
|
||||
cuda::make_zip_iterator(a.end(), b.end(), c.end()),
|
||||
a.begin(),
|
||||
cuda::zip_function{[=] _CCCL_HOST_DEVICE(const T& ai, const T& bi, const T& ci) {
|
||||
return ai + bi + scalar * ci;
|
||||
}}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(nstream, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("nstream")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
75
cccl_upstream/libcudacxx/benchmarks/bench/transform/fib.cu
Normal file
75
cccl_upstream/libcudacxx/benchmarks/bench/transform/fib.cu
Normal file
@@ -0,0 +1,75 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
template <class InT, class OutT>
|
||||
struct fib_t
|
||||
{
|
||||
__device__ OutT operator()(InT n)
|
||||
{
|
||||
OutT t1 = 0;
|
||||
OutT t2 = 1;
|
||||
|
||||
if (n <= 1)
|
||||
{
|
||||
return t1;
|
||||
}
|
||||
else if (n == 2)
|
||||
{
|
||||
return t2;
|
||||
}
|
||||
for (InT i = 3; i <= n; ++i)
|
||||
{
|
||||
const auto next = t1 + t2;
|
||||
t1 = t2;
|
||||
t2 = next;
|
||||
}
|
||||
|
||||
return t2;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void fib(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> input = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<nvbench::uint32_t>(elements);
|
||||
|
||||
fib_t<T, nvbench::uint32_t> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::transform(cuda_policy(alloc, launch), input.cbegin(), input.cend(), output.begin(), op));
|
||||
});
|
||||
}
|
||||
|
||||
using types = nvbench::type_list<nvbench::uint32_t, nvbench::uint64_t>;
|
||||
|
||||
NVBENCH_BENCH_TYPES(fib, NVBENCH_TYPE_AXES(types))
|
||||
.set_name("fib")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,53 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/transform_scan.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct times_two
|
||||
{
|
||||
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
|
||||
{
|
||||
return 2 * val;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_exclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, cuda::std::plus<T>{}, times_two<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,79 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/transform_scan.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct times_two
|
||||
{
|
||||
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
|
||||
{
|
||||
return 2 * val;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("basic")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,49 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void binary(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_reduce(
|
||||
cuda_policy(alloc, launch),
|
||||
in.begin(),
|
||||
in.end(),
|
||||
cuda::constant_iterator<int>{42},
|
||||
42,
|
||||
cuda::std::plus<T>{},
|
||||
cuda::std::multiplies<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(binary, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct plus_one
|
||||
{
|
||||
template <class U>
|
||||
[[nodiscard]] __device__ constexpr T operator()(const U val) const noexcept
|
||||
{
|
||||
return static_cast<T>(val + 1);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void unary(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_reduce(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), 42, cuda::std::plus<T>{}, plus_one<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(unary, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
88
cccl_upstream/libcudacxx/benchmarks/bench/unique/basic.cu
Normal file
88
cccl_upstream/libcudacxx/benchmarks/bench/unique/basic.cu
Normal file
@@ -0,0 +1,88 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/iterator/counting_iterator.h>
|
||||
#include <thrust/transform.h>
|
||||
#include <thrust/unique.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream_ref>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
|
||||
template <typename T>
|
||||
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
|
||||
{
|
||||
in.resize(elements);
|
||||
thrust::transform(
|
||||
thrust::counting_iterator<std::size_t>(0),
|
||||
thrust::counting_iterator<std::size_t>(elements),
|
||||
in.begin(),
|
||||
[] __device__(std::size_t i) {
|
||||
// This seems like a clang-tidy bug. Yes we end up converting to double, but the division
|
||||
// is done entirely in integer land...
|
||||
return static_cast<T>(i / 2ULL); // NOLINT(bugprone-integer-division)
|
||||
});
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end(), cuda::std::equal_to<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,90 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/iterator/counting_iterator.h>
|
||||
#include <thrust/transform.h>
|
||||
#include <thrust/unique.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream_ref>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
|
||||
template <typename T>
|
||||
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
|
||||
{
|
||||
in.resize(elements);
|
||||
thrust::transform(
|
||||
thrust::counting_iterator<std::size_t>(0),
|
||||
thrust::counting_iterator<std::size_t>(elements),
|
||||
in.begin(),
|
||||
[] __device__(std::size_t i) {
|
||||
const auto run = i / 2;
|
||||
return static_cast<T>(run);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique_copy writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique_copy writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique_copy(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::equal_to<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
2558
cccl_upstream/libcudacxx/cmake/AddLLVM.cmake
Normal file
2558
cccl_upstream/libcudacxx/cmake/AddLLVM.cmake
Normal file
File diff suppressed because it is too large
Load Diff
11
cccl_upstream/libcudacxx/cmake/DetermineGCCCompatible.cmake
Normal file
11
cccl_upstream/libcudacxx/cmake/DetermineGCCCompatible.cmake
Normal file
@@ -0,0 +1,11 @@
|
||||
# Determine if the compiler has GCC-compatible command-line syntax.
|
||||
|
||||
if (NOT DEFINED LLVM_COMPILER_IS_GCC_COMPATIBLE)
|
||||
if (CMAKE_COMPILER_IS_GNUCXX)
|
||||
set(LLVM_COMPILER_IS_GCC_COMPATIBLE ON)
|
||||
elseif (MSVC)
|
||||
set(LLVM_COMPILER_IS_GCC_COMPATIBLE OFF)
|
||||
elseif ("${CMAKE_CXX_COMPILER_ID}" MATCHES "Clang")
|
||||
set(LLVM_COMPILER_IS_GCC_COMPATIBLE ON)
|
||||
endif()
|
||||
endif()
|
||||
35
cccl_upstream/libcudacxx/cmake/GetHostTriple.cmake
Normal file
35
cccl_upstream/libcudacxx/cmake/GetHostTriple.cmake
Normal file
@@ -0,0 +1,35 @@
|
||||
# Returns the host triple.
|
||||
# Invokes config.guess
|
||||
|
||||
function(get_host_triple var)
|
||||
if (MSVC)
|
||||
if (CMAKE_SIZEOF_VOID_P EQUAL 8)
|
||||
set(value "x86_64-pc-windows-msvc")
|
||||
else()
|
||||
set(value "i686-pc-windows-msvc")
|
||||
endif()
|
||||
elseif (MINGW AND NOT MSYS)
|
||||
if (CMAKE_SIZEOF_VOID_P EQUAL 8)
|
||||
set(value "x86_64-w64-windows-gnu")
|
||||
else()
|
||||
set(value "i686-pc-windows-gnu")
|
||||
endif()
|
||||
else(MSVC)
|
||||
if (CMAKE_HOST_SYSTEM_NAME STREQUAL Windows AND NOT MSYS)
|
||||
message(WARNING "unable to determine host target triple")
|
||||
else()
|
||||
set(config_guess ${LLVM_PATH}/cmake/config.guess)
|
||||
execute_process(
|
||||
COMMAND sh ${config_guess}
|
||||
RESULT_VARIABLE TT_RV
|
||||
OUTPUT_VARIABLE TT_OUT
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
if (NOT TT_RV EQUAL 0)
|
||||
message(FATAL_ERROR "Failed to execute ${config_guess}")
|
||||
endif(NOT TT_RV EQUAL 0)
|
||||
set(value ${TT_OUT})
|
||||
endif()
|
||||
endif(MSVC)
|
||||
set(${var} ${value} PARENT_SCOPE)
|
||||
endfunction(get_host_triple var)
|
||||
376
cccl_upstream/libcudacxx/cmake/LLVM-Config.cmake
Normal file
376
cccl_upstream/libcudacxx/cmake/LLVM-Config.cmake
Normal file
@@ -0,0 +1,376 @@
|
||||
function(get_system_libs return_var)
|
||||
message(AUTHOR_WARNING "get_system_libs no longer needed")
|
||||
set(${return_var} "" PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
function(link_system_libs target)
|
||||
message(AUTHOR_WARNING "link_system_libs no longer needed")
|
||||
endfunction()
|
||||
|
||||
# is_llvm_target_library(
|
||||
# library
|
||||
# Name of the LLVM library to check
|
||||
# return_var
|
||||
# Output variable name
|
||||
# ALL_TARGETS;INCLUDED_TARGETS;OMITTED_TARGETS
|
||||
# ALL_TARGETS - default looks at the full list of known targets
|
||||
# INCLUDED_TARGETS - looks only at targets being configured
|
||||
# OMITTED_TARGETS - looks only at targets that are not being configured
|
||||
# )
|
||||
function(is_llvm_target_library library return_var)
|
||||
cmake_parse_arguments(
|
||||
ARG
|
||||
"ALL_TARGETS;INCLUDED_TARGETS;OMITTED_TARGETS"
|
||||
""
|
||||
""
|
||||
${ARGN}
|
||||
)
|
||||
# Sets variable `return_var' to ON if `library' corresponds to a
|
||||
# LLVM supported target. To OFF if it doesn't.
|
||||
set(${return_var} OFF PARENT_SCOPE)
|
||||
string(TOUPPER "${library}" capitalized_lib)
|
||||
if (ARG_INCLUDED_TARGETS)
|
||||
string(TOUPPER "${LLVM_TARGETS_TO_BUILD}" targets)
|
||||
elseif (ARG_OMITTED_TARGETS)
|
||||
set(omitted_targets ${LLVM_ALL_TARGETS})
|
||||
list(REMOVE_ITEM omitted_targets ${LLVM_TARGETS_TO_BUILD})
|
||||
string(TOUPPER "${omitted_targets}" targets)
|
||||
else()
|
||||
string(TOUPPER "${LLVM_ALL_TARGETS}" targets)
|
||||
endif()
|
||||
foreach (t ${targets})
|
||||
if (
|
||||
capitalized_lib STREQUAL t
|
||||
OR capitalized_lib STREQUAL "${t}"
|
||||
OR capitalized_lib STREQUAL "${t}DESC"
|
||||
OR capitalized_lib STREQUAL "${t}CODEGEN"
|
||||
OR capitalized_lib STREQUAL "${t}ASMPARSER"
|
||||
OR capitalized_lib STREQUAL "${t}ASMPRINTER"
|
||||
OR capitalized_lib STREQUAL "${t}DISASSEMBLER"
|
||||
OR capitalized_lib STREQUAL "${t}INFO"
|
||||
OR capitalized_lib STREQUAL "${t}UTILS"
|
||||
)
|
||||
set(${return_var} ON PARENT_SCOPE)
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
endfunction(is_llvm_target_library)
|
||||
|
||||
function(is_llvm_target_specifier library return_var)
|
||||
is_llvm_target_library(${library} ${return_var} ${ARGN})
|
||||
string(TOUPPER "${library}" capitalized_lib)
|
||||
if (NOT ${return_var})
|
||||
if (
|
||||
capitalized_lib STREQUAL "ALLTARGETSASMPARSERS"
|
||||
OR capitalized_lib STREQUAL "ALLTARGETSDESCS"
|
||||
OR capitalized_lib STREQUAL "ALLTARGETSDISASSEMBLERS"
|
||||
OR capitalized_lib STREQUAL "ALLTARGETSINFOS"
|
||||
OR capitalized_lib STREQUAL "NATIVE"
|
||||
OR capitalized_lib STREQUAL "NATIVECODEGEN"
|
||||
)
|
||||
set(${return_var} ON PARENT_SCOPE)
|
||||
endif()
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
macro(llvm_config executable)
|
||||
cmake_parse_arguments(ARG "USE_SHARED" "" "" ${ARGN})
|
||||
set(link_components ${ARG_UNPARSED_ARGUMENTS})
|
||||
|
||||
if (ARG_USE_SHARED)
|
||||
# If USE_SHARED is specified, then we link against libLLVM,
|
||||
# but also against the component libraries below. This is
|
||||
# done in case libLLVM does not contain all of the components
|
||||
# the target requires.
|
||||
#
|
||||
# Strip LLVM_DYLIB_COMPONENTS out of link_components.
|
||||
# To do this, we need special handling for "all", since that
|
||||
# may imply linking to libraries that are not included in
|
||||
# libLLVM.
|
||||
|
||||
if (DEFINED link_components AND DEFINED LLVM_DYLIB_COMPONENTS)
|
||||
if ("${LLVM_DYLIB_COMPONENTS}" STREQUAL "all")
|
||||
set(link_components "")
|
||||
else()
|
||||
list(REMOVE_ITEM link_components ${LLVM_DYLIB_COMPONENTS})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
target_link_libraries(${executable} PRIVATE LLVM)
|
||||
endif()
|
||||
|
||||
explicit_llvm_config(${executable} ${link_components})
|
||||
endmacro(llvm_config)
|
||||
|
||||
function(explicit_llvm_config executable)
|
||||
set(link_components ${ARGN})
|
||||
|
||||
llvm_map_components_to_libnames(LIBRARIES ${link_components})
|
||||
get_target_property(t ${executable} TYPE)
|
||||
if (t STREQUAL "STATIC_LIBRARY")
|
||||
target_link_libraries(${executable} INTERFACE ${LIBRARIES})
|
||||
elseif (
|
||||
t STREQUAL "EXECUTABLE"
|
||||
OR t STREQUAL "SHARED_LIBRARY"
|
||||
OR t STREQUAL "MODULE_LIBRARY"
|
||||
)
|
||||
target_link_libraries(${executable} PRIVATE ${LIBRARIES})
|
||||
else()
|
||||
# Use plain form for legacy user.
|
||||
target_link_libraries(${executable} ${LIBRARIES})
|
||||
endif()
|
||||
endfunction(explicit_llvm_config)
|
||||
|
||||
# This is Deprecated
|
||||
function(llvm_map_components_to_libraries OUT_VAR)
|
||||
message(
|
||||
AUTHOR_WARNING
|
||||
"Using llvm_map_components_to_libraries() is deprecated. Use llvm_map_components_to_libnames() instead"
|
||||
)
|
||||
explicit_map_components_to_libraries(result ${ARGN})
|
||||
set(${OUT_VAR} ${result} ${sys_result} PARENT_SCOPE)
|
||||
endfunction(llvm_map_components_to_libraries)
|
||||
|
||||
# Expand pseudo-components into real components.
|
||||
# Does not cover 'native', 'backend', or 'engine' as these require special
|
||||
# handling. Also does not cover 'all' as we only have a list of the libnames
|
||||
# available and not a list of the components.
|
||||
function(llvm_expand_pseudo_components out_components)
|
||||
set(link_components ${ARGN})
|
||||
foreach (c ${link_components})
|
||||
# add codegen, asmprinter, asmparser, disassembler
|
||||
list(FIND LLVM_TARGETS_TO_BUILD ${c} idx)
|
||||
if (NOT idx LESS 0)
|
||||
if (TARGET LLVM${c}CodeGen)
|
||||
list(APPEND expanded_components "${c}CodeGen")
|
||||
else()
|
||||
if (TARGET LLVM${c})
|
||||
list(APPEND expanded_components "${c}")
|
||||
else()
|
||||
message(FATAL_ERROR "Target ${c} is not in the set of libraries.")
|
||||
endif()
|
||||
endif()
|
||||
if (TARGET LLVM${c}AsmPrinter)
|
||||
list(APPEND expanded_components "${c}AsmPrinter")
|
||||
endif()
|
||||
if (TARGET LLVM${c}AsmParser)
|
||||
list(APPEND expanded_components "${c}AsmParser")
|
||||
endif()
|
||||
if (TARGET LLVM${c}Desc)
|
||||
list(APPEND expanded_components "${c}Desc")
|
||||
endif()
|
||||
if (TARGET LLVM${c}Disassembler)
|
||||
list(APPEND expanded_components "${c}Disassembler")
|
||||
endif()
|
||||
if (TARGET LLVM${c}Info)
|
||||
list(APPEND expanded_components "${c}Info")
|
||||
endif()
|
||||
if (TARGET LLVM${c}Utils)
|
||||
list(APPEND expanded_components "${c}Utils")
|
||||
endif()
|
||||
elseif (c STREQUAL "nativecodegen")
|
||||
if (TARGET LLVM${LLVM_NATIVE_ARCH}CodeGen)
|
||||
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}CodeGen")
|
||||
endif()
|
||||
if (TARGET LLVM${LLVM_NATIVE_ARCH}Desc)
|
||||
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}Desc")
|
||||
endif()
|
||||
if (TARGET LLVM${LLVM_NATIVE_ARCH}Info)
|
||||
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}Info")
|
||||
endif()
|
||||
elseif (c STREQUAL "AllTargetsCodeGens")
|
||||
# Link all the codegens from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}CodeGen)
|
||||
list(APPEND expanded_components "${t}CodeGen")
|
||||
endif()
|
||||
endforeach(t)
|
||||
elseif (c STREQUAL "AllTargetsAsmParsers")
|
||||
# Link all the asm parsers from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}AsmParser)
|
||||
list(APPEND expanded_components "${t}AsmParser")
|
||||
endif()
|
||||
endforeach(t)
|
||||
elseif (c STREQUAL "AllTargetsDescs")
|
||||
# Link all the descs from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}Desc)
|
||||
list(APPEND expanded_components "${t}Desc")
|
||||
endif()
|
||||
endforeach(t)
|
||||
elseif (c STREQUAL "AllTargetsDisassemblers")
|
||||
# Link all the disassemblers from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}Disassembler)
|
||||
list(APPEND expanded_components "${t}Disassembler")
|
||||
endif()
|
||||
endforeach(t)
|
||||
elseif (c STREQUAL "AllTargetsInfos")
|
||||
# Link all the infos from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}Info)
|
||||
list(APPEND expanded_components "${t}Info")
|
||||
endif()
|
||||
endforeach(t)
|
||||
else()
|
||||
list(APPEND expanded_components "${c}")
|
||||
endif()
|
||||
endforeach()
|
||||
set(${out_components} ${expanded_components} PARENT_SCOPE)
|
||||
endfunction(llvm_expand_pseudo_components out_components)
|
||||
|
||||
# This is a variant intended for the final user:
|
||||
# Map LINK_COMPONENTS to actual libnames.
|
||||
function(llvm_map_components_to_libnames out_libs)
|
||||
set(link_components ${ARGN})
|
||||
if (NOT LLVM_AVAILABLE_LIBS)
|
||||
# Inside LLVM itself available libs are in a global property.
|
||||
get_property(LLVM_AVAILABLE_LIBS GLOBAL PROPERTY LLVM_LIBS)
|
||||
endif()
|
||||
string(TOUPPER "${LLVM_AVAILABLE_LIBS}" capitalized_libs)
|
||||
|
||||
get_property(LLVM_TARGETS_CONFIGURED GLOBAL PROPERTY LLVM_TARGETS_CONFIGURED)
|
||||
|
||||
# Generally in our build system we avoid order-dependence. Unfortunately since
|
||||
# not all targets create the same set of libraries we actually need to ensure
|
||||
# that all build targets associated with a target are added before we can
|
||||
# process target dependencies.
|
||||
if (NOT LLVM_TARGETS_CONFIGURED)
|
||||
foreach (c ${link_components})
|
||||
is_llvm_target_specifier(${c} iltl_result ALL_TARGETS)
|
||||
if (iltl_result)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Specified target library before target registration is complete."
|
||||
)
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
# Expand some keywords:
|
||||
list(FIND LLVM_TARGETS_TO_BUILD "${LLVM_NATIVE_ARCH}" have_native_backend)
|
||||
list(FIND link_components "engine" engine_required)
|
||||
if (NOT engine_required EQUAL -1)
|
||||
list(FIND LLVM_TARGETS_WITH_JIT "${LLVM_NATIVE_ARCH}" have_jit)
|
||||
if (NOT have_native_backend EQUAL -1 AND NOT have_jit EQUAL -1)
|
||||
list(APPEND link_components "jit")
|
||||
list(APPEND link_components "native")
|
||||
else()
|
||||
list(APPEND link_components "interpreter")
|
||||
endif()
|
||||
endif()
|
||||
list(FIND link_components "native" native_required)
|
||||
if (NOT native_required EQUAL -1)
|
||||
if (NOT have_native_backend EQUAL -1)
|
||||
list(APPEND link_components ${LLVM_NATIVE_ARCH})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Translate symbolic component names to real libraries:
|
||||
llvm_expand_pseudo_components(link_components ${link_components})
|
||||
foreach (c ${link_components})
|
||||
get_property(c_rename GLOBAL PROPERTY LLVM_COMPONENT_NAME_${c})
|
||||
if (c_rename)
|
||||
set(c ${c_rename})
|
||||
endif()
|
||||
if (c STREQUAL "native")
|
||||
# already processed
|
||||
elseif (c STREQUAL "backend")
|
||||
# same case as in `native'.
|
||||
elseif (c STREQUAL "engine")
|
||||
# already processed
|
||||
elseif (c STREQUAL "all")
|
||||
get_property(all_components GLOBAL PROPERTY LLVM_COMPONENT_LIBS)
|
||||
list(APPEND expanded_components ${all_components})
|
||||
else()
|
||||
# Canonize the component name:
|
||||
string(TOUPPER "${c}" capitalized)
|
||||
list(FIND capitalized_libs LLVM${capitalized} lib_idx)
|
||||
if (lib_idx LESS 0)
|
||||
# The component is unknown. Maybe is an omitted target?
|
||||
is_llvm_target_library(${c} iltl_result OMITTED_TARGETS)
|
||||
if (iltl_result)
|
||||
# A missing library to a directly referenced omitted target would be bad.
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Library '${c}' is a direct reference to a target library for an omitted target."
|
||||
)
|
||||
else()
|
||||
# If it is not an omitted target we should assume it is a component
|
||||
# that hasn't yet been processed by CMake. Missing components will
|
||||
# cause errors later in the configuration, so we can safely assume
|
||||
# that this is valid here.
|
||||
list(APPEND expanded_components LLVM${c})
|
||||
endif()
|
||||
else(lib_idx LESS 0)
|
||||
list(GET LLVM_AVAILABLE_LIBS ${lib_idx} canonical_lib)
|
||||
list(APPEND expanded_components ${canonical_lib})
|
||||
endif(lib_idx LESS 0)
|
||||
endif(c STREQUAL "native")
|
||||
endforeach(c)
|
||||
|
||||
set(${out_libs} ${expanded_components} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
# Perform a post-order traversal of the dependency graph.
|
||||
# This duplicates the algorithm used by llvm-config, originally
|
||||
# in tools/llvm-config/llvm-config.cpp, function ComputeLibsForComponents.
|
||||
function(expand_topologically name required_libs visited_libs)
|
||||
list(FIND visited_libs ${name} found)
|
||||
if (found LESS 0)
|
||||
list(APPEND visited_libs ${name})
|
||||
set(visited_libs ${visited_libs} PARENT_SCOPE)
|
||||
|
||||
#
|
||||
get_property(libname GLOBAL PROPERTY LLVM_COMPONENT_NAME_${name})
|
||||
if (libname)
|
||||
set(cname LLVM${libname})
|
||||
elseif (TARGET ${name})
|
||||
set(cname ${name})
|
||||
elseif (TARGET LLVM${name})
|
||||
set(cname LLVM${name})
|
||||
else()
|
||||
message(FATAL_ERROR "unknown component ${name}")
|
||||
endif()
|
||||
|
||||
get_property(lib_deps TARGET ${cname} PROPERTY LLVM_LINK_COMPONENTS)
|
||||
foreach (lib_dep ${lib_deps})
|
||||
expand_topologically(${lib_dep} "${required_libs}" "${visited_libs}")
|
||||
set(required_libs ${required_libs} PARENT_SCOPE)
|
||||
set(visited_libs ${visited_libs} PARENT_SCOPE)
|
||||
endforeach()
|
||||
|
||||
list(APPEND required_libs ${cname})
|
||||
set(required_libs ${required_libs} PARENT_SCOPE)
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
# Expand dependencies while topologically sorting the list of libraries:
|
||||
function(llvm_expand_dependencies out_libs)
|
||||
set(expanded_components ${ARGN})
|
||||
|
||||
set(required_libs)
|
||||
set(visited_libs)
|
||||
foreach (lib ${expanded_components})
|
||||
expand_topologically(${lib} "${required_libs}" "${visited_libs}")
|
||||
endforeach()
|
||||
|
||||
if (required_libs)
|
||||
list(REVERSE required_libs)
|
||||
endif()
|
||||
set(${out_libs} ${required_libs} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
function(explicit_map_components_to_libraries out_libs)
|
||||
llvm_map_components_to_libnames(link_libs ${ARGN})
|
||||
llvm_expand_dependencies(expanded_components ${link_libs})
|
||||
# Return just the libraries included in this build:
|
||||
set(result)
|
||||
foreach (c ${expanded_components})
|
||||
if (TARGET ${c})
|
||||
set(result ${result} ${c})
|
||||
endif()
|
||||
endforeach(c)
|
||||
set(${out_libs} ${result} PARENT_SCOPE)
|
||||
endfunction(explicit_map_components_to_libraries)
|
||||
129
cccl_upstream/libcudacxx/cmake/LLVMProcessSources.cmake
Normal file
129
cccl_upstream/libcudacxx/cmake/LLVMProcessSources.cmake
Normal file
@@ -0,0 +1,129 @@
|
||||
include(AddFileDependencies)
|
||||
include(CMakeParseArguments)
|
||||
|
||||
function(llvm_replace_compiler_option var old new)
|
||||
# Replaces a compiler option or switch `old' in `var' by `new'.
|
||||
# If `old' is not in `var', appends `new' to `var'.
|
||||
# Example: llvm_replace_compiler_option(CMAKE_CXX_FLAGS_RELEASE "-O3" "-O2")
|
||||
# If the option already is on the variable, don't add it:
|
||||
if ("${${var}}" MATCHES "(^| )${new}($| )")
|
||||
set(n "")
|
||||
else()
|
||||
set(n "${new}")
|
||||
endif()
|
||||
if ("${${var}}" MATCHES "(^| )${old}($| )")
|
||||
string(REGEX REPLACE "(^| )${old}($| )" " ${n} " ${var} "${${var}}")
|
||||
else()
|
||||
set(${var} "${${var}} ${n}")
|
||||
endif()
|
||||
set(${var} "${${var}}" PARENT_SCOPE)
|
||||
endfunction(llvm_replace_compiler_option)
|
||||
|
||||
macro(add_td_sources srcs)
|
||||
file(GLOB tds *.td)
|
||||
if (tds)
|
||||
source_group("TableGen descriptions" FILES ${tds})
|
||||
set_source_files_properties(${tds} PROPERTIES HEADER_FILE_ONLY ON)
|
||||
list(APPEND ${srcs} ${tds})
|
||||
endif()
|
||||
endmacro(add_td_sources)
|
||||
|
||||
function(add_header_files_for_glob hdrs_out glob)
|
||||
file(GLOB hds ${glob})
|
||||
set(filtered)
|
||||
foreach (file ${hds})
|
||||
# Explicit existence check is necessary to filter dangling symlinks
|
||||
# out. See https://bugs.gentoo.org/674662.
|
||||
if (EXISTS ${file})
|
||||
list(APPEND filtered ${file})
|
||||
endif()
|
||||
endforeach()
|
||||
set(${hdrs_out} ${filtered} PARENT_SCOPE)
|
||||
endfunction(add_header_files_for_glob)
|
||||
|
||||
function(find_all_header_files hdrs_out additional_headerdirs)
|
||||
add_header_files_for_glob(hds *.h)
|
||||
list(APPEND all_headers ${hds})
|
||||
|
||||
foreach (additional_dir ${additional_headerdirs})
|
||||
add_header_files_for_glob(hds "${additional_dir}/*.h")
|
||||
list(APPEND all_headers ${hds})
|
||||
add_header_files_for_glob(hds "${additional_dir}/*.inc")
|
||||
list(APPEND all_headers ${hds})
|
||||
endforeach(additional_dir)
|
||||
|
||||
set(${hdrs_out} ${all_headers} PARENT_SCOPE)
|
||||
endfunction(find_all_header_files)
|
||||
|
||||
function(llvm_process_sources OUT_VAR)
|
||||
cmake_parse_arguments(
|
||||
ARG
|
||||
"PARTIAL_SOURCES_INTENDED"
|
||||
""
|
||||
"ADDITIONAL_HEADERS;ADDITIONAL_HEADER_DIRS"
|
||||
${ARGN}
|
||||
)
|
||||
set(sources ${ARG_UNPARSED_ARGUMENTS})
|
||||
if (NOT ARG_PARTIAL_SOURCES_INTENDED)
|
||||
llvm_check_source_file_list(${sources})
|
||||
endif()
|
||||
|
||||
# This adds .td and .h files to the Visual Studio solution:
|
||||
add_td_sources(sources)
|
||||
find_all_header_files(hdrs "${ARG_ADDITIONAL_HEADER_DIRS}")
|
||||
if (hdrs)
|
||||
set_source_files_properties(${hdrs} PROPERTIES HEADER_FILE_ONLY ON)
|
||||
endif()
|
||||
set_source_files_properties(
|
||||
${ARG_ADDITIONAL_HEADERS}
|
||||
PROPERTIES HEADER_FILE_ONLY ON
|
||||
)
|
||||
list(APPEND sources ${ARG_ADDITIONAL_HEADERS} ${hdrs})
|
||||
|
||||
set(${OUT_VAR} ${sources} PARENT_SCOPE)
|
||||
endfunction(llvm_process_sources)
|
||||
|
||||
function(llvm_check_source_file_list)
|
||||
cmake_parse_arguments(ARG "" "SOURCE_DIR" "" ${ARGN})
|
||||
foreach (l ${ARG_UNPARSED_ARGUMENTS})
|
||||
get_filename_component(fp ${l} REALPATH)
|
||||
list(APPEND listed ${fp})
|
||||
endforeach()
|
||||
|
||||
if (ARG_SOURCE_DIR)
|
||||
file(GLOB globbed "${ARG_SOURCE_DIR}/*.c" "${ARG_SOURCE_DIR}/*.cpp")
|
||||
else()
|
||||
file(GLOB globbed *.c *.cpp)
|
||||
endif()
|
||||
|
||||
foreach (g ${globbed})
|
||||
get_filename_component(fn ${g} NAME)
|
||||
if (ARG_SOURCE_DIR)
|
||||
set(entry "${g}")
|
||||
else()
|
||||
set(entry "${fn}")
|
||||
endif()
|
||||
get_filename_component(gp ${g} REALPATH)
|
||||
|
||||
# Don't reject hidden files. Some editors create backups in the
|
||||
# same directory as the file.
|
||||
if (NOT "${fn}" MATCHES "^\\.")
|
||||
list(FIND LLVM_OPTIONAL_SOURCES ${entry} idx)
|
||||
if (idx LESS 0)
|
||||
list(FIND listed ${gp} idx)
|
||||
if (idx LESS 0)
|
||||
if (ARG_SOURCE_DIR)
|
||||
set(fn_relative "${ARG_SOURCE_DIR}/${fn}")
|
||||
else()
|
||||
set(fn_relative "${fn}")
|
||||
endif()
|
||||
message(
|
||||
SEND_ERROR
|
||||
"Found unknown source file ${fn_relative}
|
||||
Please update ${CMAKE_CURRENT_LIST_FILE}\n"
|
||||
)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
endforeach()
|
||||
endfunction(llvm_check_source_file_list)
|
||||
@@ -0,0 +1,53 @@
|
||||
# This file defines the `libcudacxx_build_compiler_targets()` function, which
|
||||
# creates the following interface targets:
|
||||
#
|
||||
# libcudacxx.compiler_interface
|
||||
# - Interface target linked into all targets in the libcudacxx developer build.
|
||||
# Defines common warning flags, definitions, etc, including those defined in
|
||||
# the global CCCL targets.
|
||||
|
||||
cccl_get_libcudacxx()
|
||||
|
||||
function(libcudacxx_build_compiler_targets)
|
||||
set(cuda_compile_options)
|
||||
set(cxx_compile_options)
|
||||
set(cxx_compile_definitions)
|
||||
|
||||
# if (CCCL_USE_LIBCXX)
|
||||
# list(APPEND cxx_compile_options "-stdlib=libc++")
|
||||
# list(APPEND cxx_compile_definitions "_ALLOW_UNSUPPORTED_LIBCPP=1")
|
||||
# endif()
|
||||
|
||||
# Set test specific flags
|
||||
list(APPEND cxx_compile_definitions "CCCL_ENABLE_ASSERTIONS")
|
||||
list(APPEND cxx_compile_definitions "CCCL_IGNORE_DEPRECATED_CPP_DIALECT")
|
||||
list(
|
||||
APPEND cxx_compile_definitions
|
||||
"CCCL_IGNORE_DEPRECATED_DISCARD_MEMORY_HEADER"
|
||||
)
|
||||
list(
|
||||
APPEND cxx_compile_definitions
|
||||
"CCCL_IGNORE_DEPRECATED_STREAM_REF_HEADER"
|
||||
)
|
||||
|
||||
if (CCCL_ENABLE_TILE)
|
||||
list(APPEND cuda_compile_options "--enable-tile")
|
||||
endif()
|
||||
|
||||
cccl_build_compiler_interface(
|
||||
libcudacxx.compiler_flags
|
||||
"${cuda_compile_options}"
|
||||
"${cxx_compile_options}"
|
||||
"${cxx_compile_definitions}"
|
||||
)
|
||||
|
||||
add_library(libcudacxx.compiler_interface INTERFACE)
|
||||
target_link_libraries(
|
||||
libcudacxx.compiler_interface
|
||||
INTERFACE
|
||||
# order matters here, we need the libcudacxx options to override the cccl options.
|
||||
cccl.compiler_interface
|
||||
libcudacxx.compiler_flags
|
||||
libcudacxx::libcudacxx
|
||||
)
|
||||
endfunction()
|
||||
@@ -0,0 +1,125 @@
|
||||
# For every public header, build a translation unit containing `#include <header>`
|
||||
# to let the compiler try to figure out warnings in that header if it is not otherwise
|
||||
# included in tests, and also to verify if the headers are modular enough.
|
||||
# .inl files are not globbed for, because they are not supposed to be used as public
|
||||
# entrypoints.
|
||||
|
||||
cccl_get_cudatoolkit()
|
||||
|
||||
# Meta target for all configs' header builds:
|
||||
add_custom_target(libcudacxx.test.internal_headers)
|
||||
|
||||
# Grep all internal headers
|
||||
file(
|
||||
GLOB_RECURSE internal_headers
|
||||
RELATIVE "${libcudacxx_SOURCE_DIR}/include/"
|
||||
CONFIGURE_DEPENDS
|
||||
${libcudacxx_SOURCE_DIR}/include/cuda/__*/*.h
|
||||
${libcudacxx_SOURCE_DIR}/include/cuda/std/__*/*.h
|
||||
)
|
||||
|
||||
# Exclude <cuda/std/__cccl/(prologue|epilogue|visibility).h> from the test
|
||||
list(
|
||||
FILTER internal_headers
|
||||
EXCLUDE
|
||||
REGEX "__cccl/(prologue|epilogue|visibility)\.h"
|
||||
)
|
||||
|
||||
# headers in `__cuda` are meant to come after the related "cuda" headers so they do not compile on their own
|
||||
list(FILTER internal_headers EXCLUDE REGEX "__cuda/*")
|
||||
|
||||
# generated cuda::ptx headers are not standalone
|
||||
list(FILTER internal_headers EXCLUDE REGEX "__ptx/instructions/generated")
|
||||
|
||||
# don't check nvtx3.h - it's not our header
|
||||
list(FILTER internal_headers EXCLUDE REGEX ".*/__nvtx/nvtx3.h")
|
||||
|
||||
function(libcudacxx_add_internal_header_test_target target_name)
|
||||
if (NOT ARGN)
|
||||
return()
|
||||
endif()
|
||||
|
||||
cccl_generate_header_tests(
|
||||
${target_name}
|
||||
libcudacxx/include
|
||||
NO_METATARGETS
|
||||
LANGUAGE CUDA
|
||||
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
|
||||
HEADERS ${ARGN}
|
||||
)
|
||||
|
||||
target_compile_definitions(${target_name} PRIVATE _CCCL_HEADER_TEST)
|
||||
target_link_libraries(
|
||||
${target_name}
|
||||
PUBLIC #
|
||||
libcudacxx.compiler_interface
|
||||
CUDA::cudart
|
||||
)
|
||||
add_dependencies(libcudacxx.test.internal_headers ${target_name})
|
||||
endfunction()
|
||||
|
||||
libcudacxx_add_internal_header_test_target(
|
||||
libcudacxx.test.internal_headers.base
|
||||
${internal_headers}
|
||||
)
|
||||
|
||||
# We have fallbacks for some type traits that we want to explicitly test so that they do not bitrot.
|
||||
set(internal_headers_fallback)
|
||||
set(internal_headers_fallback_per_header_defines)
|
||||
foreach (header IN LISTS internal_headers)
|
||||
# MSVC cannot handle some of the fallbacks.
|
||||
if ("MSVC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
|
||||
if (
|
||||
"${header}" MATCHES "is_base_of"
|
||||
OR "${header}" MATCHES "is_nothrow_destructible"
|
||||
OR "${header}" MATCHES "is_polymorphic"
|
||||
)
|
||||
continue()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
file(READ "${libcudacxx_SOURCE_DIR}/include/${header}" header_file)
|
||||
string(REGEX MATCH "_LIBCUDACXX_[A-Z_]*_FALLBACK" fallback "${header_file}")
|
||||
if (fallback)
|
||||
list(APPEND internal_headers_fallback "${header}")
|
||||
string(
|
||||
REGEX REPLACE
|
||||
"([][+.*^$()|?\\\\])"
|
||||
"\\\\\\1"
|
||||
header_regex
|
||||
"${header}"
|
||||
)
|
||||
list(
|
||||
APPEND internal_headers_fallback_per_header_defines
|
||||
DEFINE
|
||||
"${fallback}"
|
||||
"^${header_regex}$"
|
||||
)
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if (internal_headers_fallback)
|
||||
cccl_generate_header_tests(
|
||||
libcudacxx.test.internal_headers.fallback
|
||||
libcudacxx/include
|
||||
NO_METATARGETS
|
||||
LANGUAGE CUDA
|
||||
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
|
||||
HEADERS ${internal_headers_fallback}
|
||||
PER_HEADER_DEFINES ${internal_headers_fallback_per_header_defines}
|
||||
)
|
||||
target_compile_definitions(
|
||||
libcudacxx.test.internal_headers.fallback
|
||||
PRIVATE _CCCL_HEADER_TEST
|
||||
)
|
||||
target_link_libraries(
|
||||
libcudacxx.test.internal_headers.fallback
|
||||
PUBLIC #
|
||||
libcudacxx.compiler_interface
|
||||
CUDA::cudart
|
||||
)
|
||||
add_dependencies(
|
||||
libcudacxx.test.internal_headers
|
||||
libcudacxx.test.internal_headers.fallback
|
||||
)
|
||||
endif()
|
||||
@@ -0,0 +1,47 @@
|
||||
# For every public header, build a translation unit containing `#include <header>`
|
||||
# to let the compiler try to figure out warnings in that header if it is not otherwise
|
||||
# included in tests, and also to verify if the headers are modular enough.
|
||||
# .inl files are not globbed for, because they are not supposed to be used as public
|
||||
# entrypoints.
|
||||
|
||||
# Meta target for all configs' header builds:
|
||||
add_custom_target(libcudacxx.test.public_headers)
|
||||
|
||||
# Grep all public headers
|
||||
file(
|
||||
GLOB public_headers
|
||||
LIST_DIRECTORIES false
|
||||
RELATIVE "${libcudacxx_SOURCE_DIR}/include"
|
||||
CONFIGURE_DEPENDS
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/*"
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/std/*"
|
||||
)
|
||||
|
||||
# annotated_ptr does not work with clang cuda due to __nv_associate_access_property
|
||||
if ("Clang" STREQUAL "${CMAKE_CUDA_COMPILER_ID}")
|
||||
list(REMOVE_ITEM public_headers "annotated_ptr")
|
||||
endif()
|
||||
|
||||
function(libcudacxx_add_public_header_test_target target_name)
|
||||
if (NOT ARGN)
|
||||
return()
|
||||
endif()
|
||||
|
||||
cccl_generate_header_tests(
|
||||
${target_name}
|
||||
libcudacxx/include
|
||||
NO_METATARGETS
|
||||
LANGUAGE CUDA
|
||||
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
|
||||
HEADERS ${ARGN}
|
||||
)
|
||||
|
||||
target_compile_definitions(${target_name} PRIVATE _CCCL_HEADER_TEST)
|
||||
target_link_libraries(${target_name} PUBLIC libcudacxx.compiler_interface)
|
||||
add_dependencies(libcudacxx.test.public_headers ${target_name})
|
||||
endfunction()
|
||||
|
||||
libcudacxx_add_public_header_test_target(
|
||||
libcudacxx.test.public_headers.base
|
||||
${public_headers}
|
||||
)
|
||||
@@ -0,0 +1,75 @@
|
||||
# For every public header, build a translation unit containing `#include <header>`
|
||||
# to let the compiler try to figure out warnings in that header if it is not otherwise
|
||||
# included in tests, and also to verify if the headers are modular enough.
|
||||
# .inl files are not globbed for, because they are not supposed to be used as public
|
||||
# entrypoints.
|
||||
|
||||
cccl_get_cudatoolkit()
|
||||
|
||||
# Meta target for all configs' header builds:
|
||||
add_custom_target(libcudacxx.test.public_headers_host_only)
|
||||
add_custom_target(libcudacxx.test.public_headers_host_only_with_ctk)
|
||||
|
||||
if (CCCL_ENABLE_TILE) # TODO(miscco): For now only test public headers with tile
|
||||
return()
|
||||
endif()
|
||||
|
||||
# Grep all public headers
|
||||
file(
|
||||
GLOB public_headers_host_only
|
||||
LIST_DIRECTORIES false
|
||||
RELATIVE "${libcudacxx_SOURCE_DIR}/include"
|
||||
CONFIGURE_DEPENDS
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/*"
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/std/*"
|
||||
)
|
||||
|
||||
set(public_host_header_cxx_compile_options)
|
||||
set(public_host_header_cxx_compile_definitions)
|
||||
|
||||
# Specifically add libc++ testing if requested to the libcudacxx host suite
|
||||
if (CCCL_USE_LIBCXX)
|
||||
list(APPEND public_host_header_cxx_compile_options "-stdlib=libc++")
|
||||
endif()
|
||||
|
||||
function(
|
||||
libcudacxx_add_public_header_test_host_target
|
||||
target_name
|
||||
parent_target
|
||||
with_ctk
|
||||
)
|
||||
cccl_generate_header_tests(
|
||||
${target_name}
|
||||
libcudacxx/include
|
||||
NO_METATARGETS
|
||||
LANGUAGE CXX
|
||||
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
|
||||
HEADERS ${public_headers_host_only}
|
||||
)
|
||||
target_compile_definitions(
|
||||
${target_name}
|
||||
PRIVATE #
|
||||
${public_host_header_cxx_compile_definitions}
|
||||
_CCCL_HEADER_TEST
|
||||
)
|
||||
target_compile_options(
|
||||
${target_name}
|
||||
PRIVATE ${public_host_header_cxx_compile_options}
|
||||
)
|
||||
target_link_libraries(${target_name} PUBLIC libcudacxx.compiler_interface)
|
||||
if (with_ctk)
|
||||
target_link_libraries(${target_name} PUBLIC CUDA::cudart)
|
||||
endif()
|
||||
add_dependencies(${parent_target} ${target_name})
|
||||
endfunction()
|
||||
|
||||
libcudacxx_add_public_header_test_host_target(
|
||||
libcudacxx.test.public_headers_host_only.base
|
||||
libcudacxx.test.public_headers_host_only
|
||||
OFF
|
||||
)
|
||||
libcudacxx_add_public_header_test_host_target(
|
||||
libcudacxx.test.public_headers_host_only_with_ctk.base
|
||||
libcudacxx.test.public_headers_host_only_with_ctk
|
||||
ON
|
||||
)
|
||||
1569
cccl_upstream/libcudacxx/cmake/config.guess
vendored
Executable file
1569
cccl_upstream/libcudacxx/cmake/config.guess
vendored
Executable file
File diff suppressed because it is too large
Load Diff
23
cccl_upstream/libcudacxx/cmake/header_test.cpp.in
Normal file
23
cccl_upstream/libcudacxx/cmake/header_test.cpp.in
Normal file
@@ -0,0 +1,23 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// ignore deprecation warnings
|
||||
#if defined(__clang__)
|
||||
# pragma clang diagnostic ignored "-Wdeprecated"
|
||||
# pragma clang diagnostic ignored "-Wdeprecated-declarations"
|
||||
#elif defined(_MSC_VER)
|
||||
# pragma warning (disable: 4996)
|
||||
#else
|
||||
# pragma GCC diagnostic ignored "-Wdeprecated"
|
||||
# pragma GCC diagnostic ignored "-Wdeprecated-declarations"
|
||||
#endif
|
||||
|
||||
// This file tests that the respective header is includable on its own with a cuda compiler
|
||||
#include <@header@>
|
||||
1
cccl_upstream/libcudacxx/cmake/libcudacxxAddSubdir.cmake
Normal file
1
cccl_upstream/libcudacxx/cmake/libcudacxxAddSubdir.cmake
Normal file
@@ -0,0 +1 @@
|
||||
cccl_add_subdir_helper(libcudacxx)
|
||||
1
cccl_upstream/libcudacxx/codegen/.gitignore
vendored
Normal file
1
cccl_upstream/libcudacxx/codegen/.gitignore
vendored
Normal file
@@ -0,0 +1 @@
|
||||
build
|
||||
51
cccl_upstream/libcudacxx/codegen/CMakeLists.txt
Normal file
51
cccl_upstream/libcudacxx/codegen/CMakeLists.txt
Normal file
@@ -0,0 +1,51 @@
|
||||
## Codegen adds the following build targets
|
||||
# libcudacxx.atomics.codegen
|
||||
# libcudacxx.atomics.codegen.install
|
||||
## Test targets:
|
||||
# libcudacxx.test.atomics.codegen.diff
|
||||
|
||||
add_executable(codegen EXCLUDE_FROM_ALL codegen.cpp)
|
||||
|
||||
target_compile_features(codegen PRIVATE cxx_std_20)
|
||||
|
||||
set(
|
||||
atomic_generated_output
|
||||
"${libcudacxx_BINARY_DIR}/codegen/cuda_ptx_generated.h"
|
||||
)
|
||||
set(
|
||||
atomic_install_location
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/std/__atomic/functions"
|
||||
)
|
||||
|
||||
add_custom_target(
|
||||
libcudacxx.atomics.codegen
|
||||
COMMAND codegen "${atomic_generated_output}"
|
||||
BYPRODUCTS "${atomic_generated_output}"
|
||||
)
|
||||
|
||||
add_custom_target(
|
||||
libcudacxx.atomics.codegen.install
|
||||
# gersemi: off
|
||||
COMMAND
|
||||
"${CMAKE_COMMAND}" -E copy
|
||||
"${atomic_generated_output}"
|
||||
"${atomic_install_location}/cuda_ptx_generated.h"
|
||||
# gersemi: on
|
||||
DEPENDS libcudacxx.atomics.codegen
|
||||
BYPRODUCTS "${atomic_install_location}/cuda_ptx_generated.h"
|
||||
)
|
||||
|
||||
add_test(
|
||||
NAME libcudacxx.test.atomics.codegen.diff
|
||||
# gersemi: off
|
||||
COMMAND
|
||||
"${CMAKE_COMMAND}" -E compare_files
|
||||
"${atomic_install_location}/cuda_ptx_generated.h"
|
||||
"${atomic_generated_output}"
|
||||
# gersemi: on
|
||||
)
|
||||
|
||||
set_tests_properties(
|
||||
libcudacxx.test.atomics.codegen.diff
|
||||
PROPERTIES REQUIRED_FILES "${atomic_generated_output}"
|
||||
)
|
||||
164
cccl_upstream/libcudacxx/codegen/add_ptx_instruction.py
Executable file
164
cccl_upstream/libcudacxx/codegen/add_ptx_instruction.py
Executable file
@@ -0,0 +1,164 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
##===----------------------------------------------------------------------===##
|
||||
##
|
||||
## Part of libcu++, the C++ Standard Library for your entire system,
|
||||
## under the Apache License v2.0 with LLVM Exceptions.
|
||||
## See https://llvm.org/LICENSE.txt for license information.
|
||||
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
##
|
||||
##===----------------------------------------------------------------------===##
|
||||
|
||||
import argparse
|
||||
import os
|
||||
|
||||
import cccl_paths
|
||||
|
||||
docs = os.path.join(cccl_paths.DOCS_LIBCUDACXX_DIR, "ptx", "instructions")
|
||||
test = os.path.join(cccl_paths.LIBCUDACXX_TEST_DIR, "libcudacxx", "cuda", "ptx")
|
||||
src = os.path.join(cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "__ptx", "instructions")
|
||||
ptx_header = os.path.join(cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "ptx")
|
||||
instr_docs = os.path.join(cccl_paths.DOCS_LIBCUDACXX_DIR, "ptx", "instructions.rst")
|
||||
|
||||
|
||||
def add_docs(ptx_instr, url):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
underbar = "=" * len(ptx_instr)
|
||||
|
||||
(docs / f"{cpp_instr}.rst").write_text(
|
||||
f""".. _libcudacxx-ptx-instructions-{ptx_instr.replace(".", "-")}:
|
||||
|
||||
{ptx_instr}
|
||||
{underbar}
|
||||
|
||||
- PTX ISA:
|
||||
`{ptx_instr} <{url}>`__
|
||||
|
||||
.. include:: generated/{cpp_instr}.rst
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def add_test(ptx_instr):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
dst = test / f"ptx.{ptx_instr}.compile.pass.cpp"
|
||||
dst.write_text(
|
||||
f"""//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
|
||||
// <cuda/ptx>
|
||||
|
||||
#include <cuda/ptx>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include "generated/{cpp_instr}.h"
|
||||
|
||||
int main(int, char**)
|
||||
{{
|
||||
return 0;
|
||||
}}
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def add_src(ptx_instr):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
(src / f"{cpp_instr}.h").write_text(
|
||||
f"""// -*- C++ -*-
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA_PTX_{cpp_instr.upper()}_H_
|
||||
#define _CUDA_PTX_{cpp_instr.upper()}_H_
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__ptx/ptx_dot_variants.h>
|
||||
#include <cuda/__ptx/ptx_helper_functions.h>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <nv/target> // __CUDA_MINIMUM_ARCH__ and friends
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_PTX
|
||||
|
||||
#include <cuda/__ptx/instructions/generated/{cpp_instr}.h>
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_PTX
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA_PTX_{cpp_instr.upper()}_H_
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def add_ptx_header_include(ptx_instr):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
txt = ptx_header.read_text()
|
||||
# just add as first new include. clang-format will sort it in
|
||||
idx = txt.index("#include <cuda/__ptx/instructions")
|
||||
txt = (
|
||||
txt[:idx]
|
||||
+ f"""#include <cuda/__ptx/instructions/{cpp_instr}.h>\n"""
|
||||
+ txt[idx:]
|
||||
)
|
||||
ptx_header.write_text(txt)
|
||||
|
||||
|
||||
def add_docs_include(ptx_instr):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
txt = instr_docs.read_text()
|
||||
# just add as first new include
|
||||
idx = txt.index(" instructions/")
|
||||
txt = txt[:idx] + f" instructions/{cpp_instr}\n" + txt[idx:]
|
||||
instr_docs.write_text(txt)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("ptx_instruction", type=str)
|
||||
parser.add_argument("url", type=str)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
ptx_instr = args.ptx_instruction
|
||||
url = args.url
|
||||
|
||||
# Enable using internal urls in the command-line, to be automatically converted to public URLs.
|
||||
if url.startswith("index.html"):
|
||||
url = url.replace(
|
||||
"index.html",
|
||||
"https://docs.nvidia.com/cuda/parallel-thread-execution/index.html",
|
||||
)
|
||||
|
||||
add_test(ptx_instr)
|
||||
add_docs(ptx_instr, url)
|
||||
add_src(ptx_instr)
|
||||
add_ptx_header_include(ptx_instr)
|
||||
add_docs_include(ptx_instr)
|
||||
21
cccl_upstream/libcudacxx/codegen/cccl_paths.py
Normal file
21
cccl_upstream/libcudacxx/codegen/cccl_paths.py
Normal file
@@ -0,0 +1,21 @@
|
||||
##===----------------------------------------------------------------------===##
|
||||
##
|
||||
## Part of libcu++, the C++ Standard Library for your entire system,
|
||||
## under the Apache License v2.0 with LLVM Exceptions.
|
||||
## See https://llvm.org/LICENSE.txt for license information.
|
||||
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
##
|
||||
##===----------------------------------------------------------------------===##
|
||||
|
||||
import os
|
||||
|
||||
LIBCUDACXX_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
LIBCUDACXX_CMAKE_DIR = os.path.join(LIBCUDACXX_DIR, "cmake")
|
||||
LIBCUDACXX_CODEGEN_DIR = os.path.join(LIBCUDACXX_DIR, "codegen")
|
||||
LIBCUDACXX_INCLUDE_DIR = os.path.join(LIBCUDACXX_DIR, "include")
|
||||
LIBCUDACXX_TEST_DIR = os.path.join(LIBCUDACXX_DIR, "test")
|
||||
|
||||
DOCS_DIR = os.path.dirname(LIBCUDACXX_DIR)
|
||||
DOCS_LIBCUDACXX_DIR = os.path.join(DOCS_DIR, "libcudacxx")
|
||||
45
cccl_upstream/libcudacxx/codegen/codegen.cpp
Normal file
45
cccl_upstream/libcudacxx/codegen/codegen.cpp
Normal file
@@ -0,0 +1,45 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <ostream>
|
||||
|
||||
#include "generators/compare_and_swap.h"
|
||||
#include "generators/exchange.h"
|
||||
#include "generators/fence.h"
|
||||
#include "generators/fetch_ops.h"
|
||||
#include "generators/header.h"
|
||||
#include "generators/ld_st.h"
|
||||
|
||||
using namespace std::string_literals;
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
std::fstream filestream;
|
||||
|
||||
if (argc == 2)
|
||||
{
|
||||
filestream.open(argv[1], filestream.out);
|
||||
}
|
||||
|
||||
std::ostream& stream = filestream.is_open() ? filestream : std::cout;
|
||||
|
||||
FormatHeader(stream);
|
||||
FormatFence(stream);
|
||||
FormatLoad(stream);
|
||||
FormatStore(stream);
|
||||
FormatCompareAndSwap(stream);
|
||||
FormatExchange(stream);
|
||||
FormatFetchOps(stream);
|
||||
FormatTail(stream);
|
||||
|
||||
return 0;
|
||||
}
|
||||
245
cccl_upstream/libcudacxx/codegen/generate_prologue_epilogue.py
Executable file
245
cccl_upstream/libcudacxx/codegen/generate_prologue_epilogue.py
Executable file
@@ -0,0 +1,245 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
##===----------------------------------------------------------------------===##
|
||||
##
|
||||
## Part of libcu++, the C++ Standard Library for your entire system,
|
||||
## under the Apache License v2.0 with LLVM Exceptions.
|
||||
## See https://llvm.org/LICENSE.txt for license information.
|
||||
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
##
|
||||
##===----------------------------------------------------------------------===##
|
||||
|
||||
import datetime
|
||||
import os
|
||||
|
||||
import cccl_paths
|
||||
|
||||
PROLOGUE_FILE = os.path.join(
|
||||
cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "std", "__cccl", "prologue.h"
|
||||
)
|
||||
EPILOGUE_FILE = os.path.join(
|
||||
cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "std", "__cccl", "epilogue.h"
|
||||
)
|
||||
|
||||
HEADER = f"""\
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) {datetime.datetime.now().year} NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// !!! DO NOT EDIT THIS FILE !!! This file is generated by utils/generate_prologue_epilogue.py.
|
||||
|
||||
// NO include guards here (this file is included multiple times)"""
|
||||
|
||||
FOOTER = """\
|
||||
// NO include guards here (this file is included multiple times)
|
||||
"""
|
||||
|
||||
PUSH_POP_MACROS = {
|
||||
"__declspec modifiers": [
|
||||
"align",
|
||||
"allocate",
|
||||
"allocator",
|
||||
"appdomain",
|
||||
"code_seg",
|
||||
"deprecated",
|
||||
"dllimport",
|
||||
"dllexport",
|
||||
"empty_bases",
|
||||
"hybrid_patchable",
|
||||
"jitintrinsic",
|
||||
"lifetimebound",
|
||||
"naked",
|
||||
"noalias",
|
||||
"noinline",
|
||||
"noreturn",
|
||||
"nothrow",
|
||||
"novtable",
|
||||
"no_sanitize_address",
|
||||
"process",
|
||||
"property",
|
||||
"restrict",
|
||||
"safebuffers",
|
||||
"selectany",
|
||||
"spectre",
|
||||
"thread",
|
||||
"uuid",
|
||||
],
|
||||
"[[msvc::attribute]] attributes": [
|
||||
"msvc",
|
||||
"flatten",
|
||||
"forceinline",
|
||||
"forceinline_calls",
|
||||
"intrinsic",
|
||||
"noinline",
|
||||
"noinline_calls",
|
||||
"no_tls_guard",
|
||||
],
|
||||
"Windows nasty macros": ["min", "max", "interface"],
|
||||
"sal.h on Windows": ["__valid", "__callback"],
|
||||
"other macros": ["clang"],
|
||||
"sys/sysmacros.h on linux": ["major", "minor", "makedev"],
|
||||
}
|
||||
|
||||
|
||||
def write_section(file, section):
|
||||
file.write(section)
|
||||
file.write("\n\n")
|
||||
|
||||
|
||||
def make_prologue(file):
|
||||
# Write common header.
|
||||
write_section(file, HEADER)
|
||||
|
||||
# Add prologue/epilogue include logic check.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
#if defined(_CCCL_PROLOGUE_INCLUDED)
|
||||
# error \\
|
||||
"cccl internal error: <cuda/std/__cccl/epilogue.h> must be included before next <cuda/std/__cccl/prologue.h> is reincluded"
|
||||
#endif
|
||||
#define _CCCL_PROLOGUE_INCLUDED() 1""",
|
||||
)
|
||||
|
||||
# Add necessary includes.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
#include <cuda/std/__cccl/compiler.h>
|
||||
#include <cuda/std/__cccl/diagnostic.h>
|
||||
#include <cuda/std/__cccl/dialect.h>""",
|
||||
)
|
||||
|
||||
# Add push macros.
|
||||
for group_name, macros in PUSH_POP_MACROS.items():
|
||||
write_section(file, f"// {group_name}")
|
||||
for macro in macros:
|
||||
write_section(
|
||||
file,
|
||||
f"""\
|
||||
#if defined({macro})
|
||||
# pragma push_macro("{macro}")
|
||||
# undef {macro}
|
||||
# define _CCCL_POP_MACRO_{macro}
|
||||
#endif // defined({macro})""",
|
||||
)
|
||||
|
||||
# Add warnings suppressions.
|
||||
write_section(
|
||||
file,
|
||||
'''\
|
||||
_CCCL_DIAG_PUSH
|
||||
_CCCL_NV_DIAG_PUSH()
|
||||
|
||||
// disable some msvc warnings
|
||||
// https://github.com/microsoft/STL/blob/master/stl/inc/yvals_core.h#L353
|
||||
// warning C4100: 'quack': unreferenced formal parameter
|
||||
// warning C4127: conditional expression is constant
|
||||
// warning C4180: qualifier applied to function type has no meaning; ignored
|
||||
// warning C4197: 'purr': top-level volatile in cast is ignored
|
||||
// warning C4324: 'roar': structure was padded due to alignment specifier
|
||||
// warning C4455: literal suffix identifiers that do not start with an underscore are reserved
|
||||
// warning C4503: 'hum': decorated name length exceeded, name was truncated
|
||||
// warning C4522: 'woof' : multiple assignment operators specified
|
||||
// warning C4668: 'meow' is not defined as a preprocessor macro, replacing with '0' for '#if/#elif'
|
||||
// warning C4800: 'boo': forcing value to bool 'true' or 'false' (performance warning)
|
||||
// warning C4996: 'meow': was declared deprecated
|
||||
_CCCL_DIAG_SUPPRESS_MSVC(4100 4127 4180 4197 4296 4324 4455 4503 4522 4668 4800 4996)
|
||||
|
||||
// Suppress compiler warnings about C++ extensions.
|
||||
|
||||
#if _CCCL_COMPILER(GCC, >=, 12)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wc++20-extensions")
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wc++23-extensions")
|
||||
#endif // _CCCL_COMPILER(GCC, >=, 12)
|
||||
#if _CCCL_COMPILER(GCC, >=, 14)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wc++26-extensions")
|
||||
#endif // _CCCL_COMPILER(GCC, >=, 14)
|
||||
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++20-extensions")
|
||||
#if _CCCL_COMPILER(CLANG, >=, 17)
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++23-extensions")
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++26-extensions")
|
||||
#else // ^^^ _CCCL_COMPILER(CLANG, >=, 17) ^^^ / vvv _CCCL_COMPILER(CLANG, <, 17) vvv
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++2b-extensions")
|
||||
#endif // ^^^ _CCCL_COMPILER(CLANG, <, 17) ^^^
|
||||
|
||||
// Suppress `if consteval`-related warnings.
|
||||
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(if_consteval_nonstandard)
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(is_constant_evaluated_in_nonconstexpr_context)
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(if_consteval_in_nonconstexpr_function)
|
||||
|
||||
_CCCL_DIAG_SUPPRESS_NVCC(3215) // "if consteval" and "if not consteval" are not standard in this mode
|
||||
_CCCL_DIAG_SUPPRESS_NVCC(3206) // "if consteval" and "if not consteval" are meaningless in a non-constexpr function
|
||||
_CCCL_DIAG_SUPPRESS_NVCC(3060) // call to __builtin_is_constant_evaluated appearing in a non-constexpr function always
|
||||
// produces "false"''',
|
||||
)
|
||||
|
||||
# Write the common footer.
|
||||
file.write(FOOTER)
|
||||
|
||||
|
||||
def make_epilogue(file):
|
||||
# Write common header.
|
||||
write_section(file, HEADER)
|
||||
|
||||
# Write includes.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
#include <cuda/std/__cccl/compiler.h>
|
||||
#include <cuda/std/__cccl/diagnostic.h>""",
|
||||
)
|
||||
|
||||
# Add prologue/epilogue include logic check.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
#if !defined(_CCCL_PROLOGUE_INCLUDED)
|
||||
# error "cccl internal error: <cuda/std/__cccl/prologue.h> must be included before <cuda/std/__cccl/epilogue.h>"
|
||||
#endif
|
||||
#undef _CCCL_PROLOGUE_INCLUDED""",
|
||||
)
|
||||
|
||||
# Pop warning suppressions.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
_CCCL_NV_DIAG_POP()
|
||||
_CCCL_DIAG_POP""",
|
||||
)
|
||||
|
||||
# Add pop macros.
|
||||
for group_name, macros in PUSH_POP_MACROS.items():
|
||||
write_section(file, f"// {group_name}")
|
||||
for macro in macros:
|
||||
write_section(
|
||||
file,
|
||||
f"""\
|
||||
#if defined({macro})
|
||||
# error \\
|
||||
"cccl internal error: macro `{macro}` was redefined between <cuda/std/__cccl/prologue.h> and <cuda/std/__cccl/epilogue.h>"
|
||||
#elif defined(_CCCL_POP_MACRO_{macro})
|
||||
# pragma pop_macro("{macro}")
|
||||
# undef _CCCL_POP_MACRO_{macro}
|
||||
#endif""",
|
||||
)
|
||||
|
||||
# Write the common footer.
|
||||
file.write(FOOTER)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
with open(PROLOGUE_FILE, "w") as file:
|
||||
make_prologue(file)
|
||||
|
||||
with open(EPILOGUE_FILE, "w") as file:
|
||||
make_epilogue(file)
|
||||
201
cccl_upstream/libcudacxx/codegen/generators/compare_and_swap.h
Normal file
201
cccl_upstream/libcudacxx/codegen/generators/compare_and_swap.h
Normal file
@@ -0,0 +1,201 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef COMPARED_AND_SWAP_H
|
||||
#define COMPARED_AND_SWAP_H
|
||||
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline void FormatCompareAndSwap(std::ostream& out)
|
||||
{
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE bool __cuda_atomic_compare_swap_memory_order_dispatch(_Fn& __cuda_cas, int __success_memorder, int __failure_memorder, _Sco) {
|
||||
bool __res = false;
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__stronger_order_cuda(__success_memorder, __failure_memorder)) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __res = __cuda_cas(__atomic_cuda_acquire{}); break;
|
||||
case __ATOMIC_ACQ_REL: __res = __cuda_cas(__atomic_cuda_acq_rel{}); break;
|
||||
case __ATOMIC_RELEASE: __res = __cuda_cas(__atomic_cuda_release{}); break;
|
||||
case __ATOMIC_RELAXED: __res = __cuda_cas(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__stronger_order_cuda(__success_memorder, __failure_memorder)) {
|
||||
case __ATOMIC_SEQ_CST: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __res = __cuda_cas(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __res = __cuda_cas(__atomic_cuda_volatile{}); break;
|
||||
case __ATOMIC_RELAXED: __res = __cuda_cas(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
return __res;
|
||||
}
|
||||
)XXX";
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Operand Type
|
||||
// 1 - Operand Size
|
||||
// 2 - Type Constraint
|
||||
// 3 - Memory Order
|
||||
// 4 - Memory Order function tag
|
||||
// 5 - Scope Constraint
|
||||
// 6 - Scope function tag
|
||||
constexpr auto asm_intrinsic_format_128 = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE bool __cuda_atomic_compare_exchange(
|
||||
_Type* __ptr, _Type& __dst, _Type __cmp, _Type __op, {4}, __atomic_cuda_operand_{0}{1}, {6})
|
||||
{{
|
||||
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b CAS is not supported until PTX ISA version 840");
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_90, (),
|
||||
NV_ANY_TARGET, (__atomic_cas_128b_unsupported_before_SM_90();)
|
||||
)
|
||||
asm volatile(R"YYY(
|
||||
{{
|
||||
.reg .b128 _d;
|
||||
.reg .b128 _v;
|
||||
mov.b128 _d, {{%3, %4}};
|
||||
mov.b128 _v, {{%5, %6}};
|
||||
atom.cas{3}{5}.b128 _d,[%2],_d,_v;
|
||||
mov.b128 {{%0, %1}}, _d;
|
||||
}}
|
||||
)YYY" : "=l"(__dst.__x),"=l"(__dst.__y) : "l"(__ptr), "l"(__cmp.__x),"l"(__cmp.__y), "l"(__op.__x),"l"(__op.__y) : "memory"); return __dst.__x == __cmp.__x && __dst.__y == __cmp.__y; }})XXX";
|
||||
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE bool __cuda_atomic_compare_exchange(
|
||||
_Type* __ptr, _Type& __dst, _Type __cmp, _Type __op, {4}, __atomic_cuda_operand_{0}{1}, {6})
|
||||
{{ asm volatile("atom.cas{3}{5}.{0}{1} %0,[%1],%2,%3;" : "={2}"(__dst) : "l"(__ptr), "{2}"(__cmp), "{2}"(__op) : "memory"); return __dst == __cmp; }})XXX";
|
||||
|
||||
constexpr Operand supported_types[] = {
|
||||
Operand::Bit,
|
||||
};
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
32,
|
||||
64,
|
||||
128,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Acquire,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Release,
|
||||
Semantic::Acq_Rel,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
for (auto type : supported_types)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
if (size == 2 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (size == 128 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if (size == 128)
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format_128,
|
||||
operand(type),
|
||||
size,
|
||||
constraints(type, size),
|
||||
semantic(sem),
|
||||
semantic_tag(sem),
|
||||
scope(sco),
|
||||
scope_tag(sco));
|
||||
}
|
||||
else
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
operand(type),
|
||||
size,
|
||||
constraints(type, size),
|
||||
semantic(sem),
|
||||
semantic_tag(sem),
|
||||
scope(sco),
|
||||
scope_tag(sco));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco>
|
||||
struct __cuda_atomic_bind_compare_exchange {
|
||||
_Type* __ptr;
|
||||
_Type* __exp;
|
||||
_Type* __des;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE bool operator()(_Atomic_Memorder) {
|
||||
return __cuda_atomic_compare_exchange(__ptr, *__exp, *__exp, *__des, _Atomic_Memorder{}, _Tag{}, _Sco{});
|
||||
}
|
||||
};
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE bool __atomic_compare_exchange_cuda(_Type* __ptr, _Type* __exp, _Type __des, bool, int __success_memorder, int __failure_memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
|
||||
__proxy_t* __exp_proxy = reinterpret_cast<__proxy_t*>(__exp);
|
||||
__proxy_t* __des_proxy = reinterpret_cast<__proxy_t*>(&__des);
|
||||
bool __res = false;
|
||||
if (__cuda_compare_exchange_weak_if_local(__ptr_proxy, __exp_proxy, __des_proxy, &__res)) {return __res;}
|
||||
__cuda_atomic_bind_compare_exchange<__proxy_t, __proxy_tag, _Sco> __bound_compare_swap{__ptr_proxy, __exp_proxy, __des_proxy};
|
||||
return __cuda_atomic_compare_swap_memory_order_dispatch(__bound_compare_swap, __success_memorder, __failure_memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE bool __atomic_compare_exchange_cuda(_Type volatile* __ptr, _Type* __exp, _Type __des, bool, int __success_memorder, int __failure_memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __exp_proxy = reinterpret_cast<__proxy_t*>(__exp);
|
||||
__proxy_t* __des_proxy = reinterpret_cast<__proxy_t*>(&__des);
|
||||
bool __res = false;
|
||||
if (__cuda_compare_exchange_weak_if_local(__ptr_proxy, __exp_proxy, __des_proxy, &__res)) {return __res;}
|
||||
__cuda_atomic_bind_compare_exchange<__proxy_t, __proxy_tag, _Sco> __bound_compare_swap{__ptr_proxy, __exp_proxy, __des_proxy};
|
||||
return __cuda_atomic_compare_swap_memory_order_dispatch(__bound_compare_swap, __success_memorder, __failure_memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // COMPARED_AND_SWAP_H
|
||||
192
cccl_upstream/libcudacxx/codegen/generators/definitions.h
Normal file
192
cccl_upstream/libcudacxx/codegen/generators/definitions.h
Normal file
@@ -0,0 +1,192 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef DEFINITIONS_H
|
||||
#define DEFINITIONS_H
|
||||
|
||||
#include <format>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <type_traits>
|
||||
#include <vector>
|
||||
|
||||
enum class Mmio
|
||||
{
|
||||
Disabled,
|
||||
Enabled,
|
||||
};
|
||||
|
||||
inline std::string mmio(Mmio m)
|
||||
{
|
||||
static const char* mmio_map[]{
|
||||
"",
|
||||
".mmio",
|
||||
};
|
||||
return mmio_map[std::underlying_type_t<Mmio>(m)];
|
||||
}
|
||||
|
||||
inline std::string mmio_tag(Mmio m)
|
||||
{
|
||||
static const char* mmio_map[]{
|
||||
"__atomic_cuda_mmio_disable",
|
||||
"__atomic_cuda_mmio_enable",
|
||||
};
|
||||
return mmio_map[std::underlying_type_t<Mmio>(m)];
|
||||
}
|
||||
|
||||
enum class Operand
|
||||
{
|
||||
Floating,
|
||||
Unsigned,
|
||||
Signed,
|
||||
Bit,
|
||||
};
|
||||
|
||||
inline std::string operand(Operand op)
|
||||
{
|
||||
static std::map op_map = {
|
||||
std::pair{Operand::Floating, "f"},
|
||||
std::pair{Operand::Unsigned, "u"},
|
||||
std::pair{Operand::Signed, "s"},
|
||||
std::pair{Operand::Bit, "b"},
|
||||
};
|
||||
return op_map[op];
|
||||
}
|
||||
|
||||
inline std::string operand_proxy_type(Operand op, size_t sz)
|
||||
{
|
||||
if (op == Operand::Floating)
|
||||
{
|
||||
if (sz == 32)
|
||||
{
|
||||
return {"float"};
|
||||
}
|
||||
else
|
||||
{
|
||||
return {"double"};
|
||||
}
|
||||
}
|
||||
else if (op == Operand::Signed)
|
||||
{
|
||||
return std::format("int{}_t", sz);
|
||||
}
|
||||
// Binary and unsigned can be the same proxy_type
|
||||
return std::format("uint{}_t", sz);
|
||||
}
|
||||
|
||||
inline std::string constraints(Operand op, size_t sz)
|
||||
{
|
||||
static std::map constraint_map = {
|
||||
std::pair{32,
|
||||
std::map{
|
||||
std::pair{Operand::Bit, "r"},
|
||||
std::pair{Operand::Unsigned, "r"},
|
||||
std::pair{Operand::Signed, "r"},
|
||||
std::pair{Operand::Floating, "f"},
|
||||
}},
|
||||
std::pair{64,
|
||||
std::map{
|
||||
std::pair{Operand::Bit, "l"},
|
||||
std::pair{Operand::Unsigned, "l"},
|
||||
std::pair{Operand::Signed, "l"},
|
||||
std::pair{Operand::Floating, "d"},
|
||||
}},
|
||||
std::pair{128,
|
||||
std::map{
|
||||
std::pair{Operand::Bit, "l"},
|
||||
std::pair{Operand::Unsigned, "l"},
|
||||
std::pair{Operand::Signed, "l"},
|
||||
std::pair{Operand::Floating, "d"},
|
||||
}},
|
||||
};
|
||||
|
||||
if (sz == 16)
|
||||
{
|
||||
return {"h"};
|
||||
}
|
||||
else
|
||||
{
|
||||
return constraint_map[sz][op];
|
||||
}
|
||||
}
|
||||
|
||||
enum class Semantic
|
||||
{
|
||||
Relaxed,
|
||||
Release,
|
||||
Acquire,
|
||||
Acq_Rel,
|
||||
Seq_Cst,
|
||||
Volatile,
|
||||
};
|
||||
|
||||
inline std::string semantic(Semantic sem)
|
||||
{
|
||||
static std::map sem_map = {
|
||||
std::pair{Semantic::Relaxed, ".relaxed"},
|
||||
std::pair{Semantic::Release, ".release"},
|
||||
std::pair{Semantic::Acquire, ".acquire"},
|
||||
std::pair{Semantic::Acq_Rel, ".acq_rel"},
|
||||
std::pair{Semantic::Seq_Cst, ".sc"},
|
||||
std::pair{Semantic::Volatile, ""},
|
||||
};
|
||||
return sem_map[sem];
|
||||
}
|
||||
|
||||
inline std::string semantic_tag(Semantic sem)
|
||||
{
|
||||
static std::map sem_map = {
|
||||
std::pair{Semantic::Relaxed, "__atomic_cuda_relaxed"},
|
||||
std::pair{Semantic::Release, "__atomic_cuda_release"},
|
||||
std::pair{Semantic::Acquire, "__atomic_cuda_acquire"},
|
||||
std::pair{Semantic::Acq_Rel, "__atomic_cuda_acq_rel"},
|
||||
std::pair{Semantic::Seq_Cst, "__atomic_cuda_seq_cst"},
|
||||
std::pair{Semantic::Volatile, "__atomic_cuda_volatile"},
|
||||
};
|
||||
return sem_map[sem];
|
||||
}
|
||||
|
||||
enum class Scope
|
||||
{
|
||||
Thread,
|
||||
Warp,
|
||||
CTA,
|
||||
Cluster,
|
||||
GPU,
|
||||
System,
|
||||
};
|
||||
|
||||
inline std::string scope(Scope sco)
|
||||
{
|
||||
static std::map sco_map = {
|
||||
std::pair{Scope::Thread, ""},
|
||||
std::pair{Scope::Warp, ""},
|
||||
std::pair{Scope::CTA, ".cta"},
|
||||
std::pair{Scope::Cluster, ".cluster"},
|
||||
std::pair{Scope::GPU, ".gpu"},
|
||||
std::pair{Scope::System, ".sys"},
|
||||
};
|
||||
return sco_map[sco];
|
||||
}
|
||||
|
||||
inline std::string scope_tag(Scope sco)
|
||||
{
|
||||
static std::map sco_map = {
|
||||
std::pair{Scope::Thread, "__thread_scope_thread_tag"},
|
||||
std::pair{Scope::Warp, ""},
|
||||
std::pair{Scope::CTA, "__thread_scope_block_tag"},
|
||||
std::pair{Scope::Cluster, "__thread_scope_cluster_tag"},
|
||||
std::pair{Scope::GPU, "__thread_scope_device_tag"},
|
||||
std::pair{Scope::System, "__thread_scope_system_tag"},
|
||||
};
|
||||
return sco_map[sco];
|
||||
}
|
||||
|
||||
#endif // DEFINITIONS_H
|
||||
197
cccl_upstream/libcudacxx/codegen/generators/exchange.h
Normal file
197
cccl_upstream/libcudacxx/codegen/generators/exchange.h
Normal file
@@ -0,0 +1,197 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef EXCHANGE_H
|
||||
#define EXCHANGE_H
|
||||
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline void FormatExchange(std::ostream& out)
|
||||
{
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_exchange_memory_order_dispatch(_Fn& __cuda_exch, int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_exch(__atomic_cuda_acquire{}); break;
|
||||
case __ATOMIC_ACQ_REL: __cuda_exch(__atomic_cuda_acq_rel{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_exch(__atomic_cuda_release{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_exch(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_exch(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __cuda_exch(__atomic_cuda_volatile{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_exch(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Operand Type
|
||||
// 1 - Operand Size
|
||||
// 2 - Type Constraint
|
||||
// 3 - Memory Order
|
||||
// 4 - Memory Order function tag
|
||||
// 5 - Scope Constraint
|
||||
// 6 - Scope function tag
|
||||
constexpr auto asm_intrinsic_format_128 = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_exchange(
|
||||
_Type* __ptr, _Type& __old, _Type __new, {4}, __atomic_cuda_operand_{0}{1}, {6})
|
||||
{{
|
||||
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b exchange is not supported until PTX ISA version 840");
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_90, (),
|
||||
NV_ANY_TARGET, (__atomic_exchange_128b_unsupported_before_SM_90();)
|
||||
)
|
||||
asm volatile(R"YYY(
|
||||
{{
|
||||
.reg .b128 _d;
|
||||
.reg .b128 _v;
|
||||
mov.b128 _v, {{%3, %4}};
|
||||
atom.exch{3}{5}.b128 _d,[%2],_v;
|
||||
mov.b128 {{%0, %1}}, _d;
|
||||
}}
|
||||
)YYY" : "=l"(__old.__x),"=l"(__old.__y) : "l"(__ptr), "l"(__new.__x),"l"(__new.__y) : "memory");
|
||||
}})XXX";
|
||||
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_exchange(
|
||||
_Type* __ptr, _Type& __old, _Type __new, {4}, __atomic_cuda_operand_{0}{1}, {6})
|
||||
{{ asm volatile("atom.exch{3}{5}.{0}{1} %0,[%1],%2;" : "={2}"(__old) : "l"(__ptr), "{2}"(__new) : "memory"); }})XXX";
|
||||
|
||||
constexpr Operand supported_types[] = {
|
||||
Operand::Bit,
|
||||
};
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
32,
|
||||
64,
|
||||
128,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Acquire,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Release,
|
||||
Semantic::Acq_Rel,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
for (auto type : supported_types)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
if (size == 2 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (size == 128 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if (size == 128)
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format_128,
|
||||
operand(type),
|
||||
size,
|
||||
constraints(type, size),
|
||||
semantic(sem),
|
||||
semantic_tag(sem),
|
||||
scope(sco),
|
||||
scope_tag(sco));
|
||||
}
|
||||
else
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
operand(type),
|
||||
size,
|
||||
constraints(type, size),
|
||||
semantic(sem),
|
||||
semantic_tag(sem),
|
||||
scope(sco),
|
||||
scope_tag(sco));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco>
|
||||
struct __cuda_atomic_bind_exchange {
|
||||
_Type* __ptr;
|
||||
_Type* __old;
|
||||
_Type* __new;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
|
||||
__cuda_atomic_exchange(__ptr, *__old, *__new, _Atomic_Memorder{}, _Tag{}, _Sco{});
|
||||
}
|
||||
};
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_exchange_cuda(_Type* __ptr, _Type& __old, _Type __new, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
|
||||
__proxy_t* __old_proxy = reinterpret_cast<__proxy_t*>(&__old);
|
||||
__proxy_t* __new_proxy = reinterpret_cast<__proxy_t*>(&__new);
|
||||
if(__cuda_exchange_weak_if_local(__ptr_proxy, __new_proxy, __old_proxy)) {{return;}}
|
||||
__cuda_atomic_bind_exchange<__proxy_t, __proxy_tag, _Sco> __bound_swap{__ptr_proxy, __old_proxy, __new_proxy};
|
||||
__cuda_atomic_exchange_memory_order_dispatch(__bound_swap, __memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_exchange_cuda(_Type volatile* __ptr, _Type& __old, _Type __new, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __old_proxy = reinterpret_cast<__proxy_t*>(&__old);
|
||||
__proxy_t* __new_proxy = reinterpret_cast<__proxy_t*>(&__new);
|
||||
if(__cuda_exchange_weak_if_local(__ptr_proxy, __new_proxy, __old_proxy)) {{return;}}
|
||||
__cuda_atomic_bind_exchange<__proxy_t, __proxy_tag, _Sco> __bound_swap{__ptr_proxy, __old_proxy, __new_proxy};
|
||||
__cuda_atomic_exchange_memory_order_dispatch(__bound_swap, __memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // EXCHANGE_H
|
||||
110
cccl_upstream/libcudacxx/codegen/generators/fence.h
Normal file
110
cccl_upstream/libcudacxx/codegen/generators/fence.h
Normal file
@@ -0,0 +1,110 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef FENCE_H
|
||||
#define FENCE_H
|
||||
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline std::string membar_scope(Scope sco)
|
||||
{
|
||||
static std::map scope_map{
|
||||
std::pair{Scope::GPU, ".gl"},
|
||||
std::pair{Scope::System, ".sys"},
|
||||
std::pair{Scope::CTA, ".cta"},
|
||||
};
|
||||
|
||||
return scope_map[sco];
|
||||
}
|
||||
|
||||
inline void FormatFence(std::ostream& out)
|
||||
{
|
||||
// Argument ID Reference
|
||||
// 0 - Membar scope tag
|
||||
// 1 - Membar scope
|
||||
constexpr auto intrinsic_membar = R"XXX(
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_membar({0})
|
||||
{{ asm volatile("membar{1};" ::: "memory"); }})XXX";
|
||||
|
||||
const std::map membar_scopes{
|
||||
std::pair{Scope::GPU, ".gl"},
|
||||
std::pair{Scope::System, ".sys"},
|
||||
std::pair{Scope::CTA, ".cta"},
|
||||
};
|
||||
|
||||
for (const auto& sco : membar_scopes)
|
||||
{
|
||||
out << std::format(intrinsic_membar, scope_tag(sco.first), sco.second);
|
||||
}
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Fence scope tag
|
||||
// 1 - Fence scope
|
||||
// 2 - Fence order tag
|
||||
// 3 - Fence order
|
||||
constexpr auto intrinsic_fence = R"XXX(
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_fence({0}, {2})
|
||||
{{ asm volatile("fence{1}{3};" ::: "memory"); }})XXX";
|
||||
|
||||
const Scope fence_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
const Semantic fence_semantics[] = {
|
||||
Semantic::Acq_Rel,
|
||||
Semantic::Seq_Cst,
|
||||
};
|
||||
|
||||
for (const auto& sco : fence_scopes)
|
||||
{
|
||||
for (const auto& sem : fence_semantics)
|
||||
{
|
||||
out << std::format(intrinsic_fence, scope_tag(sco), semantic(sem), semantic_tag(sem), scope(sco));
|
||||
}
|
||||
}
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_thread_fence_cuda(int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); break;
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: [[fallthrough]];
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_fence(_Sco{}, __atomic_cuda_acq_rel{}); break;
|
||||
case __ATOMIC_RELAXED: break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: [[fallthrough]];
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELAXED: break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // FENCE_H
|
||||
219
cccl_upstream/libcudacxx/codegen/generators/fetch_ops.h
Normal file
219
cccl_upstream/libcudacxx/codegen/generators/fetch_ops.h
Normal file
@@ -0,0 +1,219 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef FETCH_OPS_H
|
||||
#define FETCH_OPS_H
|
||||
|
||||
#include <array>
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline std::string fetch_op_skip_v(std::string fetch_op)
|
||||
{
|
||||
if (fetch_op == "add")
|
||||
{
|
||||
return "constexpr auto __skip_v = __atomic_ptr_skip_t<_Type>::__skip;";
|
||||
}
|
||||
return "constexpr auto __skip_v = 1;";
|
||||
}
|
||||
|
||||
inline void FormatFetchOps(std::ostream& out)
|
||||
{
|
||||
const std::vector arithmetic_types = {
|
||||
Operand::Floating,
|
||||
Operand::Unsigned,
|
||||
Operand::Signed,
|
||||
};
|
||||
|
||||
const std::vector minmax_types = {
|
||||
Operand::Unsigned,
|
||||
Operand::Signed,
|
||||
};
|
||||
|
||||
const std::vector bitwise_types = {Operand::Bit};
|
||||
|
||||
const std::map op_support_map{
|
||||
std::pair{std::string{"add"}, std::pair{arithmetic_types, std::string{"arithmetic"}}},
|
||||
std::pair{std::string{"min"}, std::pair{minmax_types, std::string{"minmax"}}},
|
||||
std::pair{std::string{"max"}, std::pair{minmax_types, std::string{"minmax"}}},
|
||||
std::pair{std::string{"or"}, std::pair{bitwise_types, std::string{"bitwise"}}},
|
||||
std::pair{std::string{"xor"}, std::pair{bitwise_types, std::string{"bitwise"}}},
|
||||
std::pair{std::string{"and"}, std::pair{bitwise_types, std::string{"bitwise"}}},
|
||||
};
|
||||
|
||||
// Memory order dispatcher
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_fetch_memory_order_dispatch(_Fn& __cuda_fetch, int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_fetch(__atomic_cuda_acquire{}); break;
|
||||
case __ATOMIC_ACQ_REL: __cuda_fetch(__atomic_cuda_acq_rel{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_fetch(__atomic_cuda_release{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_fetch(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_fetch(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __cuda_fetch(__atomic_cuda_volatile{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_fetch(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Atomic Operation
|
||||
// 1 - Operand Type
|
||||
// 2 - Operand Size
|
||||
// 3 - Type Constraint
|
||||
// 4 - Memory Order
|
||||
// 5 - Memory Order function tag
|
||||
// 6 - Scope Constraint
|
||||
// 7 - Scope function tag
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_fetch_{0}(
|
||||
_Type* __ptr, _Type& __dst, _Type __op, {5}, __atomic_cuda_operand_{1}{2}, {7})
|
||||
{{ asm volatile("atom.{0}{4}{6}.{1}{2} %0,[%1],%2;" : "={3}"(__dst) : "l"(__ptr), "{3}"(__op) : "memory"); }})XXX";
|
||||
|
||||
// 0 - Atomic Operation
|
||||
// 1 - Operand type constraint
|
||||
// 2 - Pointer op skip_v
|
||||
constexpr auto fetch_bind_invoke = R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco>
|
||||
struct __cuda_atomic_bind_fetch_{0} {{
|
||||
_Type* __ptr;
|
||||
_Type* __dst;
|
||||
_Type* __op;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {{
|
||||
__cuda_atomic_fetch_{0}(__ptr, *__dst, *__op, _Atomic_Memorder{{}}, _Tag{{}}, _Sco{{}});
|
||||
}}
|
||||
}};
|
||||
template <class _Type, class _Up, class _Sco, __atomic_enable_if_native_{1}<_Type> = 0>
|
||||
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_{0}_cuda(_Type* __ptr, _Up __op, int __memorder, _Sco)
|
||||
{{
|
||||
{2}
|
||||
__op = __op * __skip_v;
|
||||
using __proxy_t = typename __atomic_cuda_deduce_{1}<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_{1}<_Type>::__tag;
|
||||
_Type __dst{{}};
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
|
||||
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
|
||||
__proxy_t* __op_proxy = reinterpret_cast<__proxy_t*>(&__op);
|
||||
if (__cuda_fetch_{0}_weak_if_local(__ptr_proxy, *__op_proxy, __dst_proxy)) {{return __dst;}}
|
||||
__cuda_atomic_bind_fetch_{0}<__proxy_t, __proxy_tag, _Sco> __bound_{0}{{__ptr_proxy, __dst_proxy, __op_proxy}};
|
||||
__cuda_atomic_fetch_memory_order_dispatch(__bound_{0}, __memorder, _Sco{{}});
|
||||
return __dst;
|
||||
}}
|
||||
template <class _Type, class _Up, class _Sco, __atomic_enable_if_native_{1}<_Type> = 0>
|
||||
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_{0}_cuda(_Type volatile* __ptr, _Up __op, int __memorder, _Sco)
|
||||
{{
|
||||
{2}
|
||||
__op = __op * __skip_v;
|
||||
using __proxy_t = typename __atomic_cuda_deduce_{1}<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_{1}<_Type>::__tag;
|
||||
_Type __dst{{}};
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
|
||||
__proxy_t* __op_proxy = reinterpret_cast<__proxy_t*>(&__op);
|
||||
if (__cuda_fetch_{0}_weak_if_local(__ptr_proxy, *__op_proxy, __dst_proxy)) {{return __dst;}}
|
||||
__cuda_atomic_bind_fetch_{0}<__proxy_t, __proxy_tag, _Sco> __bound_{0}{{__ptr_proxy, __dst_proxy, __op_proxy}};
|
||||
__cuda_atomic_fetch_memory_order_dispatch(__bound_{0}, __memorder, _Sco{{}});
|
||||
return __dst;
|
||||
}}
|
||||
)XXX";
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
32,
|
||||
64,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Acquire,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Release,
|
||||
Semantic::Acq_Rel,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
for (auto& op_kp : op_support_map)
|
||||
{
|
||||
const auto& op_name = op_kp.first;
|
||||
const auto& op_type_kp = op_kp.second;
|
||||
const auto& type_list = op_type_kp.first;
|
||||
const auto& deduction = op_type_kp.second;
|
||||
for (auto type : type_list)
|
||||
{
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
const std::string proxy_type = operand_proxy_type(type, size);
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
// There is no atom.add.s64
|
||||
if (op_name == "add" && type == Operand::Signed && size == 64)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
/* 0 */ op_name,
|
||||
/* 1 */ operand(type),
|
||||
/* 2 */ size,
|
||||
/* 3 */ constraints(type, size),
|
||||
/* 4 */ semantic(sem),
|
||||
/* 5 */ semantic_tag(sem),
|
||||
/* 6 */ scope(sco),
|
||||
/* 7 */ scope_tag(sco));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out << "\n" << std::format(fetch_bind_invoke, op_name, deduction, fetch_op_skip_v(op_name));
|
||||
}
|
||||
|
||||
out << R"XXX(
|
||||
template <class _Type, class _Up, class _Sco>
|
||||
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_sub_cuda(_Type* __ptr, _Up __op, int __memorder, _Sco)
|
||||
{
|
||||
return __atomic_fetch_add_cuda(__ptr, -__op, __memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Up, class _Sco>
|
||||
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_sub_cuda(_Type volatile* __ptr, _Up __op, int __memorder, _Sco)
|
||||
{
|
||||
return __atomic_fetch_add_cuda(__ptr, -__op, __memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // FETCH_OPS_H
|
||||
88
cccl_upstream/libcudacxx/codegen/generators/header.h
Normal file
88
cccl_upstream/libcudacxx/codegen/generators/header.h
Normal file
@@ -0,0 +1,88 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef HEADER_H
|
||||
#define HEADER_H
|
||||
|
||||
#include <string>
|
||||
|
||||
inline void FormatHeader(std::ostream& out)
|
||||
{
|
||||
constexpr auto header = R"XXX(//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// This is an autogenerated file, we want to ensure that it contains exactly the contents we want to generate
|
||||
// clang-format off
|
||||
|
||||
#ifndef _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
|
||||
#define _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <cuda/std/__type_traits/enable_if.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/is_unsigned.h>
|
||||
|
||||
#include <cuda/std/__atomic/scopes.h>
|
||||
#include <cuda/std/__atomic/order.h>
|
||||
#include <cuda/std/__atomic/functions/common.h>
|
||||
#include <cuda/std/__atomic/functions/cuda_ptx_generated_helper.h>
|
||||
#include <cuda/std/__atomic/functions/cuda_local.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_STD
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
extern "C" _CCCL_DEVICE void __atomic_cas_128b_unsupported_before_SM_90();
|
||||
extern "C" _CCCL_DEVICE void __atomic_exchange_128b_unsupported_before_SM_90();
|
||||
extern "C" _CCCL_DEVICE void __atomic_ldst_128b_unsupported_before_SM_70();
|
||||
)XXX";
|
||||
|
||||
out << header;
|
||||
}
|
||||
|
||||
inline void FormatTail(std::ostream& out)
|
||||
{
|
||||
constexpr auto tail = R"XXX(
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_STD
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
|
||||
|
||||
// clang-format on
|
||||
)XXX";
|
||||
|
||||
out << tail;
|
||||
}
|
||||
|
||||
#endif // HEADER_H
|
||||
407
cccl_upstream/libcudacxx/codegen/generators/ld_st.h
Normal file
407
cccl_upstream/libcudacxx/codegen/generators/ld_st.h
Normal file
@@ -0,0 +1,407 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef LD_ST_H
|
||||
#define LD_ST_H
|
||||
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline std::string semantic_ld_st(Semantic sem)
|
||||
{
|
||||
static std::map sem_map = {
|
||||
std::pair{Semantic::Relaxed, ".relaxed"},
|
||||
std::pair{Semantic::Release, ".release"},
|
||||
std::pair{Semantic::Acquire, ".acquire"},
|
||||
std::pair{Semantic::Volatile, ".volatile"},
|
||||
};
|
||||
return sem_map[sem];
|
||||
}
|
||||
|
||||
inline std::string scope_ld_st(Semantic sem, Scope sco)
|
||||
{
|
||||
if (sem == Semantic::Volatile)
|
||||
{
|
||||
return "";
|
||||
}
|
||||
return scope(sco);
|
||||
}
|
||||
|
||||
inline void FormatLoad(std::ostream& out)
|
||||
{
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_load_memory_order_dispatch(_Fn &__cuda_load, int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_load(__atomic_cuda_acquire{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_load(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_load(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_load(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Operand Type
|
||||
// 1 - Operand Size
|
||||
// 2 - Constraint
|
||||
// 3 - Memory order
|
||||
// 4 - Memory order semantic
|
||||
// 5 - Scope tag
|
||||
// 6 - Scope semantic
|
||||
// 7 - Mmio tag
|
||||
// 8 - Mmio semantic
|
||||
constexpr auto asm_intrinsic_format_128 = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_load(
|
||||
const _Type* __ptr, _Type& __dst, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
|
||||
{{
|
||||
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b ld/st is not supported until PTX ISA version 840");
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (),
|
||||
NV_ANY_TARGET, (__atomic_ldst_128b_unsupported_before_SM_70();)
|
||||
)
|
||||
asm volatile(R"YYY(
|
||||
{{
|
||||
.reg .b128 _d;
|
||||
ld{8}{4}{6}.b128 _d,[%2];
|
||||
mov.b128 {{%0, %1}}, _d;
|
||||
}}
|
||||
)YYY" : "=l"(__dst.__x),"=l"(__dst.__y) : "l"(__ptr) : "memory");
|
||||
}})XXX";
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_load(
|
||||
const _Type* __ptr, _Type& __dst, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
|
||||
{{ asm volatile("ld{8}{4}{6}.{0}{1} %0,[%1];" : "={2}"(__dst) : "l"(__ptr) : "memory"); }})XXX";
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
16,
|
||||
32,
|
||||
64,
|
||||
128,
|
||||
};
|
||||
|
||||
constexpr Operand supported_types[] = {
|
||||
Operand::Bit,
|
||||
Operand::Floating,
|
||||
Operand::Unsigned,
|
||||
Operand::Signed,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Acquire,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
constexpr Mmio mmio_states[] = {
|
||||
Mmio::Disabled,
|
||||
Mmio::Enabled,
|
||||
};
|
||||
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
for (auto type : supported_types)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
for (auto mm : mmio_states)
|
||||
{
|
||||
if (size == 16 && type == Operand::Floating)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (size == 128 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if ((mm == Mmio::Enabled) && ((sco != Scope::System) || (sem != Semantic::Relaxed)))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if (size == 128)
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format_128,
|
||||
/* 0 */ operand(type),
|
||||
/* 1 */ size,
|
||||
/* 2 */ constraints(type, size),
|
||||
/* 3 */ semantic_tag(sem),
|
||||
/* 4 */ semantic_ld_st(sem),
|
||||
/* 5 */ scope_tag(sco),
|
||||
/* 6 */ scope_ld_st(sem, sco),
|
||||
/* 7 */ mmio_tag(mm),
|
||||
/* 8 */ mmio(mm));
|
||||
}
|
||||
else
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
/* 0 */ operand(type),
|
||||
/* 1 */ size,
|
||||
/* 2 */ constraints(type, size),
|
||||
/* 3 */ semantic_tag(sem),
|
||||
/* 4 */ semantic_ld_st(sem),
|
||||
/* 5 */ scope_tag(sco),
|
||||
/* 6 */ scope_ld_st(sem, sco),
|
||||
/* 7 */ mmio_tag(mm),
|
||||
/* 8 */ mmio(mm));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco, typename _Mmio>
|
||||
struct __cuda_atomic_bind_load {
|
||||
const _Type* __ptr;
|
||||
_Type* __dst;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
|
||||
__cuda_atomic_load(__ptr, *__dst, _Atomic_Memorder{}, _Tag{}, _Sco{}, _Mmio{});
|
||||
}
|
||||
};
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_load_cuda(const _Type* __ptr, _Type& __dst, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
const __proxy_t* __ptr_proxy = reinterpret_cast<const __proxy_t*>(__ptr);
|
||||
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
|
||||
if (__cuda_load_weak_if_local(__ptr_proxy, __dst_proxy, sizeof(__proxy_t))) {{return;}}
|
||||
__cuda_atomic_bind_load<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_load{__ptr_proxy, __dst_proxy};
|
||||
__cuda_atomic_load_memory_order_dispatch(__bound_load, __memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_load_cuda(const _Type volatile* __ptr, _Type& __dst, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
const __proxy_t* __ptr_proxy = reinterpret_cast<const __proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
|
||||
if (__cuda_load_weak_if_local(__ptr_proxy, __dst_proxy, sizeof(__proxy_t))) {{return;}}
|
||||
__cuda_atomic_bind_load<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_load{__ptr_proxy, __dst_proxy};
|
||||
__cuda_atomic_load_memory_order_dispatch(__bound_load, __memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
inline void FormatStore(std::ostream& out)
|
||||
{
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_store_memory_order_dispatch(_Fn &__cuda_store, int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_RELEASE: __cuda_store(__atomic_cuda_release{}); break;
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_RELAXED: __cuda_store(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_RELEASE: [[fallthrough]];
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_RELAXED: __cuda_store(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
// Argument ID Reference
|
||||
// 0 - Operand Type
|
||||
// 1 - Operand Size
|
||||
// 2 - Constraint
|
||||
// 3 - Memory order
|
||||
// 4 - Memory order semantic
|
||||
// 5 - Scope tag
|
||||
// 6 - Scope semantic
|
||||
// 7 - Mmio tag
|
||||
// 8 - Mmio semantic
|
||||
constexpr auto asm_intrinsic_format_128 = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_store(
|
||||
_Type* __ptr, _Type& __val, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
|
||||
{{
|
||||
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b ld/st is not supported until PTX ISA version 840");
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (),
|
||||
NV_ANY_TARGET, (__atomic_ldst_128b_unsupported_before_SM_70();)
|
||||
)
|
||||
asm volatile(R"YYY(
|
||||
{{
|
||||
.reg .b128 _v;
|
||||
mov.b128 _v, {{%1, %2}};
|
||||
st{8}{4}{6}.b128 [%0],_v;
|
||||
}}
|
||||
)YYY" :: "l"(__ptr), "l"(__val.__x),"l"(__val.__y) : "memory");
|
||||
}})XXX";
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_store(
|
||||
_Type* __ptr, _Type& __val, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
|
||||
{{ asm volatile("st{8}{4}{6}.{0}{1} [%0],%1;" :: "l"(__ptr), "{2}"(__val) : "memory"); }})XXX";
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
16,
|
||||
32,
|
||||
64,
|
||||
128,
|
||||
};
|
||||
|
||||
constexpr Operand supported_types[] = {
|
||||
Operand::Bit,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Release,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
constexpr Mmio mmio_states[] = {
|
||||
Mmio::Disabled,
|
||||
Mmio::Enabled,
|
||||
};
|
||||
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
for (auto type : supported_types)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
for (auto mm : mmio_states)
|
||||
{
|
||||
if (size == 16 && type == Operand::Floating)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (size == 128 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if ((mm == Mmio::Enabled) && ((sco != Scope::System) || (sem != Semantic::Relaxed)))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if (size == 128)
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format_128,
|
||||
/* 0 */ operand(type),
|
||||
/* 1 */ size,
|
||||
/* 2 */ constraints(type, size),
|
||||
/* 3 */ semantic_tag(sem),
|
||||
/* 4 */ semantic_ld_st(sem),
|
||||
/* 5 */ scope_tag(sco),
|
||||
/* 6 */ scope_ld_st(sem, sco),
|
||||
/* 7 */ mmio_tag(mm),
|
||||
/* 8 */ mmio(mm));
|
||||
}
|
||||
else
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
/* 0 */ operand(type),
|
||||
/* 1 */ size,
|
||||
/* 2 */ constraints(type, size),
|
||||
/* 3 */ semantic_tag(sem),
|
||||
/* 4 */ semantic_ld_st(sem),
|
||||
/* 5 */ scope_tag(sco),
|
||||
/* 6 */ scope_ld_st(sem, sco),
|
||||
/* 7 */ mmio_tag(mm),
|
||||
/* 8 */ mmio(mm));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco, typename _Mmio>
|
||||
struct __cuda_atomic_bind_store {
|
||||
_Type* __ptr;
|
||||
_Type* __val;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
|
||||
__cuda_atomic_store(__ptr, *__val, _Atomic_Memorder{}, _Tag{}, _Sco{}, _Mmio{});
|
||||
}
|
||||
};
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_store_cuda(_Type* __ptr, _Type& __val, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
|
||||
__proxy_t* __val_proxy = reinterpret_cast<__proxy_t*>(&__val);
|
||||
if (__cuda_store_weak_if_local(__ptr_proxy, __val_proxy, sizeof(__proxy_t))) {{return;}}
|
||||
__cuda_atomic_bind_store<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_store{__ptr_proxy, __val_proxy};
|
||||
__cuda_atomic_store_memory_order_dispatch(__bound_store, __memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_store_cuda(volatile _Type* __ptr, _Type& __val, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __val_proxy = reinterpret_cast<__proxy_t*>(&__val);
|
||||
if (__cuda_store_weak_if_local(__ptr_proxy, __val_proxy, sizeof(__proxy_t))) {{return;}}
|
||||
__cuda_atomic_bind_store<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_store{__ptr_proxy, __val_proxy};
|
||||
__cuda_atomic_store_memory_order_dispatch(__bound_store, __memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // LD_ST_H
|
||||
68
cccl_upstream/libcudacxx/include/cuda/__algorithm/common.h
Normal file
68
cccl_upstream/libcudacxx/include/cuda/__algorithm/common.h
Normal file
@@ -0,0 +1,68 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef __CUDA___ALGORITHM_COMMON
|
||||
#define __CUDA___ALGORITHM_COMMON
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/__concepts/concept_macros.h>
|
||||
#include <cuda/std/__concepts/convertible_to.h>
|
||||
#include <cuda/std/__ranges/concepts.h>
|
||||
#include <cuda/std/__type_traits/remove_reference.h>
|
||||
#include <cuda/std/mdspan>
|
||||
#include <cuda/std/span>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
template <typename _Tp>
|
||||
using __as_span_t = ::cuda::std::span<::cuda::std::remove_reference_t<::cuda::std::ranges::range_reference_t<_Tp>>>;
|
||||
|
||||
//! @brief A concept that checks if the type can be converted to a `cuda::std::span`.
|
||||
//! The type must be a contiguous range.
|
||||
template <typename _Tp>
|
||||
_CCCL_CONCEPT __spannable = _CCCL_REQUIRES_EXPR((_Tp))( //
|
||||
requires(::cuda::std::ranges::contiguous_range<_Tp>), //
|
||||
requires(::cuda::std::convertible_to<_Tp, __as_span_t<_Tp>>));
|
||||
|
||||
template <typename _Tp>
|
||||
using __as_mdspan_t =
|
||||
::cuda::std::mdspan<typename ::cuda::std::decay_t<_Tp>::value_type,
|
||||
typename ::cuda::std::decay_t<_Tp>::extents_type,
|
||||
typename ::cuda::std::decay_t<_Tp>::layout_type,
|
||||
typename ::cuda::std::decay_t<_Tp>::accessor_type>;
|
||||
|
||||
//! @brief A concept that checks if the type can be converted to a `cuda::std::mdspan`.
|
||||
//! The type must have a conversion to `__as_mdspan_t<_Tp>`.
|
||||
template <typename _Tp>
|
||||
_CCCL_CONCEPT __mdspannable =
|
||||
_CCCL_REQUIRES_EXPR((_Tp))(requires(::cuda::std::convertible_to<_Tp, __as_mdspan_t<_Tp>>));
|
||||
|
||||
template <typename _Tp>
|
||||
[[nodiscard]] _CCCL_HOST_API constexpr auto __as_mdspan(_Tp&& __value) noexcept -> __as_mdspan_t<_Tp>
|
||||
{
|
||||
return ::cuda::std::forward<_Tp>(__value);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif //__CUDA___ALGORITHM_COMMON
|
||||
187
cccl_upstream/libcudacxx/include/cuda/__algorithm/copy.h
Normal file
187
cccl_upstream/libcudacxx/include/cuda/__algorithm/copy.h
Normal file
@@ -0,0 +1,187 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef __CUDA___ALGORITHM_COPY_H
|
||||
#define __CUDA___ALGORITHM_COPY_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
# include <cuda/__algorithm/common.h>
|
||||
# include <cuda/__stream/launch_transform.h>
|
||||
# include <cuda/__stream/stream_ref.h>
|
||||
# include <cuda/__type_traits/is_trivially_copyable.h>
|
||||
# include <cuda/std/__concepts/concept_macros.h>
|
||||
# include <cuda/std/__exception/exception_macros.h>
|
||||
# include <cuda/std/__host_stdlib/stdexcept>
|
||||
# include <cuda/std/mdspan>
|
||||
# include <cuda/std/span>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief Source access order for copy_bytes
|
||||
enum class source_access_order
|
||||
{
|
||||
# if _CCCL_CTK_AT_LEAST(13, 0)
|
||||
//! @brief Access source in stream order
|
||||
stream = ::cudaMemcpySrcAccessOrderStream,
|
||||
//! @brief Access source during the copy call, source can be destroyed after the API returns
|
||||
during_api_call = ::cudaMemcpySrcAccessOrderDuringApiCall,
|
||||
//! @brief Access source in any order, the order can change across CUDA releases
|
||||
any = ::cudaMemcpySrcAccessOrderAny,
|
||||
# else
|
||||
any = 0x3,
|
||||
# endif // _CCCL_CTK_BELOW(13, 0)
|
||||
};
|
||||
|
||||
//! @brief Configuration for copy_bytes
|
||||
struct copy_configuration
|
||||
{
|
||||
//! @brief Source memory location hint for copy_bytes, used only for managed memory
|
||||
memory_location src_location_hint = {};
|
||||
//! @brief Destination memory location hint for copy_bytes, used only for managed memory
|
||||
memory_location dst_location_hint = {};
|
||||
//! @brief Source access order for copy_bytes
|
||||
source_access_order src_access_order = source_access_order::any;
|
||||
};
|
||||
|
||||
namespace __detail
|
||||
{
|
||||
template <typename _SrcTy, typename _DstTy>
|
||||
_CCCL_HOST_API void __copy_bytes_impl(
|
||||
stream_ref __stream,
|
||||
::cuda::std::span<_SrcTy> __src,
|
||||
::cuda::std::span<_DstTy> __dst,
|
||||
[[maybe_unused]] copy_configuration __config)
|
||||
{
|
||||
static_assert(!::cuda::std::is_const_v<_DstTy>, "Copy destination can't be const");
|
||||
static_assert(::cuda::is_trivially_copyable_v<_SrcTy> && ::cuda::is_trivially_copyable_v<_DstTy>);
|
||||
|
||||
if (__src.size_bytes() > __dst.size_bytes())
|
||||
{
|
||||
_CCCL_THROW(::std::invalid_argument, "Copy destination is too small to fit the source data");
|
||||
}
|
||||
if (__src.size_bytes() == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
# if _CCCL_CTK_AT_LEAST(13, 0)
|
||||
CUmemcpyAttributes __attributes = {};
|
||||
__attributes.srcAccessOrder = static_cast<::CUmemcpySrcAccessOrder>(__config.src_access_order);
|
||||
__attributes.srcLocHint.id = __config.src_location_hint.id;
|
||||
__attributes.srcLocHint.type = static_cast<::CUmemLocationType>(__config.src_location_hint.type);
|
||||
__attributes.dstLocHint.id = __config.dst_location_hint.id;
|
||||
__attributes.dstLocHint.type = static_cast<::CUmemLocationType>(__config.dst_location_hint.type);
|
||||
|
||||
::cuda::__ensure_current_context guard(__stream);
|
||||
::cuda::__driver::__memcpyAsyncWithAttributes(
|
||||
__dst.data(), __src.data(), __src.size_bytes(), __stream.get(), __attributes);
|
||||
# else
|
||||
::cuda::__driver::__memcpyAsync(__dst.data(), __src.data(), __src.size_bytes(), __stream.get());
|
||||
# endif // _CCCL_CTK_BELOW(13, 0)
|
||||
}
|
||||
|
||||
template <typename _SrcElem,
|
||||
typename _SrcExtents,
|
||||
typename _SrcLayout,
|
||||
typename _SrcAccessor,
|
||||
typename _DstElem,
|
||||
typename _DstExtents,
|
||||
typename _DstLayout,
|
||||
typename _DstAccessor>
|
||||
_CCCL_HOST_API void __copy_bytes_impl(
|
||||
stream_ref __stream,
|
||||
::cuda::std::mdspan<_SrcElem, _SrcExtents, _SrcLayout, _SrcAccessor> __src,
|
||||
::cuda::std::mdspan<_DstElem, _DstExtents, _DstLayout, _DstAccessor> __dst,
|
||||
copy_configuration __config)
|
||||
{
|
||||
static_assert(::cuda::std::is_constructible_v<_DstExtents, _SrcExtents>,
|
||||
"Multidimensional copy requires both source and destination extents to be compatible");
|
||||
static_assert(::cuda::std::is_same_v<_SrcLayout, _DstLayout>,
|
||||
"Multidimensional copy requires both source and destination layouts to match");
|
||||
|
||||
// Check only destination, because the layout of destination is the same as source
|
||||
if (!__dst.is_exhaustive())
|
||||
{
|
||||
_CCCL_THROW(::std::invalid_argument, "copy_bytes supports only exhaustive mdspans");
|
||||
}
|
||||
|
||||
if (__src.extents() != __dst.extents())
|
||||
{
|
||||
_CCCL_THROW(::std::invalid_argument, "Copy destination size differs from the source");
|
||||
}
|
||||
|
||||
::cuda::__detail::__copy_bytes_impl(
|
||||
__stream,
|
||||
::cuda::std::span(__src.data_handle(), __src.mapping().required_span_size()),
|
||||
::cuda::std::span(__dst.data_handle(), __dst.mapping().required_span_size()),
|
||||
__config);
|
||||
}
|
||||
} // namespace __detail
|
||||
|
||||
//! @brief Launches a bytewise memory copy from source to destination into the provided
|
||||
//! stream.
|
||||
//!
|
||||
//! Both source and destination needs to be a `contiguous_range` and convert to
|
||||
//! `cuda::std::span`. The element types of both the source and destination range is
|
||||
//! required to be trivially copyable.
|
||||
//!
|
||||
//! This call might be synchronous if either source or destination is pagable host memory.
|
||||
//! It will be synchronous if both destination and copy is located in host memory.
|
||||
//!
|
||||
//! @param __stream Stream that the copy should be inserted into
|
||||
//! @param __src Source to copy from
|
||||
//! @param __dst Destination to copy into
|
||||
//! @param __config Configuration for the copy
|
||||
_CCCL_TEMPLATE(typename _SrcTy, typename _DstTy)
|
||||
_CCCL_REQUIRES( // NOLINT(modernize-type-traits)
|
||||
__spannable<transformed_device_argument_t<_SrcTy>> _CCCL_AND __spannable<transformed_device_argument_t<_DstTy>>)
|
||||
_CCCL_HOST_API void copy_bytes(stream_ref __stream, _SrcTy&& __src, _DstTy&& __dst, copy_configuration __config = {})
|
||||
{
|
||||
::cuda::__detail::__copy_bytes_impl(
|
||||
__stream,
|
||||
::cuda::std::span(launch_transform(__stream, ::cuda::std::forward<_SrcTy>(__src))),
|
||||
::cuda::std::span(launch_transform(__stream, ::cuda::std::forward<_DstTy>(__dst))),
|
||||
__config);
|
||||
}
|
||||
|
||||
//! @overload
|
||||
//! @note This overload accepts mdspan-compatible types.
|
||||
_CCCL_TEMPLATE(typename _SrcTy, typename _DstTy)
|
||||
_CCCL_REQUIRES( // NOLINT(modernize-type-traits)
|
||||
__mdspannable<transformed_device_argument_t<_SrcTy>> _CCCL_AND __mdspannable<transformed_device_argument_t<_DstTy>>)
|
||||
_CCCL_HOST_API void copy_bytes(stream_ref __stream, _SrcTy&& __src, _DstTy&& __dst, copy_configuration __config = {})
|
||||
{
|
||||
::cuda::__detail::__copy_bytes_impl(
|
||||
__stream,
|
||||
::cuda::__as_mdspan(launch_transform(__stream, ::cuda::std::forward<_SrcTy>(__src))),
|
||||
::cuda::__as_mdspan(launch_transform(__stream, ::cuda::std::forward<_DstTy>(__dst))),
|
||||
__config);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
#endif // __CUDA___ALGORITHM_COPY_H
|
||||
101
cccl_upstream/libcudacxx/include/cuda/__algorithm/fill.h
Normal file
101
cccl_upstream/libcudacxx/include/cuda/__algorithm/fill.h
Normal file
@@ -0,0 +1,101 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef __CUDA___ALGORITHM_FILL
|
||||
#define __CUDA___ALGORITHM_FILL
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
# include <cuda/__algorithm/common.h>
|
||||
# include <cuda/__stream/launch_transform.h>
|
||||
# include <cuda/__stream/stream_ref.h>
|
||||
# include <cuda/__type_traits/is_trivially_copyable.h>
|
||||
# include <cuda/std/__concepts/concept_macros.h>
|
||||
# include <cuda/std/__exception/exception_macros.h>
|
||||
# include <cuda/std/__host_stdlib/stdexcept>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
namespace __detail
|
||||
{
|
||||
template <typename _DstTy, ::cuda::std::size_t _DstSize>
|
||||
_CCCL_HOST_API void
|
||||
__fill_bytes_impl(stream_ref __stream, ::cuda::std::span<_DstTy, _DstSize> __dst, ::cuda::std::uint8_t __value)
|
||||
{
|
||||
static_assert(!::cuda::std::is_const_v<_DstTy>, "Fill destination can't be const");
|
||||
static_assert(::cuda::is_trivially_copyable_v<_DstTy>);
|
||||
|
||||
// TODO do a host callback if not device accessible?
|
||||
::cuda::__driver::__memsetAsync(__dst.data(), __value, __dst.size_bytes(), __stream.get());
|
||||
}
|
||||
|
||||
template <typename _DstElem, typename _DstExtents, typename _DstLayout, typename _DstAccessor>
|
||||
_CCCL_HOST_API void __fill_bytes_impl(stream_ref __stream,
|
||||
::cuda::std::mdspan<_DstElem, _DstExtents, _DstLayout, _DstAccessor> __dst,
|
||||
::cuda::std::uint8_t __value)
|
||||
{
|
||||
// Check if the mdspan is exhaustive
|
||||
if (!__dst.is_exhaustive())
|
||||
{
|
||||
_CCCL_THROW(::std::invalid_argument, "fill_bytes supports only exhaustive mdspans");
|
||||
}
|
||||
|
||||
::cuda::__detail::__fill_bytes_impl(
|
||||
__stream, ::cuda::std::span(__dst.data_handle(), __dst.mapping().required_span_size()), __value);
|
||||
}
|
||||
} // namespace __detail
|
||||
|
||||
//! @brief Launches an operation to bytewise fill the memory into the provided stream.
|
||||
//!
|
||||
//! The destination needs to be or launch_transform to a `contiguous_range` and convert to `cuda::std::span`.
|
||||
//! The element type of the destination is required to be trivially copyable.
|
||||
//!
|
||||
//! The destination cannot reside in pagable host memory.
|
||||
//!
|
||||
//! @param __stream Stream that the copy should be inserted into
|
||||
//! @param __dst Destination memory to fill
|
||||
//! @param __value Value to fill into every byte in the destination
|
||||
_CCCL_TEMPLATE(typename _DstTy)
|
||||
_CCCL_REQUIRES(__spannable<transformed_device_argument_t<_DstTy>>)
|
||||
_CCCL_HOST_API void fill_bytes(stream_ref __stream, _DstTy&& __dst, ::cuda::std::uint8_t __value)
|
||||
{
|
||||
::cuda::__detail::__fill_bytes_impl(
|
||||
__stream, ::cuda::std::span(launch_transform(__stream, ::cuda::std::forward<_DstTy>(__dst))), __value);
|
||||
}
|
||||
|
||||
//! @overload
|
||||
//! @note This overload accepts mdspan-compatible types.
|
||||
_CCCL_TEMPLATE(typename _DstTy)
|
||||
_CCCL_REQUIRES(__mdspannable<transformed_device_argument_t<_DstTy>>)
|
||||
_CCCL_HOST_API void fill_bytes(stream_ref __stream, _DstTy&& __dst, ::cuda::std::uint8_t __value)
|
||||
{
|
||||
::cuda::__detail::__fill_bytes_impl(
|
||||
__stream, __as_mdspan(launch_transform(__stream, ::cuda::std::forward<_DstTy>(__dst))), __value);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
#endif // __CUDA___ALGORITHM_FILL
|
||||
@@ -0,0 +1,170 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_H
|
||||
#define _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__annotated_ptr/access_property_encoding.h>
|
||||
#include <cuda/std/cstddef>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
template <typename>
|
||||
class __annotated_ptr_base; // forward declaration
|
||||
|
||||
class access_property
|
||||
{
|
||||
private:
|
||||
uint64_t __descriptor = __l2_interleave_normal;
|
||||
|
||||
friend class __annotated_ptr_base<access_property>;
|
||||
|
||||
// needed by __annotated_ptr_base
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(uint64_t __descriptor1) noexcept
|
||||
: __descriptor{__descriptor1}
|
||||
{}
|
||||
|
||||
public:
|
||||
struct shared
|
||||
{};
|
||||
struct global
|
||||
{};
|
||||
struct persisting
|
||||
{
|
||||
#if _CCCL_HAS_CTK()
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr operator ::cudaAccessProperty() const noexcept
|
||||
{
|
||||
return ::cudaAccessProperty::cudaAccessPropertyPersisting;
|
||||
}
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
};
|
||||
struct streaming
|
||||
{
|
||||
#if _CCCL_HAS_CTK()
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr operator ::cudaAccessProperty() const noexcept
|
||||
{
|
||||
return ::cudaAccessProperty::cudaAccessPropertyStreaming;
|
||||
}
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
};
|
||||
struct normal
|
||||
{
|
||||
#if _CCCL_HAS_CTK()
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr operator ::cudaAccessProperty() const noexcept
|
||||
{
|
||||
return ::cudaAccessProperty::cudaAccessPropertyNormal;
|
||||
}
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
};
|
||||
|
||||
_CCCL_HIDE_FROM_ABI access_property() noexcept = default;
|
||||
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(normal, float __fraction) noexcept
|
||||
: __descriptor{
|
||||
::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_Unchanged, __fraction)}
|
||||
{}
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(streaming, float __fraction) noexcept
|
||||
: __descriptor{
|
||||
::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_Unchanged, __fraction)}
|
||||
{}
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(persisting, float __fraction) noexcept
|
||||
: __descriptor{
|
||||
::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_Unchanged, __fraction)}
|
||||
{}
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(normal, float __fraction, streaming) noexcept
|
||||
: __descriptor{
|
||||
::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_First, __fraction)}
|
||||
{}
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(persisting, float __fraction, streaming) noexcept
|
||||
: __descriptor{::cuda::__l2_interleave(__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_First, __fraction)}
|
||||
{}
|
||||
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(global) noexcept {}
|
||||
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(normal) noexcept
|
||||
: access_property{normal{}, 1.0f}
|
||||
{}
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(streaming) noexcept
|
||||
: access_property{streaming{}, 1.0f}
|
||||
{}
|
||||
_CCCL_HOST_DEVICE_API constexpr access_property(persisting) noexcept
|
||||
: access_property{persisting{}, 1.0f}
|
||||
{}
|
||||
|
||||
_CCCL_HOST_DEVICE_API inline access_property(
|
||||
void* __ptr, size_t __primary_bytes, size_t __total_bytes, normal) noexcept
|
||||
: __descriptor{::cuda::__block_encoding(
|
||||
__l2_evict_t::_L2_Evict_Normal_Demote,
|
||||
__l2_evict_t::_L2_Evict_Unchanged,
|
||||
__ptr,
|
||||
__primary_bytes,
|
||||
__total_bytes)}
|
||||
{}
|
||||
|
||||
_CCCL_HOST_DEVICE_API inline access_property(
|
||||
void* __ptr, size_t __primary_bytes, size_t __total_bytes, streaming) noexcept
|
||||
: __descriptor{::cuda::__block_encoding(
|
||||
__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_Unchanged, __ptr, __primary_bytes, __total_bytes)}
|
||||
{}
|
||||
|
||||
_CCCL_HOST_DEVICE_API inline access_property(
|
||||
void* __ptr, size_t __primary_bytes, size_t __total_bytes, persisting) noexcept
|
||||
: __descriptor{::cuda::__block_encoding(
|
||||
__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_Unchanged, __ptr, __primary_bytes, __total_bytes)}
|
||||
{}
|
||||
|
||||
_CCCL_HOST_DEVICE_API inline access_property(
|
||||
void* __ptr, size_t __primary_bytes, size_t __total_bytes, global, streaming) noexcept
|
||||
: __descriptor{::cuda::__block_encoding(
|
||||
__l2_evict_t::_L2_Evict_Unchanged, __l2_evict_t::_L2_Evict_First, __ptr, __primary_bytes, __total_bytes)}
|
||||
{}
|
||||
|
||||
_CCCL_HOST_DEVICE_API inline access_property(
|
||||
void* __ptr, size_t __primary_bytes, size_t __total_bytes, normal, streaming) noexcept
|
||||
: __descriptor{::cuda::__block_encoding(
|
||||
__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_First, __ptr, __primary_bytes, __total_bytes)}
|
||||
{}
|
||||
|
||||
_CCCL_HOST_DEVICE_API inline access_property(
|
||||
void* __ptr, size_t __primary_bytes, size_t __total_bytes, streaming, streaming) noexcept
|
||||
: __descriptor{::cuda::__block_encoding(
|
||||
__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_First, __ptr, __primary_bytes, __total_bytes)}
|
||||
{}
|
||||
|
||||
_CCCL_HOST_DEVICE_API inline access_property(
|
||||
void* __ptr, size_t __primary_bytes, size_t __total_bytes, persisting, streaming) noexcept
|
||||
: __descriptor{::cuda::__block_encoding(
|
||||
__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_First, __ptr, __primary_bytes, __total_bytes)}
|
||||
{}
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr explicit operator uint64_t() const noexcept
|
||||
{
|
||||
return __descriptor;
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_H
|
||||
@@ -0,0 +1,171 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_ENCODING_H
|
||||
#define _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_ENCODING_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__annotated_ptr/createpolicy.h>
|
||||
#include <cuda/__cmath/ilog.h>
|
||||
#include <cuda/std/__algorithm/clamp.h>
|
||||
#include <cuda/std/__algorithm/max.h>
|
||||
#include <cuda/std/__bit/bit_cast.h>
|
||||
#include <cuda/std/__numeric/saturating_sub.h>
|
||||
#include <cuda/std/__utility/to_underlying.h>
|
||||
#include <cuda/std/cstddef>
|
||||
#include <cuda/std/cstdint>
|
||||
#include <cuda/std/limits>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
enum class __l2_descriptor_mode_t : uint32_t
|
||||
{
|
||||
_Desc_Implicit = 0,
|
||||
_Desc_Interleaved = 2,
|
||||
_Desc_Block_Type = 3
|
||||
};
|
||||
|
||||
/***********************************************************************************************************************
|
||||
* Range Block Descriptor
|
||||
**********************************************************************************************************************/
|
||||
|
||||
// MemoryDescriptor:blockDesc_t reference
|
||||
//
|
||||
// struct __block_desc_t // 64 bits
|
||||
// {
|
||||
// uint64_t __reserved1 : 37;
|
||||
// uint32_t __block_count : 7;
|
||||
// uint32_t __block_start : 7;
|
||||
// uint32_t __reserved2 : 1;
|
||||
// uint32_t __block_size_enum : 4; // 56 bits
|
||||
//
|
||||
// uint32_t __l2_cop_off : 1;
|
||||
// uint32_t __l2_cop_on : 2;
|
||||
// uint32_t __l2_descriptor_mode : 2;
|
||||
// uint32_t __l1_inv_dont_allocate : 1;
|
||||
// uint32_t __l2_sector_promote_256B : 1;
|
||||
// uint32_t __reserved3 : 1;
|
||||
// };
|
||||
|
||||
#if !_CCCL_CUDA_COMPILER(NVRTC)
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_API inline uint64_t __block_encoding_host(
|
||||
__l2_evict_t __primary, __l2_evict_t __secondary, const void* __ptr, uint32_t __primary_bytes, uint32_t __total_bytes)
|
||||
{
|
||||
_CCCL_ASSERT(__primary_bytes > 0, "primary_size must be greater than 0");
|
||||
_CCCL_ASSERT(__primary_bytes <= __total_bytes, "primary_size must be less than or equal to total_size");
|
||||
_CCCL_ASSERT(__secondary == __l2_evict_t::_L2_Evict_First || __secondary == __l2_evict_t::_L2_Evict_Unchanged,
|
||||
"secondary policy must be evict_first or evict_unchanged");
|
||||
auto __raw_ptr = ::cuda::std::bit_cast<uintptr_t>(__ptr);
|
||||
auto __log2_total_size = ::cuda::ceil_ilog2(__total_bytes);
|
||||
auto __block_size_enum = ::cuda::std::saturating_sub<uint32_t>(__log2_total_size, 19); // min block size = 4K
|
||||
auto __log2_block_size = 12u + __block_size_enum;
|
||||
auto __block_size = 1u << __log2_block_size;
|
||||
auto __block_start = static_cast<uint32_t>(__raw_ptr >> __log2_block_size); // ptr / block_size
|
||||
// vvvv block_end = ceil_div(ptr + primary_size, block_size)
|
||||
auto __block_end = static_cast<uint32_t>((__raw_ptr + __primary_bytes + __block_size - 1) >> __log2_block_size);
|
||||
_CCCL_ASSERT(__block_end >= __block_start, "block_end < block_start");
|
||||
// NOTE: there is a bug in PTX createpolicy when __block_size_enum == 13. The *incorrect* behavior matches the
|
||||
// following code:
|
||||
// auto __block_count = (__block_size_enum == 13)
|
||||
// ? ((__block_end - __block_start <= 127u) ? (__block_end - __block_start) : 1)
|
||||
// : ::cuda::std::clamp(__block_end - __block_start, 1u, 127u);
|
||||
auto __block_count = ::cuda::std::clamp(__block_end - __block_start, 1u, 127u);
|
||||
auto __l2_cop_off = ::cuda::std::to_underlying(__secondary);
|
||||
auto __l2_cop_on = ::cuda::std::to_underlying(__primary);
|
||||
auto __l2_descriptor_mode = ::cuda::std::to_underlying(__l2_descriptor_mode_t::_Desc_Block_Type);
|
||||
return static_cast<uint64_t>(__block_count) << 37 //
|
||||
| static_cast<uint64_t>(__block_start) << 44 //
|
||||
| static_cast<uint64_t>(__block_size_enum) << 52 //
|
||||
| static_cast<uint64_t>(__l2_cop_off) << 56 //
|
||||
| static_cast<uint64_t>(__l2_cop_on) << 57 //
|
||||
| static_cast<uint64_t>(__l2_descriptor_mode) << 59;
|
||||
}
|
||||
|
||||
#endif // !_CCCL_CUDA_COMPILER(NVRTC)
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline uint64_t __block_encoding(
|
||||
__l2_evict_t __primary, __l2_evict_t __secondary, const void* __ptr, size_t __primary_bytes, size_t __total_bytes)
|
||||
{
|
||||
_CCCL_ASSERT(__primary_bytes <= size_t{0xFFFFFFFF}, "primary size must be less than 4GB");
|
||||
_CCCL_ASSERT(__total_bytes <= size_t{0xFFFFFFFF}, "total size must be less than 4GB");
|
||||
auto __primary_bytes1 = static_cast<uint32_t>(__primary_bytes);
|
||||
auto __total_bytes1 = static_cast<uint32_t>(__total_bytes);
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_HOST,
|
||||
(return ::cuda::__block_encoding_host(__primary, __secondary, __ptr, __primary_bytes1, __total_bytes1);),
|
||||
(return ::cuda::__createpolicy_range(__primary, __secondary, __ptr, __primary_bytes1, __total_bytes1);))
|
||||
}
|
||||
|
||||
/***********************************************************************************************************************
|
||||
* Interleaved Descriptor
|
||||
**********************************************************************************************************************/
|
||||
|
||||
// MemoryDescriptor:interleaveDesc_t reference
|
||||
//
|
||||
// struct __interleaved_desc_t // 64 bits
|
||||
// {
|
||||
// uint64_t : 52;
|
||||
// uint32_t __fraction : 4; // 56 bits
|
||||
//
|
||||
// uint32_t __l2_cop_off : 1;
|
||||
// uint32_t __l2_cop_on : 2;
|
||||
// uint32_t __l2_descriptor_mode : 2;
|
||||
// uint32_t __l1_inv_dont_allocate : 1;
|
||||
// uint32_t __l2_sector_promote_256B : 1;
|
||||
// uint32_t : 1;
|
||||
// };
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr uint64_t
|
||||
__l2_interleave(__l2_evict_t __primary, __l2_evict_t __secondary, float __fraction)
|
||||
{
|
||||
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
|
||||
{
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_PROVIDES_SM_80, (return ::cuda::__createpolicy_fraction(__primary, __secondary, __fraction);), (return 0;))
|
||||
}
|
||||
_CCCL_ASSERT(__fraction > 0.0f && __fraction <= 1.0f, "fraction must be between 0.0f and 1.0f");
|
||||
_CCCL_ASSERT(__secondary == __l2_evict_t::_L2_Evict_First || __secondary == __l2_evict_t::_L2_Evict_Unchanged,
|
||||
"secondary policy must be evict_first or evict_unchanged");
|
||||
constexpr auto __epsilon = ::cuda::std::numeric_limits<float>::epsilon();
|
||||
auto __num = static_cast<uint32_t>((__fraction - __epsilon) * 16.0f); // fraction = num / 16
|
||||
auto __l2_cop_off = ::cuda::std::to_underlying(__secondary);
|
||||
auto __l2_cop_on = ::cuda::std::to_underlying(__primary);
|
||||
auto __l2_descriptor_mode = ::cuda::std::to_underlying(__l2_descriptor_mode_t::_Desc_Interleaved);
|
||||
return static_cast<uint64_t>(__num) << 52 //
|
||||
| static_cast<uint64_t>(__l2_cop_off) << 56 //
|
||||
| static_cast<uint64_t>(__l2_cop_on) << 57 //
|
||||
| static_cast<uint64_t>(__l2_descriptor_mode) << 59;
|
||||
}
|
||||
|
||||
inline constexpr auto __l2_interleave_normal = uint64_t{0x10F0000000000000};
|
||||
|
||||
inline constexpr auto __l2_interleave_streaming = uint64_t{0x12F0000000000000};
|
||||
|
||||
inline constexpr auto __l2_interleave_persisting = uint64_t{0x14F0000000000000};
|
||||
|
||||
inline constexpr auto __l2_interleave_normal_demote = uint64_t{0x16F0000000000000};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___ANNOTATED_PTR_ACCESS_PROPERTY_ENCODING_H
|
||||
@@ -0,0 +1,216 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_H
|
||||
#define _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__annotated_ptr/access_property.h>
|
||||
#include <cuda/__annotated_ptr/annotated_ptr_base.h>
|
||||
#include <cuda/__memcpy_async/memcpy_async.h>
|
||||
#include <cuda/__memory/address_space.h>
|
||||
#include <cuda/std/cstddef>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
template <typename _Tp, typename _Property>
|
||||
class annotated_ptr : private ::cuda::__annotated_ptr_base<_Property>
|
||||
{
|
||||
public:
|
||||
using value_type = _Tp;
|
||||
using size_type = size_t;
|
||||
using reference = value_type&;
|
||||
using pointer = value_type*;
|
||||
using const_pointer = const value_type*;
|
||||
using difference_type = ptrdiff_t;
|
||||
|
||||
private:
|
||||
static_assert(__is_access_property_v<_Property>);
|
||||
|
||||
static constexpr bool __is_smem = ::cuda::std::is_same_v<_Property, access_property::shared>;
|
||||
|
||||
// Converting from a 64-bit to 32-bit shared pointer and maybe back just for storage might or might not be profitable.
|
||||
pointer __repr = nullptr;
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline pointer __get(difference_type __n = 0) const noexcept
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(auto __repr1 = const_cast<void*>(static_cast<const volatile void*>(__repr + __n));
|
||||
return static_cast<pointer>(this->__apply_prop(__repr1));))
|
||||
return __repr + __n;
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline pointer __offset(difference_type __n) const noexcept
|
||||
{
|
||||
return __get(__n);
|
||||
}
|
||||
|
||||
public:
|
||||
_CCCL_HIDE_FROM_ABI annotated_ptr() noexcept = default;
|
||||
|
||||
_CCCL_HOST_DEVICE_API explicit constexpr annotated_ptr(pointer __p) noexcept
|
||||
: __repr{__p}
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, (_CCCL_ASSERT(!__is_smem, "shared memory pointer is not supported on the host");))
|
||||
if constexpr (__is_smem)
|
||||
{
|
||||
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(_CCCL_ASSERT(::cuda::device::is_address_from(__p, ::cuda::device::address_space::shared),
|
||||
"__p must be shared");))
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_ASSERT(__p != nullptr, "__p must not be null");
|
||||
_CCCL_IF_NOT_CONSTEVAL_DEFAULT
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(_CCCL_ASSERT(::cuda::device::is_address_from(__p, ::cuda::device::address_space::global),
|
||||
"__p must be global");))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename _RuntimeProperty>
|
||||
_CCCL_HOST_DEVICE_API inline annotated_ptr(pointer __p, _RuntimeProperty __prop) noexcept
|
||||
: ::cuda::__annotated_ptr_base<_Property>{access_property{__prop}}
|
||||
, __repr{__p}
|
||||
{
|
||||
static_assert(::cuda::std::is_same_v<_Property, access_property>,
|
||||
"This method requires annotated_ptr<T, cuda::access_property>");
|
||||
static_assert(__is_global_access_property_v<_RuntimeProperty>,
|
||||
"This method requires RuntimeProperty=global|normal|streaming|persisting|access_property");
|
||||
_CCCL_ASSERT(__p != nullptr, "__p must not be null");
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(_CCCL_ASSERT(::cuda::device::is_address_from(__p, ::cuda::device::address_space::global),
|
||||
"__p must be global");))
|
||||
}
|
||||
|
||||
// cannot be constexpr because of get()
|
||||
template <typename _OtherType, class _OtherProperty>
|
||||
_CCCL_HOST_DEVICE_API inline annotated_ptr(const annotated_ptr<_OtherType, _OtherProperty>& __other) noexcept
|
||||
: ::cuda::__annotated_ptr_base<_Property>{__other.__property()}
|
||||
, __repr{__other.get()}
|
||||
{
|
||||
using namespace ::cuda::std;
|
||||
static_assert(is_assignable_v<pointer&, _OtherType*>, "pointer must be assignable from other pointer");
|
||||
static_assert(is_same_v<_Property, _OtherProperty>
|
||||
|| (is_same_v<_Property, access_property> && !is_same_v<_OtherProperty, access_property::shared>),
|
||||
"Both properties must have same address space, or current property is access_property and "
|
||||
"OtherProperty is not shared");
|
||||
}
|
||||
|
||||
// cannot be constexpr because is_constant_evaluated is not supported by clang-14, gcc-8.
|
||||
// when the method is called in these platforms, it needs to be called at run-time.
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline pointer operator->() const noexcept
|
||||
{
|
||||
return __get();
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline reference operator*() const noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__get() != nullptr, "dereference of null annotated_ptr");
|
||||
return *__get();
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline reference operator[](difference_type __n) const noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__offset(__n) != nullptr, "dereference of null annotated_ptr");
|
||||
return *__offset(__n);
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr difference_type operator-(annotated_ptr __other) const noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__repr >= __other.__repr, "underflow");
|
||||
return __repr - __other.__repr;
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr explicit operator bool() const noexcept
|
||||
{
|
||||
return (__repr != nullptr);
|
||||
}
|
||||
|
||||
// cannot be constexpr because of operator->()
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline pointer get() const noexcept
|
||||
{
|
||||
return (__is_smem || __repr == nullptr)
|
||||
? __repr
|
||||
: annotated_ptr<value_type, access_property::global>{__repr}.operator->();
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr _Property __property() const noexcept
|
||||
{
|
||||
return this->__get_property();
|
||||
}
|
||||
};
|
||||
|
||||
//----------------------------------------------------------------------------------------------------------------------
|
||||
// memcpy_async
|
||||
|
||||
template <typename _Dst, typename _Src, typename _SrcProperty, typename _Shape, typename _Sync>
|
||||
_CCCL_HOST_DEVICE_API inline void
|
||||
memcpy_async(_Dst* __dst, annotated_ptr<_Src, _SrcProperty> __src, _Shape __shape, _Sync& __sync) noexcept
|
||||
{
|
||||
::cuda::memcpy_async(__dst, __src.operator->(), __shape, __sync);
|
||||
}
|
||||
|
||||
template <typename _Dst, typename _DstProperty, typename _Src, typename _SrcProperty, typename _Shape, typename _Sync>
|
||||
_CCCL_HOST_DEVICE_API inline void memcpy_async(
|
||||
annotated_ptr<_Dst, _DstProperty> __dst,
|
||||
annotated_ptr<_Src, _SrcProperty> __src,
|
||||
_Shape __shape,
|
||||
_Sync& __sync) noexcept
|
||||
{
|
||||
::cuda::memcpy_async(__dst.operator->(), __src.operator->(), __shape, __sync);
|
||||
}
|
||||
|
||||
template <typename _Group, typename _Dst, typename _Src, typename _SrcProperty, typename _Shape, typename _Sync>
|
||||
_CCCL_HOST_DEVICE_API inline void memcpy_async(
|
||||
const _Group& __group, _Dst* __dst, annotated_ptr<_Src, _SrcProperty> __src, _Shape __shape, _Sync& __sync) noexcept
|
||||
{
|
||||
::cuda::memcpy_async(__group, __dst, __src.operator->(), __shape, __sync);
|
||||
}
|
||||
|
||||
template <typename _Group,
|
||||
typename _Dst,
|
||||
typename _DstProperty,
|
||||
typename _Src,
|
||||
typename _SrcProperty,
|
||||
typename _Shape,
|
||||
typename _Sync>
|
||||
_CCCL_HOST_DEVICE_API inline void memcpy_async(
|
||||
const _Group& __group,
|
||||
annotated_ptr<_Dst, _DstProperty> __dst,
|
||||
annotated_ptr<_Src, _SrcProperty> __src,
|
||||
_Shape __shape,
|
||||
_Sync& __sync) noexcept
|
||||
{
|
||||
::cuda::memcpy_async(__group, __dst.operator->(), __src.operator->(), __shape, __sync);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_H
|
||||
@@ -0,0 +1,100 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_BASE_H
|
||||
#define _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_BASE_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__annotated_ptr/access_property.h>
|
||||
#include <cuda/__annotated_ptr/associate_access_property.h>
|
||||
#include <cuda/std/__type_traits/is_same.h>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
template <typename _AccessProperty>
|
||||
class __annotated_ptr_base
|
||||
{
|
||||
protected:
|
||||
_CCCL_HOST_DEVICE_API static constexpr uint64_t __default_property() noexcept
|
||||
{
|
||||
return ::cuda::std::is_same_v<_AccessProperty, access_property::global> ? __l2_interleave_normal
|
||||
: ::cuda::std::is_same_v<_AccessProperty, access_property::normal> ? __l2_interleave_normal_demote
|
||||
: ::cuda::std::is_same_v<_AccessProperty, access_property::persisting> ? __l2_interleave_persisting
|
||||
: ::cuda::std::is_same_v<_AccessProperty, access_property::streaming>
|
||||
? __l2_interleave_streaming
|
||||
: 0; // access_property::shared;
|
||||
}
|
||||
|
||||
static constexpr uint64_t __prop = __default_property();
|
||||
|
||||
_CCCL_HIDE_FROM_ABI __annotated_ptr_base() noexcept = default;
|
||||
|
||||
_CCCL_HOST_DEVICE_API constexpr __annotated_ptr_base(_AccessProperty) noexcept {}
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
[[nodiscard]] _CCCL_DEVICE_API void* __apply_prop(void* __p) const
|
||||
{
|
||||
return ::cuda::__associate(__p, _AccessProperty{});
|
||||
}
|
||||
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr _AccessProperty __get_property() const noexcept
|
||||
{
|
||||
return _AccessProperty{};
|
||||
}
|
||||
};
|
||||
|
||||
//----------------------------------------------------------------------------------------------------------------------
|
||||
// Specialization for dynamic access property
|
||||
|
||||
template <>
|
||||
class __annotated_ptr_base<access_property>
|
||||
{
|
||||
protected:
|
||||
uint64_t __prop = static_cast<uint64_t>(access_property{});
|
||||
|
||||
_CCCL_HOST_DEVICE_API constexpr __annotated_ptr_base(access_property __property) noexcept
|
||||
: __prop{static_cast<uint64_t>(__property)}
|
||||
{}
|
||||
|
||||
_CCCL_HIDE_FROM_ABI __annotated_ptr_base() noexcept = default;
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
[[nodiscard]] _CCCL_DEVICE_API void* __apply_prop(void* __p) const
|
||||
{
|
||||
return ::cuda::__associate_raw_descriptor(__p, __prop);
|
||||
}
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API constexpr access_property __get_property() const noexcept
|
||||
{
|
||||
return access_property{__prop};
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___ANNOTATED_PTR_ANNOTATED_PTR_BASE_H
|
||||
@@ -0,0 +1,83 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___ANNOTATED_PTR_APPLY_ACCESS_PROPERTY_H
|
||||
#define _CUDA___ANNOTATED_PTR_APPLY_ACCESS_PROPERTY_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__annotated_ptr/access_property.h>
|
||||
#include <cuda/__memory/address_space.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
template <typename _Shape>
|
||||
_CCCL_HOST_DEVICE_API inline void apply_access_property(
|
||||
[[maybe_unused]] const volatile void* __ptr,
|
||||
[[maybe_unused]] _Shape __shape,
|
||||
[[maybe_unused]] access_property::persisting __prop) noexcept
|
||||
{
|
||||
// clang-format off
|
||||
NV_IF_TARGET(
|
||||
NV_PROVIDES_SM_80,
|
||||
(_CCCL_ASSERT(__ptr != nullptr, "null pointer");
|
||||
if (!::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::global))
|
||||
{
|
||||
return;
|
||||
}
|
||||
constexpr size_t __line_size = 128;
|
||||
auto __p = reinterpret_cast<uint8_t*>(const_cast<void*>(__ptr));
|
||||
auto __nbytes = static_cast<size_t>(__shape);
|
||||
// Apply to all 128 bytes aligned cache lines inclusive of __p
|
||||
for (size_t __i = 0; __i < __nbytes; __i += __line_size) {
|
||||
asm volatile("prefetch.global.L2::evict_last [%0];" ::"l"(__p + __i) :);
|
||||
}))
|
||||
// clang-format on
|
||||
}
|
||||
|
||||
template <typename _Shape>
|
||||
_CCCL_HOST_DEVICE_API inline void apply_access_property(
|
||||
[[maybe_unused]] const volatile void* __ptr,
|
||||
[[maybe_unused]] _Shape __shape,
|
||||
[[maybe_unused]] access_property::normal __prop) noexcept
|
||||
{
|
||||
// clang-format off
|
||||
NV_IF_TARGET(
|
||||
NV_PROVIDES_SM_80,
|
||||
(_CCCL_ASSERT(__ptr != nullptr, "null pointer");
|
||||
if (!::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::global))
|
||||
{
|
||||
return;
|
||||
}
|
||||
constexpr size_t __line_size = 128;
|
||||
auto __p = reinterpret_cast<uint8_t*>(const_cast<void*>(__ptr));
|
||||
auto __nbytes = static_cast<size_t>(__shape);
|
||||
// Apply to all 128 bytes aligned cache lines inclusive of __p
|
||||
for (size_t __i = 0; __i < __nbytes; __i += __line_size) {
|
||||
asm volatile("prefetch.global.L2::evict_normal [%0];" ::"l"(__p + __i) :);
|
||||
}))
|
||||
// clang-format on
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___ANNOTATED_PTR_APPLY_ACCESS_PROPERTY_H
|
||||
@@ -0,0 +1,127 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___ANNOTATED_PTR_ASSOCIATE_ACCESS_PROPERTY_H
|
||||
#define _CUDA___ANNOTATED_PTR_ASSOCIATE_ACCESS_PROPERTY_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__annotated_ptr/access_property.h>
|
||||
#include <cuda/__memory/address_space.h>
|
||||
#include <cuda/std/__type_traits/always_false.h>
|
||||
#include <cuda/std/__type_traits/is_one_of.h>
|
||||
#include <cuda/std/__type_traits/is_same.h>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//----------------------------------------------------------------------------------------------------------------------
|
||||
// Private access property methods
|
||||
|
||||
template <typename _Property>
|
||||
inline constexpr bool __is_access_property_v =
|
||||
::cuda::std::__is_one_of_v<_Property,
|
||||
access_property::shared,
|
||||
access_property::global,
|
||||
access_property::normal,
|
||||
access_property::persisting,
|
||||
access_property::streaming,
|
||||
access_property>;
|
||||
|
||||
template <typename _Property>
|
||||
inline constexpr bool __is_global_access_property_v =
|
||||
::cuda::std::__is_one_of_v<_Property,
|
||||
access_property::global,
|
||||
access_property::normal,
|
||||
access_property::persisting,
|
||||
access_property::streaming,
|
||||
access_property>;
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
template <typename _Property>
|
||||
[[nodiscard]] _CCCL_DEVICE_API void* __associate_address_space(void* __ptr, [[maybe_unused]] _Property __prop)
|
||||
{
|
||||
if constexpr (::cuda::std::is_same_v<_Property, access_property::shared>)
|
||||
{
|
||||
[[maybe_unused]] bool __b = ::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::shared);
|
||||
_CCCL_ASSERT(__b, "");
|
||||
_CCCL_ASSUME(__b);
|
||||
}
|
||||
else if constexpr (__is_global_access_property_v<_Property>)
|
||||
{
|
||||
[[maybe_unused]] bool __b = ::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::global);
|
||||
_CCCL_ASSERT(__b, "");
|
||||
_CCCL_ASSUME(__b);
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(::cuda::std::__always_false_v<_Property>, "invalid access_property");
|
||||
}
|
||||
return __ptr;
|
||||
}
|
||||
|
||||
_CCCL_DEVICE_API inline void* __associate_raw_descriptor(void* __ptr, [[maybe_unused]] uint64_t __prop)
|
||||
{
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_80, (return ::__nv_associate_access_property(__ptr, __prop);))
|
||||
return __ptr;
|
||||
}
|
||||
|
||||
template <typename _Property>
|
||||
[[nodiscard]] _CCCL_DEVICE_API void* __associate_descriptor(void* __ptr, _Property __prop)
|
||||
{
|
||||
static_assert(__is_access_property_v<_Property>, "invalid cuda::access_property");
|
||||
if constexpr (!::cuda::std::is_same_v<_Property, access_property::shared>)
|
||||
{
|
||||
[[maybe_unused]] auto __raw_prop = static_cast<uint64_t>(access_property{__prop});
|
||||
return ::cuda::__associate_raw_descriptor(__ptr, __raw_prop);
|
||||
}
|
||||
return __ptr;
|
||||
}
|
||||
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
template <typename _Type, typename _Property>
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline _Type* __associate(_Type* __ptr, [[maybe_unused]] _Property __prop) noexcept
|
||||
{
|
||||
static_assert(__is_access_property_v<_Property>, "invalid cuda::access_property");
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(auto __void_ptr = const_cast<void*>(static_cast<const void*>(__ptr));
|
||||
auto __associated_ptr = ::cuda::__associate_address_space(__void_ptr, __prop);
|
||||
return static_cast<_Type*>(::cuda::__associate_descriptor(__associated_ptr, __prop));),
|
||||
(return __ptr;))
|
||||
}
|
||||
|
||||
//----------------------------------------------------------------------------------------------------------------------
|
||||
// Public access property methods
|
||||
|
||||
template <typename _Tp, typename _Property>
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE_API inline _Tp* associate_access_property(_Tp* __ptr, _Property __prop) noexcept
|
||||
{
|
||||
static_assert(__is_access_property_v<_Property>, "invalid cuda::access_property");
|
||||
return ::cuda::__associate(__ptr, __prop);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___ANNOTATED_PTR_ASSOCIATE_ACCESS_PROPERTY_H
|
||||
@@ -0,0 +1,210 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___ANNOTATED_PTR_CREATEPOLICY_H
|
||||
#define _CUDA___ANNOTATED_PTR_CREATEPOLICY_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__memory/address_space.h>
|
||||
#include <cuda/std/cstddef>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
enum class __l2_evict_t : uint32_t
|
||||
{
|
||||
_L2_Evict_Unchanged = 0, // called "_L2_Evict_Normal" at lower level
|
||||
_L2_Evict_First = 1,
|
||||
_L2_Evict_Last = 2,
|
||||
_L2_Evict_Normal_Demote = 3
|
||||
};
|
||||
|
||||
/***********************************************************************************************************************
|
||||
* PTX MAPPING
|
||||
**********************************************************************************************************************/
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
template <typename = void>
|
||||
[[nodiscard]] _CCCL_CONST _CCCL_DEVICE_API uint64_t __createpolicy_range_ptx(
|
||||
__l2_evict_t __primary, __l2_evict_t __secondary, size_t __gmem_ptr, uint32_t __primary_size, uint32_t __total_size)
|
||||
{
|
||||
uint64_t __policy;
|
||||
if (__secondary == __l2_evict_t::_L2_Evict_Unchanged)
|
||||
{
|
||||
if (__primary == __l2_evict_t::_L2_Evict_Last)
|
||||
{
|
||||
asm("createpolicy.range.global.L2::evict_last.b64 %0, [%1], %2, %3;"
|
||||
: "=l"(__policy)
|
||||
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_Normal_Demote)
|
||||
{
|
||||
asm("createpolicy.range.global.L2::evict_normal.b64 %0, [%1], %2, %3;"
|
||||
: "=l"(__policy)
|
||||
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_First)
|
||||
{
|
||||
asm("createpolicy.range.global.L2::evict_first.b64 %0, [%1], %2, %3;"
|
||||
: "=l"(__policy)
|
||||
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_Unchanged)
|
||||
{
|
||||
asm("createpolicy.range.global.L2::evict_unchanged.b64 %0, [%1], %2, %3;"
|
||||
: "=l"(__policy)
|
||||
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_UNREACHABLE();
|
||||
}
|
||||
}
|
||||
else // __secondary == _L2_Evict_First
|
||||
{
|
||||
if (__primary == __l2_evict_t::_L2_Evict_Last)
|
||||
{
|
||||
asm("createpolicy.range.global.L2::evict_last.L2::evict_first.b64 %0, [%1], %2, %3;"
|
||||
: "=l"(__policy)
|
||||
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_Normal_Demote)
|
||||
{
|
||||
asm("createpolicy.range.global.L2::evict_normal.L2::evict_first.b64 %0, [%1], %2, %3;"
|
||||
: "=l"(__policy)
|
||||
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_First)
|
||||
{
|
||||
asm("createpolicy.range.global.L2::evict_first.L2::evict_first.b64 %0, [%1], %2, %3;"
|
||||
: "=l"(__policy)
|
||||
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_Unchanged)
|
||||
{
|
||||
asm("createpolicy.range.global.L2::evict_unchanged.L2::evict_first.b64 %0, [%1], %2, %3;"
|
||||
: "=l"(__policy)
|
||||
: "l"(__gmem_ptr), "r"(__primary_size), "r"(__total_size));
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_UNREACHABLE();
|
||||
}
|
||||
}
|
||||
return __policy;
|
||||
}
|
||||
|
||||
template <typename = void>
|
||||
[[nodiscard]] _CCCL_CONST _CCCL_DEVICE_API uint64_t
|
||||
__createpolicy_fraction_ptx(__l2_evict_t __primary, __l2_evict_t __secondary, float __fraction)
|
||||
{
|
||||
uint64_t __policy;
|
||||
if (__secondary == __l2_evict_t::_L2_Evict_Unchanged)
|
||||
{
|
||||
if (__primary == __l2_evict_t::_L2_Evict_Last)
|
||||
{
|
||||
asm("createpolicy.fractional.L2::evict_last.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_Normal_Demote)
|
||||
{
|
||||
asm("createpolicy.fractional.L2::evict_normal.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_First)
|
||||
{
|
||||
asm("createpolicy.fractional.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_Unchanged)
|
||||
{
|
||||
asm("createpolicy.fractional.L2::evict_unchanged.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_UNREACHABLE();
|
||||
}
|
||||
}
|
||||
else // __secondary == _L2_Evict_First
|
||||
{
|
||||
if (__primary == __l2_evict_t::_L2_Evict_Last)
|
||||
{
|
||||
asm("createpolicy.fractional.L2::evict_last.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_Normal_Demote)
|
||||
{
|
||||
asm("createpolicy.fractional.L2::evict_normal.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_First)
|
||||
{
|
||||
asm("createpolicy.fractional.L2::evict_first.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
|
||||
}
|
||||
else if (__primary == __l2_evict_t::_L2_Evict_Unchanged)
|
||||
{
|
||||
asm("createpolicy.fractional.L2::evict_unchanged.L2::evict_first.b64 %0, %1;" : "=l"(__policy) : "f"(__fraction));
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_UNREACHABLE();
|
||||
}
|
||||
}
|
||||
return __policy;
|
||||
}
|
||||
|
||||
/***********************************************************************************************************************
|
||||
* C++ API
|
||||
**********************************************************************************************************************/
|
||||
|
||||
extern "C" _CCCL_DEVICE void __createpolicy_is_not_supported_before_SM_80();
|
||||
|
||||
template <typename T = void>
|
||||
[[nodiscard]] _CCCL_CONST _CCCL_DEVICE_API uint64_t __createpolicy_range(
|
||||
__l2_evict_t __primary, __l2_evict_t __secondary, const void* __ptr, uint32_t __primary_size, uint32_t __total_size)
|
||||
{
|
||||
_CCCL_ASSERT(::cuda::device::is_address_from(__ptr, ::cuda::device::address_space::global), "ptr must be global");
|
||||
_CCCL_ASSERT(__primary_size > 0, "primary_size must be greater than zero");
|
||||
_CCCL_ASSERT(__primary_size <= __total_size, "primary_size must be less than or equal to total_size");
|
||||
_CCCL_ASSERT(__secondary == __l2_evict_t::_L2_Evict_First || __secondary == __l2_evict_t::_L2_Evict_Unchanged,
|
||||
"secondary policy must be evict_first or evict_unchanged");
|
||||
[[maybe_unused]] auto __gmem_ptr = ::__cvta_generic_to_global(__ptr);
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_PROVIDES_SM_80,
|
||||
(return ::cuda::__createpolicy_range_ptx(__primary, __secondary, __gmem_ptr, __primary_size, __total_size);),
|
||||
(::cuda::__createpolicy_is_not_supported_before_SM_80(); return 0;))
|
||||
}
|
||||
|
||||
template <typename T = void>
|
||||
[[nodiscard]] _CCCL_CONST _CCCL_DEVICE_API uint64_t
|
||||
__createpolicy_fraction(__l2_evict_t __primary, __l2_evict_t __secondary, float __fraction = 1.0f)
|
||||
{
|
||||
_CCCL_ASSERT(__fraction > 0.0f && __fraction <= 1.0f, "fraction must be between 0.0f and 1.0f");
|
||||
_CCCL_ASSERT(__secondary == __l2_evict_t::_L2_Evict_First || __secondary == __l2_evict_t::_L2_Evict_Unchanged,
|
||||
"secondary policy must be evict_first or evict_unchanged");
|
||||
NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
|
||||
(return ::cuda::__createpolicy_fraction_ptx(__primary, __secondary, __fraction);),
|
||||
(::cuda::__createpolicy_is_not_supported_before_SM_80(); return 0;))
|
||||
}
|
||||
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___ANNOTATED_PTR_CREATEPOLICY_H
|
||||
997
cccl_upstream/libcudacxx/include/cuda/__argument/argument.h
Normal file
997
cccl_upstream/libcudacxx/include/cuda/__argument/argument.h
Normal file
@@ -0,0 +1,997 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___ARGUMENT_ARGUMENT_H
|
||||
#define _CUDA___ARGUMENT_ARGUMENT_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__argument/argument_bounds.h>
|
||||
#include <cuda/std/__algorithm/max_element.h>
|
||||
#include <cuda/std/__algorithm/min_element.h>
|
||||
#include <cuda/std/__cccl/assert.h>
|
||||
#include <cuda/std/__iterator/iterator_traits.h>
|
||||
#include <cuda/std/__iterator/readable_traits.h>
|
||||
#include <cuda/std/__ranges/concepts.h>
|
||||
#include <cuda/std/__type_traits/is_arithmetic.h>
|
||||
#include <cuda/std/__type_traits/is_array.h>
|
||||
#include <cuda/std/__type_traits/is_integer.h>
|
||||
#include <cuda/std/__type_traits/is_integral.h>
|
||||
#include <cuda/std/__type_traits/is_same.h>
|
||||
#include <cuda/std/__type_traits/remove_cv.h>
|
||||
#include <cuda/std/__type_traits/remove_cvref.h>
|
||||
#include <cuda/std/__type_traits/void_t.h>
|
||||
#include <cuda/std/__utility/cmp.h>
|
||||
#include <cuda/std/__utility/declval.h>
|
||||
#include <cuda/std/__utility/forward.h>
|
||||
#include <cuda/std/__utility/move.h>
|
||||
#include <cuda/std/cstddef>
|
||||
#include <cuda/std/limits>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_ARGUMENT
|
||||
|
||||
struct __access;
|
||||
|
||||
// =====================================================================
|
||||
// __element_type_of
|
||||
// =====================================================================
|
||||
|
||||
template <class _Tp, class = void>
|
||||
struct __element_type_from_member_iterator
|
||||
{
|
||||
using type = _Tp;
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
struct __element_type_from_member_iterator<_Tp, ::cuda::std::void_t<::cuda::std::iter_value_t<typename _Tp::iterator>>>
|
||||
{
|
||||
using type = ::cuda::std::iter_value_t<typename _Tp::iterator>;
|
||||
};
|
||||
|
||||
// Fallback: element type is the type itself.
|
||||
template <class _Tp, class = void>
|
||||
struct __element_type_of : __element_type_from_member_iterator<_Tp>
|
||||
{};
|
||||
|
||||
template <class _Tp>
|
||||
struct __element_type_of<_Tp, ::cuda::std::void_t<::cuda::std::iter_value_t<_Tp>>>
|
||||
{
|
||||
using type = ::cuda::std::iter_value_t<_Tp>;
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
using __element_type_of_t = typename __element_type_of<::cuda::std::remove_cvref_t<_Tp>>::type;
|
||||
|
||||
// =====================================================================
|
||||
// __is_sequence_v
|
||||
// =====================================================================
|
||||
|
||||
template <class _Tp>
|
||||
inline constexpr bool __is_sequence_v =
|
||||
(::cuda::std::is_array_v<::cuda::std::remove_cvref_t<_Tp>> || ::cuda::std::ranges::range<_Tp>)
|
||||
|| ::cuda::std::__has_random_access_traversal<_Tp>;
|
||||
|
||||
// =====================================================================
|
||||
// constant
|
||||
// =====================================================================
|
||||
|
||||
// Non-sequence wrappers intentionally do not reject types with a distinct element type.
|
||||
// A pointer or iterator can represent either a single value or a sequence; the wrapper
|
||||
// spelling carries that intent.
|
||||
|
||||
//! @brief Wraps a compile-time constant argument value.
|
||||
template <auto _Value, class _Tp = ::cuda::std::remove_cvref_t<decltype(_Value)>>
|
||||
class constant
|
||||
{
|
||||
public:
|
||||
using value_type = ::cuda::std::remove_cvref_t<_Tp>;
|
||||
using __element_type = value_type;
|
||||
|
||||
[[nodiscard]] _CCCL_API static constexpr value_type __get_value() noexcept
|
||||
{
|
||||
return static_cast<value_type>(_Value);
|
||||
}
|
||||
};
|
||||
|
||||
//! @brief Wraps a compile-time constant argument sequence.
|
||||
template <auto _Value>
|
||||
class __constant_sequence
|
||||
{
|
||||
public:
|
||||
using value_type = ::cuda::std::remove_cvref_t<decltype(_Value)>;
|
||||
using __element_type = __element_type_of_t<value_type>;
|
||||
|
||||
static_assert(__is_sequence_v<value_type>, "The value type of __constant_sequence must be a sequence");
|
||||
};
|
||||
|
||||
// __assert_in_range
|
||||
// =====================================================================
|
||||
|
||||
template <class _To, class _From>
|
||||
_CCCL_API constexpr void __assert_in_range([[maybe_unused]] _From __val) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::__cccl_is_cv_integer_v<_To> && ::cuda::std::__cccl_is_cv_integer_v<_From>)
|
||||
{
|
||||
_CCCL_ASSERT(::cuda::std::in_range<::cuda::std::remove_cv_t<_To>>(__val),
|
||||
"runtime bound value overflows the element type");
|
||||
}
|
||||
}
|
||||
|
||||
template <class _To, class _From>
|
||||
[[nodiscard]] _CCCL_API constexpr _To __runtime_bound_cast(_From __val) noexcept
|
||||
{
|
||||
__assert_in_range<_To>(__val);
|
||||
return static_cast<_To>(__val);
|
||||
}
|
||||
|
||||
template <class _To, auto _Value>
|
||||
_CCCL_API constexpr bool __static_bound_in_range() noexcept
|
||||
{
|
||||
using _RawTo = ::cuda::std::remove_cv_t<_To>;
|
||||
using _RawFrom = ::cuda::std::remove_cv_t<decltype(_Value)>;
|
||||
|
||||
if constexpr (::cuda::std::__cccl_is_integer_v<_RawTo> && ::cuda::std::__cccl_is_integer_v<_RawFrom>)
|
||||
{
|
||||
return ::cuda::std::in_range<_RawTo>(_Value);
|
||||
}
|
||||
else if constexpr (::cuda::std::is_arithmetic_v<_RawTo> && ::cuda::std::is_arithmetic_v<_RawFrom>)
|
||||
{
|
||||
return static_cast<_RawFrom>(static_cast<_RawTo>(_Value)) == _Value;
|
||||
}
|
||||
else
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
template <class _ElementType, class _StaticBounds>
|
||||
inline constexpr bool __valid_static_bounds_v = false;
|
||||
|
||||
template <class _ElementType>
|
||||
inline constexpr bool __valid_static_bounds_v<_ElementType, no_bounds> = true;
|
||||
|
||||
template <class _ElementType, auto _Lowest, auto _Highest>
|
||||
inline constexpr bool __valid_static_bounds_v<_ElementType, static_bounds<_Lowest, _Highest>> =
|
||||
__static_bound_in_range<_ElementType, _Lowest>() && __static_bound_in_range<_ElementType, _Highest>();
|
||||
|
||||
template <class _ElementType, class _StaticBounds>
|
||||
_CCCL_API constexpr _ElementType __wrapper_static_lowest() noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_same_v<_StaticBounds, no_bounds>)
|
||||
{
|
||||
return __type_lowest<_ElementType>();
|
||||
}
|
||||
else
|
||||
{
|
||||
return static_cast<_ElementType>(_StaticBounds::lower());
|
||||
}
|
||||
}
|
||||
|
||||
template <class _ElementType, class _StaticBounds>
|
||||
_CCCL_API constexpr _ElementType __wrapper_static_highest() noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_same_v<_StaticBounds, no_bounds>)
|
||||
{
|
||||
return __type_highest<_ElementType>();
|
||||
}
|
||||
else
|
||||
{
|
||||
return static_cast<_ElementType>(_StaticBounds::upper());
|
||||
}
|
||||
}
|
||||
|
||||
template <class _ElementType, class _StaticBounds>
|
||||
_CCCL_API constexpr _ElementType __effective_lowest(runtime_bounds<_ElementType> __runtime_bounds) noexcept
|
||||
{
|
||||
const auto __static_lowest = __wrapper_static_lowest<_ElementType, _StaticBounds>();
|
||||
return __static_lowest < __runtime_bounds.lower() ? __runtime_bounds.lower() : __static_lowest;
|
||||
}
|
||||
|
||||
template <class _ElementType, class _StaticBounds>
|
||||
_CCCL_API constexpr _ElementType __effective_highest(runtime_bounds<_ElementType> __runtime_bounds) noexcept
|
||||
{
|
||||
const auto __static_highest = __wrapper_static_highest<_ElementType, _StaticBounds>();
|
||||
return __static_highest < __runtime_bounds.upper() ? __static_highest : __runtime_bounds.upper();
|
||||
}
|
||||
|
||||
template <class _ElementType, class _StaticBounds>
|
||||
_CCCL_API constexpr bool __has_bounds_intersection(runtime_bounds<_ElementType> __runtime_bounds) noexcept
|
||||
{
|
||||
return __bounds_less_equal(__effective_lowest<_ElementType, _StaticBounds>(__runtime_bounds),
|
||||
__effective_highest<_ElementType, _StaticBounds>(__runtime_bounds));
|
||||
}
|
||||
|
||||
template <class _ElementType, class _StaticBounds>
|
||||
_CCCL_API constexpr void __validate_bounds_intersection(runtime_bounds<_ElementType> __runtime_bounds) noexcept
|
||||
{
|
||||
static_assert(__valid_static_bounds_v<_ElementType, _StaticBounds>,
|
||||
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
|
||||
"values representable by the element type");
|
||||
_CCCL_VERIFY((__has_bounds_intersection<_ElementType, _StaticBounds>(__runtime_bounds)),
|
||||
"static and runtime argument bounds do not intersect");
|
||||
}
|
||||
|
||||
template <class _ElementType, class _StaticBounds>
|
||||
_CCCL_API constexpr void __validate_static_element_bounds([[maybe_unused]] const _ElementType& __val) noexcept
|
||||
{
|
||||
if constexpr (!::cuda::std::is_same_v<_StaticBounds, no_bounds>)
|
||||
{
|
||||
_CCCL_ASSERT((__bounds_greater_equal(__val, __wrapper_static_lowest<_ElementType, _StaticBounds>())),
|
||||
"immediate argument value is below static lowest bound");
|
||||
_CCCL_ASSERT((__bounds_less_equal(__val, __wrapper_static_highest<_ElementType, _StaticBounds>())),
|
||||
"immediate argument value is above static highest bound");
|
||||
}
|
||||
}
|
||||
|
||||
template <class _ElementType>
|
||||
_CCCL_API constexpr void __validate_runtime_element_bounds(
|
||||
[[maybe_unused]] const _ElementType& __val, [[maybe_unused]] runtime_bounds<_ElementType> __runtime_bounds) noexcept
|
||||
{
|
||||
_CCCL_ASSERT((__bounds_greater_equal(__val, __runtime_bounds.lower())),
|
||||
"immediate argument value is below runtime lower bound");
|
||||
_CCCL_ASSERT((__bounds_less_equal(__val, __runtime_bounds.upper())),
|
||||
"immediate argument value is above runtime upper bound");
|
||||
}
|
||||
|
||||
// =====================================================================
|
||||
// immediate
|
||||
// =====================================================================
|
||||
|
||||
//! @brief Wraps a runtime argument value with optional bounds.
|
||||
//!
|
||||
//! The value is host-accessible at API call time.
|
||||
template <class _Arg, class _StaticBounds = no_bounds>
|
||||
class immediate
|
||||
{
|
||||
public:
|
||||
using __element_type = __element_type_of_t<_Arg>;
|
||||
|
||||
static_assert(__valid_static_bounds_v<__element_type, _StaticBounds>,
|
||||
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
|
||||
"values representable by the element type");
|
||||
|
||||
private:
|
||||
friend struct __access;
|
||||
|
||||
_Arg __arg_;
|
||||
|
||||
_CCCL_API constexpr void __validate_value() const noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_same_v<::cuda::std::remove_cvref_t<_Arg>, __element_type>
|
||||
&& ::cuda::std::is_arithmetic_v<__element_type>)
|
||||
{
|
||||
__validate_static_element_bounds<__element_type, _StaticBounds>(__arg_);
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
_CCCL_API constexpr immediate(_Arg __arg) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
{
|
||||
__validate_value();
|
||||
}
|
||||
|
||||
_CCCL_API constexpr immediate(_Arg __arg, _StaticBounds) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
{
|
||||
__validate_value();
|
||||
}
|
||||
};
|
||||
|
||||
#ifndef _CCCL_DOXYGEN_INVOKED
|
||||
template <class _Arg, auto _Lowest, auto _Highest>
|
||||
_CCCL_HOST_DEVICE immediate(_Arg, static_bounds<_Lowest, _Highest>)
|
||||
-> immediate<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
|
||||
#endif // _CCCL_DOXYGEN_INVOKED
|
||||
|
||||
// =====================================================================
|
||||
// __immediate_sequence
|
||||
// =====================================================================
|
||||
|
||||
//! @brief Wraps a runtime argument sequence with optional bounds.
|
||||
template <class _Arg, class _StaticBounds = no_bounds>
|
||||
class __immediate_sequence
|
||||
{
|
||||
public:
|
||||
using __element_type = __element_type_of_t<_Arg>;
|
||||
|
||||
static_assert(__is_sequence_v<_Arg>, "immediate sequence arguments must have a distinct element type");
|
||||
static_assert(__valid_static_bounds_v<__element_type, _StaticBounds>,
|
||||
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
|
||||
"values representable by the element type");
|
||||
|
||||
private:
|
||||
friend struct __access;
|
||||
|
||||
_Arg __arg_;
|
||||
runtime_bounds<__element_type> __runtime_bounds_{};
|
||||
|
||||
_CCCL_API constexpr void __validate_bounds() const noexcept
|
||||
{
|
||||
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
|
||||
}
|
||||
|
||||
_CCCL_API constexpr void __validate_element(const __element_type& __val) const noexcept
|
||||
{
|
||||
__validate_static_element_bounds<__element_type, _StaticBounds>(__val);
|
||||
__validate_runtime_element_bounds(__val, __runtime_bounds_);
|
||||
}
|
||||
|
||||
_CCCL_API constexpr void __validate_value() const noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::__has_random_access_traversal<_Arg>)
|
||||
{ // FIXME: (miscco) This is broken. we do not know the size of the sequence
|
||||
}
|
||||
else if constexpr (__is_sequence_v<_Arg> && !::cuda::std::__has_random_access_traversal<_Arg>
|
||||
&& ::cuda::std::is_arithmetic_v<__element_type>)
|
||||
{
|
||||
for (const auto& __a : __arg_)
|
||||
{
|
||||
__validate_element(__a);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
_CCCL_API constexpr __immediate_sequence(_Arg __arg) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
{
|
||||
__validate_bounds();
|
||||
__validate_value();
|
||||
}
|
||||
|
||||
_CCCL_API constexpr __immediate_sequence(_Arg __arg, _StaticBounds) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
{
|
||||
__validate_bounds();
|
||||
__validate_value();
|
||||
}
|
||||
|
||||
template <class _BoundsTp>
|
||||
_CCCL_API constexpr __immediate_sequence(_Arg __arg, runtime_bounds<_BoundsTp> __rb) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
, __runtime_bounds_{__runtime_bound_cast<__element_type>(__rb.lower()),
|
||||
__runtime_bound_cast<__element_type>(__rb.upper())}
|
||||
{
|
||||
__validate_bounds();
|
||||
__validate_value();
|
||||
}
|
||||
|
||||
template <class _BoundsTp>
|
||||
_CCCL_API constexpr __immediate_sequence(_Arg __arg, _StaticBounds, runtime_bounds<_BoundsTp> __rb) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
, __runtime_bounds_{__runtime_bound_cast<__element_type>(__rb.lower()),
|
||||
__runtime_bound_cast<__element_type>(__rb.upper())}
|
||||
{
|
||||
__validate_bounds();
|
||||
__validate_value();
|
||||
}
|
||||
|
||||
template <class _BoundsTp>
|
||||
_CCCL_API constexpr __immediate_sequence(_Arg __arg, runtime_bounds<_BoundsTp> __rb, _StaticBounds __sb) noexcept
|
||||
: __immediate_sequence(::cuda::std::move(__arg), __sb, __rb)
|
||||
{}
|
||||
};
|
||||
|
||||
#ifndef _CCCL_DOXYGEN_INVOKED
|
||||
template <class _Arg, auto _Lowest, auto _Highest>
|
||||
_CCCL_HOST_DEVICE __immediate_sequence(_Arg, static_bounds<_Lowest, _Highest>)
|
||||
-> __immediate_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
|
||||
template <class _Arg, auto _Lowest, auto _Highest, class _Tp>
|
||||
_CCCL_HOST_DEVICE __immediate_sequence(_Arg, static_bounds<_Lowest, _Highest>, runtime_bounds<_Tp>)
|
||||
-> __immediate_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
|
||||
template <class _Arg, class _Tp, auto _Lowest, auto _Highest>
|
||||
_CCCL_HOST_DEVICE __immediate_sequence(_Arg, runtime_bounds<_Tp>, static_bounds<_Lowest, _Highest>)
|
||||
-> __immediate_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
#endif // _CCCL_DOXYGEN_INVOKED
|
||||
|
||||
// =====================================================================
|
||||
// __deferred_base / deferred / deferred_sequence
|
||||
// =====================================================================
|
||||
|
||||
//! @brief Common base for deferred argument wrappers.
|
||||
template <class _Arg, class _StaticBounds = no_bounds>
|
||||
class __deferred_base
|
||||
{
|
||||
public:
|
||||
using __element_type = __element_type_of_t<_Arg>;
|
||||
|
||||
static_assert(__valid_static_bounds_v<__element_type, _StaticBounds>,
|
||||
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
|
||||
"values representable by the element type");
|
||||
|
||||
private:
|
||||
friend struct __access;
|
||||
|
||||
_Arg __arg_;
|
||||
runtime_bounds<__element_type> __runtime_bounds_{};
|
||||
|
||||
public:
|
||||
_CCCL_API constexpr __deferred_base(_Arg __arg) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
{
|
||||
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
|
||||
}
|
||||
|
||||
_CCCL_API constexpr __deferred_base(_Arg __arg, _StaticBounds) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
{
|
||||
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
|
||||
}
|
||||
|
||||
template <class _BoundsTp>
|
||||
_CCCL_API constexpr __deferred_base(_Arg __arg, runtime_bounds<_BoundsTp> __rb) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
, __runtime_bounds_{__runtime_bound_cast<__element_type>(__rb.lower()),
|
||||
__runtime_bound_cast<__element_type>(__rb.upper())}
|
||||
{
|
||||
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
|
||||
}
|
||||
|
||||
template <class _BoundsTp>
|
||||
_CCCL_API constexpr __deferred_base(_Arg __arg, _StaticBounds, runtime_bounds<_BoundsTp> __rb) noexcept
|
||||
: __arg_{::cuda::std::move(__arg)}
|
||||
, __runtime_bounds_{__runtime_bound_cast<__element_type>(__rb.lower()),
|
||||
__runtime_bound_cast<__element_type>(__rb.upper())}
|
||||
{
|
||||
__validate_bounds_intersection<__element_type, _StaticBounds>(__runtime_bounds_);
|
||||
}
|
||||
|
||||
template <class _BoundsTp>
|
||||
_CCCL_API constexpr __deferred_base(_Arg __arg, runtime_bounds<_BoundsTp> __rb, _StaticBounds __sb) noexcept
|
||||
: __deferred_base(::cuda::std::move(__arg), __sb, __rb)
|
||||
{}
|
||||
};
|
||||
|
||||
//! @brief Wraps a reference to a single value that is potentially not available at API call time but will be available
|
||||
//! by the time the argument is consumed in stream order.
|
||||
template <class _Arg, class _StaticBounds = no_bounds>
|
||||
class deferred : public __deferred_base<_Arg, _StaticBounds>
|
||||
{
|
||||
public:
|
||||
using __deferred_base<_Arg, _StaticBounds>::__deferred_base;
|
||||
};
|
||||
|
||||
#ifndef _CCCL_DOXYGEN_INVOKED
|
||||
template <class _Arg>
|
||||
_CCCL_HOST_DEVICE deferred(_Arg) -> deferred<_Arg>;
|
||||
|
||||
template <class _Arg, auto _Lowest, auto _Highest>
|
||||
_CCCL_HOST_DEVICE deferred(_Arg, static_bounds<_Lowest, _Highest>) -> deferred<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
|
||||
template <class _Arg, class _Tp>
|
||||
_CCCL_HOST_DEVICE deferred(_Arg, runtime_bounds<_Tp>) -> deferred<_Arg>;
|
||||
|
||||
template <class _Arg, auto _Lowest, auto _Highest, class _Tp>
|
||||
_CCCL_HOST_DEVICE deferred(_Arg, static_bounds<_Lowest, _Highest>, runtime_bounds<_Tp>)
|
||||
-> deferred<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
|
||||
template <class _Arg, class _Tp, auto _Lowest, auto _Highest>
|
||||
_CCCL_HOST_DEVICE deferred(_Arg, runtime_bounds<_Tp>, static_bounds<_Lowest, _Highest>)
|
||||
-> deferred<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
#endif // _CCCL_DOXYGEN_INVOKED
|
||||
|
||||
//! @brief Wraps a reference to a sequence of values that is potentially not available at API call time but will be
|
||||
//! available by the time the argument is consumed in stream order.
|
||||
template <class _Arg, class _StaticBounds = no_bounds>
|
||||
class deferred_sequence : public __deferred_base<_Arg, _StaticBounds>
|
||||
{
|
||||
public:
|
||||
static_assert(__is_sequence_v<_Arg>, "deferred sequence arguments must have a distinct element type");
|
||||
|
||||
using __deferred_base<_Arg, _StaticBounds>::__deferred_base;
|
||||
};
|
||||
|
||||
#ifndef _CCCL_DOXYGEN_INVOKED
|
||||
template <class _Arg>
|
||||
_CCCL_HOST_DEVICE deferred_sequence(_Arg) -> deferred_sequence<_Arg>;
|
||||
|
||||
template <class _Arg, auto _Lowest, auto _Highest>
|
||||
_CCCL_HOST_DEVICE deferred_sequence(_Arg, static_bounds<_Lowest, _Highest>)
|
||||
-> deferred_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
|
||||
template <class _Arg, class _Tp>
|
||||
_CCCL_HOST_DEVICE deferred_sequence(_Arg, runtime_bounds<_Tp>) -> deferred_sequence<_Arg>;
|
||||
|
||||
template <class _Arg, auto _Lowest, auto _Highest, class _Tp>
|
||||
_CCCL_HOST_DEVICE deferred_sequence(_Arg, static_bounds<_Lowest, _Highest>, runtime_bounds<_Tp>)
|
||||
-> deferred_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
|
||||
template <class _Arg, class _Tp, auto _Lowest, auto _Highest>
|
||||
_CCCL_HOST_DEVICE deferred_sequence(_Arg, runtime_bounds<_Tp>, static_bounds<_Lowest, _Highest>)
|
||||
-> deferred_sequence<_Arg, static_bounds<_Lowest, _Highest>>;
|
||||
#endif // _CCCL_DOXYGEN_INVOKED
|
||||
|
||||
// =====================================================================
|
||||
// __access
|
||||
// =====================================================================
|
||||
|
||||
struct __access
|
||||
{
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr _Arg& __arg(immediate<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__arg_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr const _Arg& __arg(const immediate<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__arg_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr _Arg&& __arg(immediate<_Arg, _StaticBounds>&& __wrapper) noexcept
|
||||
{
|
||||
return ::cuda::std::move(__wrapper.__arg_);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr _Arg& __arg(__immediate_sequence<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__arg_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr const _Arg&
|
||||
__arg(const __immediate_sequence<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__arg_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr _Arg&& __arg(__immediate_sequence<_Arg, _StaticBounds>&& __wrapper) noexcept
|
||||
{
|
||||
return ::cuda::std::move(__wrapper.__arg_);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr _Arg& __arg(__deferred_base<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__arg_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr const _Arg&
|
||||
__arg(const __deferred_base<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__arg_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr _Arg&& __arg(__deferred_base<_Arg, _StaticBounds>&& __wrapper) noexcept
|
||||
{
|
||||
return ::cuda::std::move(__wrapper.__arg_);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr runtime_bounds<__element_type_of_t<_Arg>>&
|
||||
__runtime_bounds(__immediate_sequence<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__runtime_bounds_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr const runtime_bounds<__element_type_of_t<_Arg>>&
|
||||
__runtime_bounds(const __immediate_sequence<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__runtime_bounds_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr runtime_bounds<__element_type_of_t<_Arg>>&
|
||||
__runtime_bounds(__deferred_base<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__runtime_bounds_;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API static constexpr const runtime_bounds<__element_type_of_t<_Arg>>&
|
||||
__runtime_bounds(const __deferred_base<_Arg, _StaticBounds>& __wrapper) noexcept
|
||||
{
|
||||
return __wrapper.__runtime_bounds_;
|
||||
}
|
||||
};
|
||||
|
||||
// =====================================================================
|
||||
// __unwrap
|
||||
// =====================================================================
|
||||
|
||||
template <class _Tp>
|
||||
inline constexpr bool __is_wrapper_v = false;
|
||||
template <class _Arg, class _StaticBounds>
|
||||
inline constexpr bool __is_wrapper_v<immediate<_Arg, _StaticBounds>> = true;
|
||||
template <auto _Value, class _Tp>
|
||||
inline constexpr bool __is_wrapper_v<constant<_Value, _Tp>> = true;
|
||||
template <auto _Value>
|
||||
inline constexpr bool __is_wrapper_v<__constant_sequence<_Value>> = true;
|
||||
template <class _Arg, class _StaticBounds>
|
||||
inline constexpr bool __is_wrapper_v<__immediate_sequence<_Arg, _StaticBounds>> = true;
|
||||
template <class _Arg, class _StaticBounds>
|
||||
inline constexpr bool __is_wrapper_v<deferred<_Arg, _StaticBounds>> = true;
|
||||
template <class _Arg, class _StaticBounds>
|
||||
inline constexpr bool __is_wrapper_v<deferred_sequence<_Arg, _StaticBounds>> = true;
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES((!__is_wrapper_v<::cuda::std::remove_cvref_t<_Tp>>) )
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp&& __unwrap(_Tp&& __arg) noexcept
|
||||
{
|
||||
return ::cuda::std::forward<_Tp>(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr _Arg& __unwrap(immediate<_Arg, _StaticBounds>& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr const _Arg& __unwrap(const immediate<_Arg, _StaticBounds>& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr _Arg __unwrap(immediate<_Arg, _StaticBounds>&& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(::cuda::std::move(__arg));
|
||||
}
|
||||
|
||||
template <auto _Value, class _Tp>
|
||||
[[nodiscard]] _CCCL_API constexpr typename constant<_Value, _Tp>::value_type
|
||||
__unwrap(const constant<_Value, _Tp>&) noexcept
|
||||
{
|
||||
return constant<_Value, _Tp>::__get_value();
|
||||
}
|
||||
|
||||
template <auto _Value>
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::remove_cvref_t<decltype(_Value)>
|
||||
__unwrap(const __constant_sequence<_Value>&) noexcept
|
||||
{
|
||||
return _Value;
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr _Arg& __unwrap(__immediate_sequence<_Arg, _StaticBounds>& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr const _Arg& __unwrap(const __immediate_sequence<_Arg, _StaticBounds>& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr _Arg __unwrap(__immediate_sequence<_Arg, _StaticBounds>&& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(::cuda::std::move(__arg));
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr _Arg& __unwrap(deferred<_Arg, _StaticBounds>& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr const _Arg& __unwrap(const deferred<_Arg, _StaticBounds>& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr _Arg __unwrap(deferred<_Arg, _StaticBounds>&& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(::cuda::std::move(__arg));
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr _Arg& __unwrap(deferred_sequence<_Arg, _StaticBounds>& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr const _Arg& __unwrap(const deferred_sequence<_Arg, _StaticBounds>& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr _Arg __unwrap(deferred_sequence<_Arg, _StaticBounds>&& __arg) noexcept
|
||||
{
|
||||
return __access::__arg(::cuda::std::move(__arg));
|
||||
}
|
||||
|
||||
template <auto _Value, class _Tp>
|
||||
_CCCL_API constexpr auto __constant_compute_lowest() noexcept
|
||||
{
|
||||
return constant<_Value, _Tp>::__get_value();
|
||||
}
|
||||
|
||||
template <auto _Value, class _Tp>
|
||||
_CCCL_API constexpr auto __constant_compute_highest() noexcept
|
||||
{
|
||||
return constant<_Value, _Tp>::__get_value();
|
||||
}
|
||||
|
||||
template <auto _Value>
|
||||
_CCCL_API constexpr auto __constant_sequence_compute_lowest() noexcept
|
||||
{
|
||||
using _ElementType = __element_type_of_t<::cuda::std::remove_cvref_t<decltype(_Value)>>;
|
||||
auto __first = _Value.begin();
|
||||
auto __last = _Value.end();
|
||||
|
||||
if (__first == __last)
|
||||
{
|
||||
return __type_lowest<_ElementType>();
|
||||
}
|
||||
return static_cast<_ElementType>(*::cuda::std::min_element(__first, __last));
|
||||
}
|
||||
|
||||
template <auto _Value>
|
||||
_CCCL_API constexpr auto __constant_sequence_compute_highest() noexcept
|
||||
{
|
||||
using _ElementType = __element_type_of_t<::cuda::std::remove_cvref_t<decltype(_Value)>>;
|
||||
auto __first = _Value.begin();
|
||||
auto __last = _Value.end();
|
||||
|
||||
if (__first == __last)
|
||||
{
|
||||
return __type_highest<_ElementType>();
|
||||
}
|
||||
return static_cast<_ElementType>(*::cuda::std::max_element(__first, __last));
|
||||
}
|
||||
|
||||
// =====================================================================
|
||||
// __traits
|
||||
// =====================================================================
|
||||
|
||||
//! @brief Traits for argument wrappers and plain argument values.
|
||||
//!
|
||||
//! Models @c numeric_limits for bounds: @c lowest is the lower bound, @c highest is the upper bound.
|
||||
//! Use in @c if @c constexpr for compile-time dispatch based on bounds.
|
||||
template <class _Tp>
|
||||
struct __traits_impl
|
||||
{
|
||||
using value_type = _Tp;
|
||||
using element_type = __element_type_of_t<_Tp>;
|
||||
static constexpr bool is_constant = false;
|
||||
static constexpr bool is_deferred = false;
|
||||
static constexpr bool is_single_value = true;
|
||||
static constexpr element_type lowest = __type_lowest<element_type>();
|
||||
static constexpr element_type highest = __type_highest<element_type>();
|
||||
};
|
||||
|
||||
template <auto _Value, class _Tp>
|
||||
struct __traits_impl<constant<_Value, _Tp>>
|
||||
{
|
||||
using value_type = typename constant<_Value, _Tp>::value_type;
|
||||
using element_type = value_type;
|
||||
static constexpr bool is_constant = true;
|
||||
static constexpr bool is_deferred = false;
|
||||
static constexpr bool is_single_value = true;
|
||||
static constexpr element_type lowest = __constant_compute_lowest<_Value, _Tp>();
|
||||
static constexpr element_type highest = __constant_compute_highest<_Value, _Tp>();
|
||||
};
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
struct __traits_impl<immediate<_Arg, _StaticBounds>>
|
||||
{
|
||||
using value_type = _Arg;
|
||||
using element_type = __element_type_of_t<_Arg>;
|
||||
static_assert(__valid_static_bounds_v<element_type, _StaticBounds>,
|
||||
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
|
||||
"values representable by the element type");
|
||||
|
||||
static constexpr bool is_constant = false;
|
||||
static constexpr bool is_deferred = false;
|
||||
static constexpr bool is_single_value = true;
|
||||
static constexpr element_type lowest = __wrapper_static_lowest<element_type, _StaticBounds>();
|
||||
static constexpr element_type highest = __wrapper_static_highest<element_type, _StaticBounds>();
|
||||
};
|
||||
|
||||
template <auto _Value>
|
||||
struct __traits_impl<__constant_sequence<_Value>>
|
||||
{
|
||||
using value_type = ::cuda::std::remove_cvref_t<decltype(_Value)>;
|
||||
using element_type = __element_type_of_t<value_type>;
|
||||
static_assert(__is_sequence_v<value_type>, "The value type of __constant_sequence must be a sequence");
|
||||
static constexpr bool is_constant = true;
|
||||
static constexpr bool is_deferred = false;
|
||||
static constexpr bool is_single_value = false;
|
||||
static constexpr element_type lowest = __constant_sequence_compute_lowest<_Value>();
|
||||
static constexpr element_type highest = __constant_sequence_compute_highest<_Value>();
|
||||
};
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
struct __traits_impl<__immediate_sequence<_Arg, _StaticBounds>>
|
||||
{
|
||||
using value_type = _Arg;
|
||||
using element_type = __element_type_of_t<_Arg>;
|
||||
static_assert(__is_sequence_v<value_type>, "immediate sequence arguments must have a distinct element type");
|
||||
static_assert(__valid_static_bounds_v<element_type, _StaticBounds>,
|
||||
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
|
||||
"values representable by the element type");
|
||||
|
||||
static constexpr bool is_constant = false;
|
||||
static constexpr bool is_deferred = false;
|
||||
static constexpr bool is_single_value = false;
|
||||
static constexpr element_type lowest = __wrapper_static_lowest<element_type, _StaticBounds>();
|
||||
static constexpr element_type highest = __wrapper_static_highest<element_type, _StaticBounds>();
|
||||
};
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
struct __traits_impl<deferred<_Arg, _StaticBounds>>
|
||||
{
|
||||
using value_type = _Arg;
|
||||
using element_type = __element_type_of_t<_Arg>;
|
||||
static_assert(__valid_static_bounds_v<element_type, _StaticBounds>,
|
||||
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
|
||||
"values representable by the element type");
|
||||
|
||||
static constexpr bool is_constant = false;
|
||||
static constexpr bool is_deferred = true;
|
||||
static constexpr bool is_single_value = true;
|
||||
static constexpr element_type lowest = __wrapper_static_lowest<element_type, _StaticBounds>();
|
||||
static constexpr element_type highest = __wrapper_static_highest<element_type, _StaticBounds>();
|
||||
};
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
struct __traits_impl<deferred_sequence<_Arg, _StaticBounds>>
|
||||
{
|
||||
using value_type = _Arg;
|
||||
using element_type = __element_type_of_t<_Arg>;
|
||||
static_assert(__is_sequence_v<value_type>, "deferred sequence arguments must have a distinct element type");
|
||||
static_assert(__valid_static_bounds_v<element_type, _StaticBounds>,
|
||||
"argument wrapper bounds type must be cuda::args::no_bounds or cuda::args::static_bounds with "
|
||||
"values representable by the element type");
|
||||
|
||||
static constexpr bool is_constant = false;
|
||||
static constexpr bool is_deferred = true;
|
||||
static constexpr bool is_single_value = false;
|
||||
static constexpr element_type lowest = __wrapper_static_lowest<element_type, _StaticBounds>();
|
||||
static constexpr element_type highest = __wrapper_static_highest<element_type, _StaticBounds>();
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
struct __traits : __traits_impl<::cuda::std::remove_cvref_t<_Tp>>
|
||||
{};
|
||||
|
||||
// =====================================================================
|
||||
// __lowest_ / __highest_ — free functions
|
||||
// =====================================================================
|
||||
|
||||
//! @brief Returns the effective lowest bound, combining static and runtime bounds.
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES((!__is_wrapper_v<::cuda::std::remove_cv_t<_Tp>>) )
|
||||
[[nodiscard]] _CCCL_API constexpr auto __lowest_(_Tp) noexcept
|
||||
{
|
||||
return __type_lowest<__element_type_of_t<_Tp>>();
|
||||
}
|
||||
|
||||
template <auto _Value, class _Tp>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __lowest_(constant<_Value, _Tp>) noexcept
|
||||
{
|
||||
return __constant_compute_lowest<_Value, _Tp>();
|
||||
}
|
||||
|
||||
template <auto _Value>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __lowest_(__constant_sequence<_Value>) noexcept
|
||||
{
|
||||
return __constant_sequence_compute_lowest<_Value>();
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __lowest_(immediate<_Arg, _StaticBounds> __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __lowest_(__immediate_sequence<_Arg, _StaticBounds> __arg) noexcept
|
||||
{
|
||||
using _ET = __element_type_of_t<_Arg>;
|
||||
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
|
||||
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
|
||||
return __effective_lowest<_ET, _StaticBounds>(__runtime_bounds);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __lowest_(deferred<_Arg, _StaticBounds> __arg) noexcept
|
||||
{
|
||||
using _ET = __element_type_of_t<_Arg>;
|
||||
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
|
||||
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
|
||||
return __effective_lowest<_ET, _StaticBounds>(__runtime_bounds);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __lowest_(deferred_sequence<_Arg, _StaticBounds> __arg) noexcept
|
||||
{
|
||||
using _ET = __element_type_of_t<_Arg>;
|
||||
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
|
||||
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
|
||||
return __effective_lowest<_ET, _StaticBounds>(__runtime_bounds);
|
||||
}
|
||||
|
||||
//! @brief Returns the effective highest bound, combining static and runtime bounds.
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES((!__is_wrapper_v<::cuda::std::remove_cv_t<_Tp>>) )
|
||||
[[nodiscard]] _CCCL_API constexpr auto __highest_(_Tp) noexcept
|
||||
{
|
||||
return __type_highest<__element_type_of_t<_Tp>>();
|
||||
}
|
||||
|
||||
template <auto _Value, class _Tp>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __highest_(constant<_Value, _Tp>) noexcept
|
||||
{
|
||||
return __constant_compute_highest<_Value, _Tp>();
|
||||
}
|
||||
|
||||
template <auto _Value>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __highest_(__constant_sequence<_Value>) noexcept
|
||||
{
|
||||
return __constant_sequence_compute_highest<_Value>();
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __highest_(immediate<_Arg, _StaticBounds> __arg) noexcept
|
||||
{
|
||||
return __access::__arg(__arg);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __highest_(__immediate_sequence<_Arg, _StaticBounds> __arg) noexcept
|
||||
{
|
||||
using _ET = __element_type_of_t<_Arg>;
|
||||
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
|
||||
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
|
||||
return __effective_highest<_ET, _StaticBounds>(__runtime_bounds);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __highest_(deferred<_Arg, _StaticBounds> __arg) noexcept
|
||||
{
|
||||
using _ET = __element_type_of_t<_Arg>;
|
||||
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
|
||||
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
|
||||
return __effective_highest<_ET, _StaticBounds>(__runtime_bounds);
|
||||
}
|
||||
|
||||
template <class _Arg, class _StaticBounds>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __highest_(deferred_sequence<_Arg, _StaticBounds> __arg) noexcept
|
||||
{
|
||||
using _ET = __element_type_of_t<_Arg>;
|
||||
const auto& __runtime_bounds = __access::__runtime_bounds(__arg);
|
||||
__validate_bounds_intersection<_ET, _StaticBounds>(__runtime_bounds);
|
||||
return __effective_highest<_ET, _StaticBounds>(__runtime_bounds);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_ARGUMENT
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA___ARGUMENT_ARGUMENT_H
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user