[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,68 @@
# Byte-compiled / optimized / DLL files
__pycache__/
*.py[cod]
# C extensions
*.so
# Distribution / packaging
.Python
env/
build/
develop-eggs/
dist/
downloads/
eggs/
#lib/ # We actually have things checked in to lib/
lib64/
parts/
sdist/
var/
*.egg-info/
.installed.cfg
*.egg
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
!*.spec/
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.coverage
.cache
nosetests.xml
coverage.xml
# Translations
*.mo
*.pot
# Django stuff:
*.log
# Sphinx documentation
docs/_build/
# PyBuilder
target/
# MSVC libraries test harness
env.lst
keep.lst
# Editor by-products
.vscode/
# Random things
build-*
# Perforce files
.p4config

View File

@@ -0,0 +1,102 @@
find_package(Python COMPONENTS Interpreter)
if (NOT Python_Interpreter_FOUND)
message(
FATAL_ERROR
"Failed to find python interpreter, which is required for running tests and building a libcu++ static library."
)
endif()
# Determine the host triple to avoid invoking `${CXX} -dumpmachine`.
include(GetHostTriple)
get_host_triple(LLVM_INFERRED_HOST_TRIPLE)
set(
LLVM_HOST_TRIPLE
"${LLVM_INFERRED_HOST_TRIPLE}"
CACHE STRING
"Host on which LLVM binaries will run"
)
# By default, we target the host, but this can be overridden at CMake
# invocation time.
set(
LLVM_DEFAULT_TARGET_TRIPLE
"${LLVM_HOST_TRIPLE}"
CACHE STRING
"Default target for which LLVM will generate code."
)
set(TARGET_TRIPLE "${LLVM_DEFAULT_TARGET_TRIPLE}")
message(STATUS "LLVM host triple: ${LLVM_HOST_TRIPLE}")
message(STATUS "LLVM default target triple: ${LLVM_DEFAULT_TARGET_TRIPLE}")
set(LIT_EXTRA_ARGS "" CACHE STRING "Use for additional options (e.g. -j12)")
find_program(LLVM_DEFAULT_EXTERNAL_LIT lit)
set(LLVM_LIT_ARGS "-sv ${LIT_EXTRA_ARGS}")
# Libcudacxx's main lit tests
add_subdirectory(libcudacxx)
add_subdirectory(cmake)
# Set appropriate warning levels for MSVC/sane
if ("${CMAKE_CUDA_COMPILER_ID}" STREQUAL "NVIDIA")
# CUDA 11.5 and down do not support '-use-local-env'
if (MSVC)
set(
headertest_warning_levels_device
-Xcompiler=/W4
-Xcompiler=/WX
-Wno-deprecated-gpu-targets
)
if ("${CMAKE_CUDA_COMPILER_VERSION}" GREATER_EQUAL "11.6.0")
list(APPEND headertest_warning_levels_device --use-local-env)
endif()
else()
set(
headertest_warning_levels_device
-Wall
-Werror
all-warnings
-Wno-deprecated-gpu-targets
)
endif()
if (
CCCL_ENABLE_TILE
AND "${CMAKE_CUDA_COMPILER_VERSION}" VERSION_LESS_EQUAL "13.2"
)
message(
FATAL_ERROR
"tile programs require NVCC 13.3 or later; found NVCC ${CMAKE_CUDA_COMPILER_VERSION}"
)
endif()
# Set warnings for Clang as device compiler
elseif ("${CMAKE_CUDA_COMPILER_ID}" STREQUAL "Clang")
set(
headertest_warning_levels_device
-Wall
-Werror
-Wno-unknown-cuda-version
-Xclang=-fcuda-allow-variadic-functions
)
# If the CMAKE_CUDA_COMPILER is unknown, try to use gcc style warnings
else()
set(headertest_warning_levels_device -Wall -Werror)
endif()
# Set raw host/device warnings
if (MSVC)
set(headertest_warning_levels_host /W4 /WX)
else()
set(headertest_warning_levels_host -Wall -Werror)
endif()
# Enable building the nvrtcc project if NVRTC is enabled
if (LIBCUDACXX_TEST_WITH_NVRTC)
add_subdirectory(utils/nvidia/nvrtc)
endif()
add_subdirectory(nvtarget)
add_subdirectory(atomic_codegen)
add_subdirectory(simd_codegen)
add_subdirectory(debugging)

View File

@@ -0,0 +1,150 @@
This file is a partial list of people who have contributed to the LLVM/libc++
project. If you have contributed a patch or made some other contribution to
LLVM/libc++, please submit a patch to this file to add yourself, and it will be
done!
The list is sorted by surname and formatted to allow easy grepping and
beautification by scripts. The fields are: name (N), email (E), web-address
(W), PGP key ID and fingerprint (P), description (D), and snail-mail address
(S).
N: Saleem Abdulrasool
E: compnerd@compnerd.org
D: Minor patches and Linux fixes.
N: Dan Albert
E: danalbert@google.com
D: Android support and test runner improvements.
N: Dimitry Andric
E: dimitry@andric.com
D: Visibility fixes, minor FreeBSD portability patches.
N: Holger Arnold
E: holgerar@gmail.com
D: Minor fix.
N: Ruben Van Boxem
E: vanboxem dot ruben at gmail dot com
D: Initial Windows patches.
N: David Chisnall
E: theraven at theravensnest dot org
D: FreeBSD and Solaris ports, libcxxrt support, some atomics work.
N: Marshall Clow
E: mclow.lists@gmail.com
E: marshall@idio.com
D: C++14 support, patches and bug fixes.
N: Jonathan B Coe
E: jbcoe@me.com
D: Implementation of propagate_const.
N: Glen Joseph Fernandes
E: glenjofe@gmail.com
D: Implementation of to_address.
N: Eric Fiselier
E: eric@efcs.ca
D: LFTS support, patches and bug fixes.
N: Bill Fisher
E: william.w.fisher@gmail.com
D: Regex bug fixes.
N: Matthew Dempsky
E: matthew@dempsky.org
D: Minor patches and bug fixes.
N: Google Inc.
D: Copyright owner and contributor of the CityHash algorithm
N: Howard Hinnant
E: hhinnant@apple.com
D: Architect and primary author of libc++
N: Hyeon-bin Jeong
E: tuhertz@gmail.com
D: Minor patches and bug fixes.
N: Argyrios Kyrtzidis
E: kyrtzidis@apple.com
D: Bug fixes.
N: Bruce Mitchener, Jr.
E: bruce.mitchener@gmail.com
D: Emscripten-related changes.
N: Michel Morin
E: mimomorin@gmail.com
D: Minor patches to is_convertible.
N: Andrew Morrow
E: andrew.c.morrow@gmail.com
D: Minor patches and Linux fixes.
N: Michael Park
E: mcypark@gmail.com
D: Implementation of <variant>.
N: Arvid Picciani
E: aep at exys dot org
D: Minor patches and musl port.
N: Bjorn Reese
E: breese@users.sourceforge.net
D: Initial regex prototype
N: Nico Rieck
E: nico.rieck@gmail.com
D: Windows fixes
N: Jon Roelofs
E: jroelofS@jroelofs.com
D: Remote testing, Newlib port, baremetal/single-threaded support.
N: Jonathan Sauer
D: Minor patches, mostly related to constexpr
N: Craig Silverstein
E: csilvers@google.com
D: Implemented Cityhash as the string hash function on 64-bit machines
N: Richard Smith
D: Minor patches.
N: Joerg Sonnenberger
E: joerg@NetBSD.org
D: NetBSD port.
N: Stephan Tolksdorf
E: st@quanttec.com
D: Minor <atomic> fix
N: Michael van der Westhuizen
E: r1mikey at gmail dot com
N: Larisse Voufo
D: Minor patches.
N: Klaas de Vries
E: klaas at klaasgaaf dot nl
D: Minor bug fix.
N: Zhang Xiongpang
E: zhangxiongpang@gmail.com
D: Minor patches and bug fixes.
N: Xing Xue
E: xingxue@ca.ibm.com
D: AIX port
N: Zhihao Yuan
E: lichray@gmail.com
D: Standard compatibility fixes.
N: Jeffrey Yasskin
E: jyasskin@gmail.com
E: jyasskin@google.com
D: Linux fixes.

View File

@@ -0,0 +1,311 @@
==============================================================================
The LLVM Project is under the Apache License v2.0 with LLVM Exceptions:
==============================================================================
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
---- LLVM Exceptions to the Apache 2.0 License ----
As an exception, if, as a result of your compiling your source code, portions
of this Software are embedded into an Object form of such source code, you
may redistribute such embedded portions in such Object form without complying
with the conditions of Sections 4(a), 4(b) and 4(d) of the License.
In addition, if you combine or link compiled forms of this Software with
software that is licensed under the GPLv2 ("Combined Software") and if a
court of competent jurisdiction determines that the patent provision (Section
3), the indemnity provision (Section 9) or other Section of the License
conflicts with the conditions of the GPLv2, you may retroactively and
prospectively choose to deem waived or otherwise exclude such Section(s) of
the License, but only in their entirety and only with respect to the Combined
Software.
==============================================================================
Software from third parties included in the LLVM Project:
==============================================================================
The LLVM Project contains third party software which is under different license
terms. All such code will be identified clearly using at least one of two
mechanisms:
1) It will be in a separate directory tree with its own `LICENSE.txt` or
`LICENSE` file at the top containing the specific license and restrictions
which apply to that software, or
2) It will contain specific license and restriction terms at the top of every
file.
==============================================================================
Legacy LLVM License (https://llvm.org/docs/DeveloperPolicy.html#legacy):
==============================================================================
The libc++ library is dual licensed under both the University of Illinois
"BSD-Like" license and the MIT license. As a user of this code you may choose
to use it under either license. As a contributor, you agree to allow your code
to be used under both.
Full text of the relevant licenses is included below.
==============================================================================
University of Illinois/NCSA
Open Source License
Copyright (c) 2009-2019 by the contributors listed in CREDITS.TXT
All rights reserved.
Developed by:
LLVM Team
University of Illinois at Urbana-Champaign
http://llvm.org
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal with
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
of the Software, and to permit persons to whom the Software is furnished to do
so, subject to the following conditions:
* Redistributions of source code must retain the above copyright notice,
this list of conditions and the following disclaimers.
* Redistributions in binary form must reproduce the above copyright notice,
this list of conditions and the following disclaimers in the
documentation and/or other materials provided with the distribution.
* Neither the names of the LLVM Team, University of Illinois at
Urbana-Champaign, nor the names of its contributors may be used to
endorse or promote products derived from this Software without specific
prior written permission.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS WITH THE
SOFTWARE.
==============================================================================
Copyright (c) 2009-2014 by the contributors listed in CREDITS.TXT
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.

View File

@@ -0,0 +1,29 @@
//===---------------------------------------------------------------------===//
// Notes relating to various libc++ tasks
//===---------------------------------------------------------------------===//
This file contains notes about various libc++ tasks and processes.
//===---------------------------------------------------------------------===//
// Post-Release TODO
//===---------------------------------------------------------------------===//
These notes contain a list of things that must be done after branching for
an LLVM release.
1. Update _LIBCUDACXX_VERSION in `__config`
2. Update the __cccl_version file.
3. Update the version number in `docs/conf.py`
4. Create ABI lists for the previous release under `lib/abi`
//===---------------------------------------------------------------------===//
// Adding a new header TODO
//===---------------------------------------------------------------------===//
These notes contain a list of things that must be done upon adding a new header
to libc++.
1. Add a test under `test/libcxx` that the header defines `_LIBCUDACXX_VERSION`.
2. Update `test/libcxx/double_include.sh.cpp` to include the new header.
3. Create a submodule in `include/module.modulemap` for the new header.
4. Update the include/CMakeLists.txt file to include the new header.

View File

@@ -0,0 +1,76 @@
This is meant to be a general place to list things that should be done "someday"
CXX Runtime Library Tasks
=========================
* Fix that CMake always link to /usr/lib/libc++abi.dylib on OS X.
* Look into mirroring libsupc++'s typeinfo vtable layout when libsupc++/libstdc++
is used as the runtime library.
* Investigate and document interoperability between libc++ and libstdc++ on
linux. Do this for every supported c++ runtime library.
Atomic Related Tasks
====================
* future should use <atomic> for synchronization.
Test Suite Tasks
================
* Improve the quality and portability of the locale test data.
* Convert failure tests to use Clang Verify.
Filesystem Tasks
================
* P0492r2 - Implement National body comments for Filesystem
* INCOMPLETE - US 25: has_filename() is equivalent to just !empty()
* INCOMPLETE - US 31: Everything is defined in terms of one implicit host system
* INCOMPLETE - US 32: Meaning of 27.10.2.1 unclear
* INCOMPLETE - US 33: Definition of canonical path problematic
* INCOMPLETE - US 34: Are there attributes of a file that are not an aspect of the file system?
* INCOMPLETE - US 35: What synchronization is required to avoid a file system race?
* INCOMPLETE - US 36: Symbolic links themselves are attached to a directory via (hard) links
* INCOMPLETE - US 37: The term "redundant current directory (dot) elements" is not defined
* INCOMPLETE - US 38: Duplicates §17.3.16
* INCOMPLETE - US 39: Remove note: Dot and dot-dot are not directories
* INCOMPLETE - US 40: Not all directories have a parent.
* INCOMPLETE - US 41: The term "parent directory" for a (non-directory) file is unusual
* INCOMPLETE - US 42: Pathname resolution does not always resolve a symlink
* INCOMPLETE - US 43: Concerns about encoded character types
* INCOMPLETE - US 44: Definition of path in terms of a string requires leaky abstraction
* INCOMPLETE - US 45: Generic format portability compromised by unspecified root-name
* INCOMPLETE - US 46: filename can be empty so productions for relative-path are redundant
* INCOMPLETE - US 47: "." and ".." already match the name production
* INCOMPLETE - US 48: Multiple separators are often meaningful in a root-name
* INCOMPLETE - US 49: What does "method of conversion method" mean?
* INCOMPLETE - US 50: 27.10.8.1 ¶ 1.4 largely redundant with ¶ 1.3
* INCOMPLETE - US 51: Failing to add / when appending empty string prevents useful apps
* INCOMPLETE - US 52: remove_filename() postcondition is not by itself a definition
* INCOMPLETE - US 53: remove_filename()'s name does not correspond to its behavior
* INCOMPLETE - US 54: remove_filename() is broken
* INCOMPLETE - US 55: replace_extension()'s use of path as parameter is inappropriate
* INCOMPLETE - US 56: Remove replace_extension()'s conditional addition of period
* INCOMPLETE - US 57: On Windows, absolute paths will sort in among relative paths
* INCOMPLETE - US 58: parent_path() behavior for root paths is useless
* INCOMPLETE - US 59: filename() returning path for single path components is bizarre
* INCOMPLETE - US 60: path("/foo/").filename()==path(".") is surprising
* INCOMPLETE - US 61: Leading dots in filename() should not begin an extension
* INCOMPLETE - US 62: It is important that stem()+extension()==filename()
* INCOMPLETE - US 63: lexically_normal() inconsistently treats trailing "/" but not "/.." as directory
* INCOMPLETE - US 73, CA 2: root-name is effectively implementation defined
* INCOMPLETE - US 74, CA 3: The term "pathname" is ambiguous in some contexts
* INCOMPLETE - US 75, CA 4: Extra flag in path constructors is needed
* INCOMPLETE - US 76, CA 5: root-name definition is over-specified.
* INCOMPLETE - US 77, CA 6: operator/ and other appends not useful if arg has root-name
* INCOMPLETE - US 78, CA 7: Member absolute() in 27.10.4.1 is overspecified for non-POSIX-like O/S
* INCOMPLETE - US 79, CA 8: Some operation functions are overspecified for implementation-defined file types
* INCOMPLETE - US 185: Fold error_code and non-error_code signatures into one signature
* INCOMPLETE - FI 14: directory_entry comparisons are members
* INCOMPLETE - Late 36: permissions() error_code overload should be noexcept
* INCOMPLETE - Late 37: permissions() actions should be separate parameter
* INCOMPLETE - Late 42: resize_file() Postcondition missing argument
Misc Tasks
==========
* Find all sequences of >2 underscores and eradicate them.
* run clang-tidy on libc++
* Document the "conditionally-supported" bits of libc++
* Look at basic_string's move assignment operator, re LWG 2063 and POCMA
* Put a static_assert in std::allocator to deny const/volatile types (LWG 2447)

View File

@@ -0,0 +1,67 @@
add_custom_target(libcudacxx.test.atomics.ptx)
find_program(filecheck "FileCheck")
if (filecheck)
message("-- ${filecheck} found... building atomic codegen tests")
else()
return()
endif()
find_program(cuobjdump "cuobjdump" REQUIRED)
find_program(bash "bash" REQUIRED)
set(atomic_codegen_cuda_arch 80)
set(libcudacxx_atomic_codegen_tests)
if (NOT "NVHPC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
file(GLOB libcudacxx_atomic_codegen_tests "*.cu")
endif()
# For every atomic API compile the TU and check if the SASS/PTX matches the expected result
foreach (test_path IN LISTS libcudacxx_atomic_codegen_tests)
cmake_path(GET test_path FILENAME test_file)
cmake_path(REMOVE_EXTENSION test_file LAST_ONLY OUTPUT_VARIABLE test_name)
add_library(atomic_codegen_${test_name} STATIC "${test_path}")
set_target_properties(
atomic_codegen_${test_name}
PROPERTIES
CUDA_ARCHITECTURES "${atomic_codegen_cuda_arch}"
COMPILE_DEFINITIONS "_CCCL_ATOMIC_UNSAFE_AUTOMATIC_STORAGE=1"
)
# Clang stopped emitting PTX in clang20. Add flags to re-enable it.
if (
CMAKE_CUDA_COMPILER_ID STREQUAL Clang
AND CMAKE_CUDA_COMPILER_VERSION VERSION_GREATER_EQUAL 20
)
target_compile_options(
atomic_codegen_${test_name}
PRIVATE "--cuda-include-ptx=sm_${atomic_codegen_cuda_arch}"
)
endif()
target_compile_options(atomic_codegen_${test_name} PRIVATE "-Wno-comment")
## Important for testing the local headers
target_include_directories(
atomic_codegen_${test_name}
PRIVATE "${libcudacxx_SOURCE_DIR}/include"
)
add_dependencies(libcudacxx.test.atomics.ptx atomic_codegen_${test_name})
# Add output path to object directory
add_custom_command(
TARGET libcudacxx.test.atomics.ptx
POST_BUILD
# gersemi: off
COMMAND
"${CMAKE_CURRENT_SOURCE_DIR}/dump_and_check.bash"
$<TARGET_FILE:atomic_codegen_${test_name}>
"${test_path}"
SM8X
# gersemi: on
)
endforeach()

View File

@@ -0,0 +1,21 @@
#include <cuda/atomic>
__global__ void add_relaxed_device_non_volatile(int* data, int* out, int n)
{
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
*out = ref.fetch_add(n, cuda::std::memory_order_relaxed);
}
/*
; SM8X-LABEL: .target sm_80
; SM8X: .visible .entry [[FUNCTION:_.*add_relaxed_device_non_volatile.*]](
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#RESULT:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_2{{.*}}
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#RESULT]];
; SM8X-NEXT: {{/*[[:space:]] *}}atom.add.relaxed.gpu.s32 %r[[#DEST:]],[%rd[[#ATOM]]],%r[[#INPUT]];{{[[:space:]]/*}}
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
; SM8X-NEXT: ret;
*/

View File

@@ -0,0 +1,23 @@
#include <cuda/atomic>
__global__ void cas_device_relaxed_non_volatile(int* data, int* out, int n)
{
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
ref.compare_exchange_strong(*out, n, cuda::std::memory_order_relaxed);
}
// clang-format off
/*
; SM8X-LABEL: .target sm_80
; SM8X: .visible .entry [[FUNCTION:_.*cas_device_relaxed_non_volatile.*]](
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#EXPECTED:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_2{{.*}}
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#EXPECTED]];
; SM8X-DAG: ld.global.{{b|u}}32 %r[[#LOCALEXP:]], [%rd[[#INPUT]]];
; SM8X-NEXT: {{/*[[:space:]] *}}atom.cas.relaxed.gpu.b32 %r[[#DEST:]],[%rd[[#ATOM]]],%r[[#LOCALEXP]],%r[[#INPUT]];{{[[:space:]]/*}}
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
; SM8X-NEXT: ret;
*/

View File

@@ -0,0 +1,21 @@
#include <cuda/atomic>
__global__ void exch_device_relaxed_non_volatile(int* data, int* out, int n)
{
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
*out = ref.exchange(n, cuda::std::memory_order_relaxed);
}
/*
; SM8X-LABEL: .target sm_80
; SM8X: .visible .entry [[FUNCTION:_.*exch_device_relaxed_non_volatile.*]](
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#EXPECTED:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_2{{.*}}
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#EXPECTED]];
; SM8X-NEXT: {{/*[[:space:]] *}}atom.exch.relaxed.gpu.b32 %r[[#DEST:]],[%rd[[#ATOM]]],%r[[#INPUT]];{{[[:space:]]/*}}
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
; SM8X-NEXT: ret;
*/

View File

@@ -0,0 +1,20 @@
#include <cuda/atomic>
__global__ void load_relaxed_device_non_volatile(int* data, int* out)
{
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
*out = ref.load(cuda::std::memory_order_relaxed);
}
/*
; SM8X-LABEL: .target sm_80
; SM8X: .visible .entry [[FUNCTION:_.*load_relaxed_device_non_volatile.*]](
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#EXPECTED:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#EXPECTED]];
; SM8X-NEXT: {{/*[[:space:]] *}}ld.relaxed.gpu.b32 %r[[#DEST:]],[%rd[[#ATOM]]];{{[[:space:]]/*}}
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
; SM8X-NEXT: ret;
*/

View File

@@ -0,0 +1,18 @@
#include <cuda/atomic>
__global__ void store_relaxed_device_non_volatile(int* data, int in)
{
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
ref.store(in, cuda::std::memory_order_relaxed);
}
/*
; SM8X-LABEL: .target sm_80
; SM8X: .visible .entry [[FUNCTION:_.*store_relaxed_device_non_volatile.*]](
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
; SM8X-NEXT: {{/*[[:space:]] *}}st.relaxed.gpu.b32 [%rd[[#ATOM]]],%r[[#INPUT]];{{[[:space:]]/*}}
; SM8X-NEXT: ret;
*/

View File

@@ -0,0 +1,22 @@
#include <cuda/atomic>
__global__ void sub_relaxed_device_non_volatile(int* data, int* out, int n)
{
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
*out = ref.fetch_sub(n, cuda::std::memory_order_relaxed);
}
/*
; SM8X-LABEL: .target sm_80
; SM8X: .visible .entry [[FUNCTION:_.*sub_relaxed_device_non_volatile.*]](
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#RESULT:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_2{{.*}}
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#RESULT]];
; SM8X-NEXT: neg.s32 %r[[#NEG:]], %r[[#INPUT]];
; SM8X-NEXT: {{/*[[:space:]] *}}atom.add.relaxed.gpu.s32 %r[[#DEST:]],[%rd[[#ATOM]]],%r[[#NEG]];{{[[:space:]]/*}}
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
; SM8X-NEXT: ret;
*/

View File

@@ -0,0 +1,11 @@
#!/usr/bin/env bash
set -euo pipefail
## Usage: dump_and_check test.a test.cu PREFIXES [cuobjdump-mode]
input_archive="${1}"
input_testfile="${2}"
input_prefix="${3}"
dump_mode="${4:---dump-ptx}"
filecheck="${FILECHECK:-FileCheck}"
cuobjdump "${dump_mode}" "${input_archive}" | "${filecheck}" --match-full-lines --check-prefixes="${input_prefix}" "${input_testfile}"

View File

@@ -0,0 +1,10 @@
# Check source code for issues that can be found by pattern matching:
add_test(
NAME libcudacxx.test.cmake.check_source_files
# gersemi: off
COMMAND
"${CMAKE_COMMAND}"
-D "LIBCUDACXX_SOURCE_DIR=${libcudacxx_SOURCE_DIR}"
-P "${CMAKE_CURRENT_LIST_DIR}/check_source_files.cmake"
# gersemi: on
)

View File

@@ -0,0 +1,108 @@
# Check libcudacxx source files for issues that can be detected using pattern
# matching.
#
# This is run as a ctest test named `libcudacxx.test.cmake.check_source_files`,
# or manually with:
# cmake -D "LIBCUDACXX_SOURCE_DIR=<libcudacxx project root>" -P check_source_files.cmake
cmake_minimum_required(VERSION 3.15)
function(count_substrings input search_regex output_var)
string(REGEX MATCHALL "${search_regex}" matches "${input}")
list(LENGTH matches num_matches)
set(${output_var} ${num_matches} PARENT_SCOPE)
endfunction()
set(found_errors 0)
file(
GLOB_RECURSE libcudacxx_srcs
RELATIVE "${LIBCUDACXX_SOURCE_DIR}"
"${LIBCUDACXX_SOURCE_DIR}/include/cuda/*"
"${LIBCUDACXX_SOURCE_DIR}/include/nv/*"
)
# Exclude the imported libc++ headers from the scan. They are not under CCCL's
# direct control and intentionally mirror the upstream libc++ implementation.
list(FILTER libcudacxx_srcs EXCLUDE REGEX "^include/cuda/std/detail/libcxx/")
################################################################################
# stdpar header checks.
# Check all files in libcudacxx to make sure that they aren't including
# <algorithm>, <memory>, or <numeric>, all of which can introduce circular
# dependencies with compilers that integrate CCCL components deeply into
# their C++ standard library implementations.
#
# The following headers should be used instead:
# <algorithm> -> <cuda/std/__host_stdlib/algorithm>
# <memory> -> <cuda/std/__host_stdlib/memory>
# <numeric> -> <cuda/std/__host_stdlib/numeric>
#
set(
stdpar_header_exclusions
include/cuda/std/__host_stdlib/algorithm
include/cuda/std/__host_stdlib/memory
include/cuda/std/__host_stdlib/numeric
)
set(algorithm_regex "#[ \t]*include[ \t]+<algorithm>")
set(memory_regex "#[ \t]*include[ \t]+<memory>")
set(numeric_regex "#[ \t]*include[ \t]+<numeric>")
# Validation check for the above regex pattern:
count_substrings([=[
#include <algorithm>
# include <algorithm>
#include <algorithm>
# include <algorithm>
# include <algorithm> // ...
]=]
${algorithm_regex} valid_count
)
if (NOT valid_count EQUAL 5)
message(
FATAL_ERROR
"Validation of stdpar header regex failed: "
"Matched ${valid_count} times, expected 5."
)
endif()
################################################################################
# Read source files:
foreach (src ${libcudacxx_srcs})
if (IS_DIRECTORY "${LIBCUDACXX_SOURCE_DIR}/${src}")
continue()
endif()
file(READ "${LIBCUDACXX_SOURCE_DIR}/${src}" src_contents)
if (NOT ${src} IN_LIST stdpar_header_exclusions)
count_substrings("${src_contents}" "${algorithm_regex}" algorithm_count)
count_substrings("${src_contents}" "${memory_regex}" memory_count)
count_substrings("${src_contents}" "${numeric_regex}" numeric_count)
if (NOT algorithm_count EQUAL 0)
message(
"'${src}' includes the <algorithm> header. Replace with <cuda/std/__host_stdlib/algorithm>."
)
set(found_errors 1)
endif()
if (NOT memory_count EQUAL 0)
message(
"'${src}' includes the <memory> header. Replace with <cuda/std/__host_stdlib/memory>."
)
set(found_errors 1)
endif()
if (NOT numeric_count EQUAL 0)
message(
"'${src}' includes the <numeric> header. Replace with <cuda/std/__host_stdlib/numeric>."
)
set(found_errors 1)
endif()
endif()
endforeach()
if (NOT found_errors EQUAL 0)
message(FATAL_ERROR "Errors detected.")
endif()

View File

@@ -0,0 +1,202 @@
set(LIBCUDACXX_SUPPORTED_DEBUGGERS lldb gdb)
set(LIBCUDACXX_DEBUGGING_BASE_TARGET libcudacxx.test.debugging)
function(libcudacxx_init_debugger_testing enabled_var)
# Unconditionally create the umbrella target to make CMakePresets easier to setup. If we
# don't have the debuggers available, this target doesn't do anything
add_custom_target(${LIBCUDACXX_DEBUGGING_BASE_TARGET})
# Windows lldb is broken sometimes (see
# https://github.com/llvm/llvm-project/issues/74073) so merely finding it does not mean
# it is functional. In any case, it is good to validate the binary since even a found
# lldb/gdb on other platforms that ends up not working is annoying to work around
function(validator result_var item)
execute_process(
COMMAND ${item} --version
OUTPUT_QUIET
ERROR_QUIET
RESULT_VARIABLE result
TIMEOUT 10
)
if (result EQUAL 0)
set(${result_var} TRUE PARENT_SCOPE)
else()
set(${result_var} FALSE PARENT_SCOPE)
endif()
endfunction()
set(enabled FALSE)
foreach (debugger IN LISTS LIBCUDACXX_SUPPORTED_DEBUGGERS)
string(TOUPPER "${debugger}" DEBUGGER_UPPER)
find_program(
LIBCUDACXX_${DEBUGGER_UPPER}
NAMES "${debugger}"
VALIDATOR validator
)
if (LIBCUDACXX_${DEBUGGER_UPPER})
set(debugger_exe "${LIBCUDACXX_${DEBUGGER_UPPER}}")
message(STATUS "Found ${debugger}: ${debugger_exe}")
execute_process(
COMMAND ${debugger_exe} --version
OUTPUT_VARIABLE version
ERROR_VARIABLE version
OUTPUT_STRIP_TRAILING_WHITESPACE
ERROR_STRIP_TRAILING_WHITESPACE
COMMAND_ERROR_IS_FATAL ANY
)
message(STATUS "${debugger_exe} --version: ${version}")
endif()
set(default OFF)
if (LIBCUDACXX_${DEBUGGER_UPPER})
set(default ON)
endif()
option(
LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING
"Run libcudacxx pretty-printer tests with ${debugger}"
${default}
)
if (
LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING
AND NOT LIBCUDACXX_${DEBUGGER_UPPER}
)
message(
FATAL_ERROR
"LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING is enabled, but ${debugger} was not found"
)
endif()
set(
LIBCUDACXX_${DEBUGGER_UPPER}
"${LIBCUDACXX_${DEBUGGER_UPPER}}"
PARENT_SCOPE
)
set(
LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING
"${LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING}"
PARENT_SCOPE
)
if (LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING)
set(enabled TRUE)
endif()
endforeach()
set(${enabled_var} ${enabled} PARENT_SCOPE)
endfunction()
libcudacxx_init_debugger_testing(testing_enabled)
if (NOT testing_enabled)
return()
endif()
#[=======================================================================[.rst:
libcudacxx_add_pretty_printer_test
---------------------------------
Build a CUDA pretty-printer scenario and register its enabled debugger tests.
The function creates the executable target
``libcudacxx.test.debugging.<NAME>`` with host debug information and optimization
disabled. For each enabled debugger, it registers a serial CTest named
``libcudacxx.test.debugging.<debugger>.<NAME>``. The test uses the debugger's
formatter entry point and the ``<debugger>.expected`` file in the caller's source
directory.
Arguments
^^^^^^^^^
``NAME``
Scenario name used in the executable target, CTest names, and diagnostic output.
``SOURCES``
Source files used to build the CUDA scenario executable. Relative paths are resolved
against the caller's source directory.
``CASES``
Ordered runner arguments describing the debugger stops and expressions. Each case
has the form ``--case <breakpoint> <caller-frame-index> <section-name>
<expression>``.
#]=======================================================================]
function(libcudacxx_add_pretty_printer_test)
set(options)
set(one_value_arguments NAME)
set(multi_value_arguments SOURCES CASES)
cmake_parse_arguments(
pretty_printer
"${options}"
"${one_value_arguments}"
"${multi_value_arguments}"
${ARGN}
)
if (pretty_printer_UNPARSED_ARGUMENTS)
message(
FATAL_ERROR
"Unrecognized arguments: ${pretty_printer_UNPARSED_ARGUMENTS}"
)
endif()
if (NOT pretty_printer_NAME)
message(FATAL_ERROR "libcudacxx_add_pretty_printer_test requires NAME")
endif()
if (NOT pretty_printer_SOURCES)
message(FATAL_ERROR "libcudacxx_add_pretty_printer_test requires SOURCES")
endif()
if (NOT pretty_printer_CASES)
message(FATAL_ERROR "libcudacxx_add_pretty_printer_test requires CASES")
endif()
set(target_name "${LIBCUDACXX_DEBUGGING_BASE_TARGET}.${pretty_printer_NAME}")
cccl_add_executable(
${target_name}
DIALECT 17
NO_CLANG_TIDY
SOURCES ${pretty_printer_SOURCES}
)
target_compile_options(
${target_name}
PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:-g> $<$<COMPILE_LANGUAGE:CUDA>:-O0>
)
target_link_libraries(${target_name} PRIVATE libcudacxx.compiler_interface)
foreach (debugger IN LISTS LIBCUDACXX_SUPPORTED_DEBUGGERS)
string(TOUPPER "${debugger}" DEBUGGER_UPPER)
if (NOT LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING)
continue()
endif()
set(executable "${LIBCUDACXX_${DEBUGGER_UPPER}}")
set(
formatter
"${libcudacxx_SOURCE_DIR}/share/libcudacxx/${debugger}/__init__.py"
)
set(test_name "${target_name}.${debugger}")
add_test(
NAME ${test_name}
COMMAND
# gersemi: off
"${Python_EXECUTABLE}" "${CMAKE_CURRENT_FUNCTION_LIST_DIR}/run_pretty_printer_test.py"
--debugger "${debugger}"
--debugger-executable "${executable}"
--program "$<TARGET_FILE:${target_name}>"
--formatter-init "${formatter}"
--expected "${CMAKE_CURRENT_SOURCE_DIR}/${debugger}.expected"
--output-log "${CMAKE_CURRENT_BINARY_DIR}/${debugger}.log"
${pretty_printer_CASES}
# gersemi: on
)
set_tests_properties(${test_name} PROPERTIES TIMEOUT 60)
endforeach()
endfunction()
add_subdirectory(array)
add_subdirectory(buffer)
add_subdirectory(memory_resource)

View File

@@ -0,0 +1,13 @@
libcudacxx_add_pretty_printer_test(
NAME array
SOURCES source.cu
CASES
# gersemi: off
--case inspect_normal 1 array.normal normal
--case inspect_empty 1 array.empty empty
--case inspect_nested 1 array.nested nested
--case inspect_alias 1 array.alias alias
--case inspect_before_update 1 array.update.before updated_values
--case inspect_after_update 1 array.update.after updated_values
# gersemi: on
)

View File

@@ -0,0 +1,44 @@
=============== array.normal begin ===============
cuda::std::array<int, 3> = {
[0] = -7,
[1] = 0,
[2] = 42
}
=============== array.normal end ===============
=============== array.empty begin ===============
cuda::std::array<int, 0>
=============== array.empty end ===============
=============== array.nested begin ===============
cuda::std::array<cuda::std::array<int, 2>, 2> = {
[0] = cuda::std::array<int, 2> = {
[0] = 13,
[1] = -5
},
[1] = cuda::std::array<int, 2> = {
[0] = 0,
[1] = 88
}
}
=============== array.nested end ===============
=============== array.alias begin ===============
cuda::std::array<int, 4> = {
[0] = -31,
[1] = 17,
[2] = 8,
[3] = -64
}
=============== array.alias end ===============
=============== array.update.before begin ===============
cuda::std::array<int, 3> = {
[0] = 6,
[1] = -91,
[2] = 52
}
=============== array.update.before end ===============
=============== array.update.after begin ===============
cuda::std::array<int, 3> = {
[0] = 3,
[1] = 85,
[2] = -12
}
=============== array.update.after end ===============

View File

@@ -0,0 +1,21 @@
=============== array.normal begin ===============
(cuda::std::array<int, 3>) ([0] = -7, [1] = 0, [2] = 42)
=============== array.normal end ===============
=============== array.empty begin ===============
(cuda::std::array<int, 0>)
=============== array.empty end ===============
=============== array.nested begin ===============
(cuda::std::array<cuda::std::array<int, 2>, 2>) {
[0] = ([0] = 13, [1] = -5)
[1] = ([0] = 0, [1] = 88)
}
=============== array.nested end ===============
=============== array.alias begin ===============
(cuda::std::array<int, 4>) ([0] = -31, [1] = 17, [2] = 8, [3] = -64)
=============== array.alias end ===============
=============== array.update.before begin ===============
(cuda::std::array<int, 3>) ([0] = 6, [1] = -91, [2] = 52)
=============== array.update.before end ===============
=============== array.update.after begin ===============
(cuda::std::array<int, 3>) ([0] = 3, [1] = 85, [2] = -12)
=============== array.update.after end ===============

View File

@@ -0,0 +1,60 @@
// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
//
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include <cuda/std/array>
template <class T>
[[gnu::noinline]] void keep_for_debugger(const T& value)
{
asm volatile("" : : "g"(&value) : "memory");
}
using array_alias = cuda::std::array<int, 4>;
[[gnu::noinline]] void inspect_normal(const cuda::std::array<int, 3>& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_empty(const cuda::std::array<int, 0>& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_nested(const cuda::std::array<cuda::std::array<int, 2>, 2>& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_alias(const array_alias& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_before_update(const cuda::std::array<int, 3>& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_after_update(const cuda::std::array<int, 3>& values)
{
keep_for_debugger(values);
}
int main()
{
const cuda::std::array<int, 3> normal = {-7, 0, 42};
const cuda::std::array<int, 0> empty = {};
const cuda::std::array<cuda::std::array<int, 2>, 2> nested = {{{13, -5}, {0, 88}}};
const array_alias alias = {-31, 17, 8, -64};
cuda::std::array<int, 3> updated_values = {6, -91, 52};
inspect_normal(normal);
inspect_empty(empty);
inspect_nested(nested);
inspect_alias(alias);
inspect_before_update(updated_values);
updated_values = {3, 85, -12};
inspect_after_update(updated_values);
}

View File

@@ -0,0 +1,15 @@
libcudacxx_add_pretty_printer_test(
NAME buffer
SOURCES source.cu
CASES
# gersemi: off
--case inspect_normal 1 buffer.normal normal_values
--case inspect_alias 1 buffer.alias aliased_values
--case inspect_vector 1 buffer.vector.0 "buffer_vector[0]"
--case inspect_vector 1 buffer.vector.1 "buffer_vector[1]"
--case inspect_host_device 1 buffer.host_device host_device_values
--case inspect_empty 1 buffer.empty empty_values
--case inspect_before_update 1 buffer.update.before updated_values
--case inspect_after_update 1 buffer.update.after updated_values
# gersemi: on
)

View File

@@ -0,0 +1,63 @@
=============== buffer.normal begin ===============
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=10, align=4, data=<address> (device) = {
[0] = -56,
[1] = 22,
[2] = 94,
[3] = -13,
[4] = 7,
[5] = 41,
[6] = -82,
[7] = 0,
[8] = 63,
[9] = -5
}
=============== buffer.normal end ===============
=============== buffer.alias begin ===============
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) = {
[0] = 17,
[1] = -31,
[2] = 8,
[3] = 55
}
=============== buffer.alias end ===============
=============== buffer.vector.0 begin ===============
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=3, align=4, data=<address> (device) = {
[0] = -2,
[1] = 4,
[2] = 6
}
=============== buffer.vector.0 end ===============
=============== buffer.vector.1 begin ===============
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=3, align=4, data=<address> (device) = {
[0] = 11,
[1] = -9,
[2] = 27
}
=============== buffer.vector.1 end ===============
=============== buffer.host_device begin ===============
cuda::buffer<int, cuda::mr::device_accessible, cuda::mr::host_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (host/device) = {
[0] = 3,
[1] = 14,
[2] = -15,
[3] = 92
}
=============== buffer.host_device end ===============
=============== buffer.empty begin ===============
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=0, align=4, data=0x0 (device)
=============== buffer.empty end ===============
=============== buffer.update.before begin ===============
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) = {
[0] = 1,
[1] = 2,
[2] = 3,
[3] = 4
}
=============== buffer.update.before end ===============
=============== buffer.update.after begin ===============
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) = {
[0] = -8,
[1] = 13,
[2] = 21,
[3] = -34
}
=============== buffer.update.after end ===============

View File

@@ -0,0 +1,63 @@
=============== buffer.normal begin ===============
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=10, align=4, data=<address> (device) {
[0] = -56
[1] = 22
[2] = 94
[3] = -13
[4] = 7
[5] = 41
[6] = -82
[7] = 0
[8] = 63
[9] = -5
}
=============== buffer.normal end ===============
=============== buffer.alias begin ===============
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) {
[0] = 17
[1] = -31
[2] = 8
[3] = 55
}
=============== buffer.alias end ===============
=============== buffer.vector.0 begin ===============
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=3, align=4, data=<address> (device) {
[0] = -2
[1] = 4
[2] = 6
}
=============== buffer.vector.0 end ===============
=============== buffer.vector.1 begin ===============
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=3, align=4, data=<address> (device) {
[0] = 11
[1] = -9
[2] = 27
}
=============== buffer.vector.1 end ===============
=============== buffer.host_device begin ===============
(cuda::buffer<int, cuda::mr::device_accessible, cuda::mr::host_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (host/device) {
[0] = 3
[1] = 14
[2] = -15
[3] = 92
}
=============== buffer.host_device end ===============
=============== buffer.empty begin ===============
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=0, align=4, data=0x0 (device)
=============== buffer.empty end ===============
=============== buffer.update.before begin ===============
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) {
[0] = 1
[1] = 2
[2] = 3
[3] = 4
}
=============== buffer.update.before end ===============
=============== buffer.update.after begin ===============
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) {
[0] = -8
[1] = 13
[2] = 21
[3] = -34
}
=============== buffer.update.after end ===============

View File

@@ -0,0 +1,102 @@
// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
//
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include <cuda/buffer>
#include <cuda/memory_resource>
#include <cuda/std/array>
#include <cuda/stream>
#include <vector>
#include <cuda_runtime_api.h>
template <class T>
[[gnu::noinline]] void keep_for_debugger(const T& value)
{
asm volatile("" : : "g"(&value) : "memory");
}
[[gnu::noinline]] void inspect_normal(const cuda::device_buffer<int>& values)
{
keep_for_debugger(values);
}
using device_buffer_alias = cuda::buffer<int, cuda::mr::device_accessible>;
[[gnu::noinline]] void inspect_alias(const device_buffer_alias& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_vector(const std::vector<cuda::device_buffer<int>>& values)
{
keep_for_debugger(values[0]);
keep_for_debugger(values[1]);
}
template <class Buffer>
[[gnu::noinline]] void inspect_host_device(const Buffer& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_empty(const cuda::device_buffer<int>& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_before_update(const cuda::device_buffer<int>& values)
{
keep_for_debugger(values);
}
[[gnu::noinline]] void inspect_after_update(const cuda::device_buffer<int>& values)
{
keep_for_debugger(values);
}
int main()
{
constexpr cuda::device_ref device{0};
cuda::stream stream{device};
const cuda::std::array normal_host_values{-56, 22, 94, -13, 7, 41, -82, 0, 63, -5};
const auto normal_values = cuda::make_device_buffer<int>(stream, device, normal_host_values);
const cuda::std::array alias_host_values{17, -31, 8, 55};
const device_buffer_alias aliased_values = cuda::make_device_buffer<int>(stream, device, alias_host_values);
std::vector<cuda::device_buffer<int>> buffer_vector;
buffer_vector.emplace_back(cuda::make_device_buffer<int>(stream, device, cuda::std::array{-2, 4, 6}));
buffer_vector.emplace_back(cuda::make_device_buffer<int>(stream, device, cuda::std::array{11, -9, 27}));
cuda::mr::legacy_managed_memory_resource managed_resource;
const cuda::std::array host_device_host_values{3, 14, -15, 92};
const auto host_device_values = cuda::make_buffer<int>(stream, managed_resource, host_device_host_values);
const auto empty_values = cuda::make_device_buffer<int>(stream, device);
const cuda::std::array initial_updated_host_values{1, 2, 3, 4};
auto updated_values = cuda::make_device_buffer<int>(stream, device, initial_updated_host_values);
stream.sync();
inspect_normal(normal_values);
inspect_alias(aliased_values);
inspect_vector(buffer_vector);
inspect_host_device(host_device_values);
inspect_empty(empty_values);
inspect_before_update(updated_values);
const cuda::std::array replacement_host_values{-8, 13, 21, -34};
if (cudaMemcpyAsync(updated_values.data(),
replacement_host_values.data(),
replacement_host_values.size() * sizeof(*updated_values.data()),
cudaMemcpyDefault,
stream.get())
!= cudaSuccess)
{
return 1;
}
stream.sync();
inspect_after_update(updated_values);
}

View File

@@ -0,0 +1,10 @@
libcudacxx_add_pretty_printer_test(
NAME memory_resource
SOURCES source.cu
CASES
# gersemi: off
--case inspect_device 1 memory_resource.device device_resource
--case inspect_host_device 1 memory_resource.host_device host_device_resource
--case inspect_alias 1 memory_resource.alias aliased_resource
# gersemi: on
)

View File

@@ -0,0 +1,9 @@
=============== memory_resource.device begin ===============
cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>
=============== memory_resource.device end ===============
=============== memory_resource.host_device begin ===============
cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible> @ <address>
=============== memory_resource.host_device end ===============
=============== memory_resource.alias begin ===============
cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>
=============== memory_resource.alias end ===============

View File

@@ -0,0 +1,9 @@
=============== memory_resource.device begin ===============
(const device_resource_type) cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>
=============== memory_resource.device end ===============
=============== memory_resource.host_device begin ===============
(const host_device_resource_type) cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible> @ <address>
=============== memory_resource.host_device end ===============
=============== memory_resource.alias begin ===============
(const resource_alias) cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>
=============== memory_resource.alias end ===============

View File

@@ -0,0 +1,43 @@
// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
//
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include <cuda/memory_resource>
template <class T>
[[gnu::noinline]] void keep_for_debugger(const T& value)
{
asm volatile("" : : "g"(&value) : "memory");
}
using device_resource_type = cuda::mr::any_resource<cuda::mr::device_accessible>;
using host_device_resource_type = cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible>;
using resource_alias = device_resource_type;
[[gnu::noinline]] void inspect_device(const device_resource_type& resource)
{
keep_for_debugger(resource);
}
[[gnu::noinline]] void inspect_host_device(const host_device_resource_type& resource)
{
keep_for_debugger(resource);
}
[[gnu::noinline]] void inspect_alias(const resource_alias& resource)
{
keep_for_debugger(resource);
}
int main()
{
using adapted_resource = cuda::mr::synchronous_resource_adapter<cuda::mr::legacy_managed_memory_resource>;
const adapted_resource managed_resource{cuda::mr::legacy_managed_memory_resource{}};
const device_resource_type device_resource{managed_resource};
const host_device_resource_type host_device_resource{managed_resource};
const resource_alias aliased_resource{managed_resource};
inspect_device(device_resource);
inspect_host_device(host_device_resource);
inspect_alias(aliased_resource);
}

View File

@@ -0,0 +1,790 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""Run a libcudacxx pretty-printer scenario under LLDB or GDB."""
from __future__ import annotations
import argparse
import difflib
import re
import subprocess
import sys
from abc import ABC, abstractmethod
from collections.abc import Sequence
from dataclasses import dataclass
from enum import StrEnum
from pathlib import Path
_MARKER_EDGE = "=" * 15
_MARKER_PATTERN = re.compile(
rf"^{re.escape(_MARKER_EDGE)} (?P<section>.+) (?P<kind>begin|end) {re.escape(_MARKER_EDGE)}$"
)
_LLDB_ECHO_PATTERN = re.compile(r"^\s*\(lldb\)\s")
_GDB_VALUE_PREFIX_PATTERN = re.compile(r"^\s*\$\d+ = ")
_NONZERO_HEX_PATTERN = re.compile(r"\b0x(?!0+\b)[0-9a-fA-F]+\b")
class HarnessError(RuntimeError):
"""Report invalid test input or debugger output."""
class DebuggerError(RuntimeError):
"""Report a debugger launch, timeout, or exit failure."""
class Debugger(StrEnum):
LLDB = "lldb"
GDB = "gdb"
@dataclass(frozen=True)
class Case:
breakpoint: str
frame: int
section: str
expression: str
class CaseAction(argparse.Action):
"""Parse and validate one four-part ``--case`` option."""
def __call__(
self,
parser: argparse.ArgumentParser,
namespace: argparse.Namespace,
values: Sequence[str],
option_string: str | None = None,
) -> None:
"""Append one parsed case to the argument namespace.
Parameters
----------
parser : argparse.ArgumentParser
Parser handling the command line.
namespace : argparse.Namespace
Namespace receiving parsed cases.
values : Sequence[str]
Breakpoint, frame, section, and expression values.
option_string : str or None
Option spelling that supplied the values.
Raises
------
SystemExit
If the frame, breakpoint, section, or expression is invalid, or if
the section name is duplicated.
"""
breakpoint, raw_frame, section, expression = values
try:
frame = int(raw_frame)
except ValueError:
parser.error(f"invalid caller frame index {raw_frame!r}")
if frame < 0:
parser.error(f"caller frame index must be nonnegative: {frame}")
for label, value in (
("breakpoint", breakpoint),
("section", section),
("expression", expression),
):
if not value or "\n" in value or "\r" in value:
parser.error(f"{label} must be a nonempty single line")
cases: list[Case] = getattr(namespace, self.dest) or []
if any(case.section == section for case in cases):
parser.error(f"duplicate section name: {section}")
cases.append(Case(breakpoint, frame, section, expression))
setattr(namespace, self.dest, cases)
def marker(section: str, kind: str) -> str:
"""Build an exact marker for a captured section.
Parameters
----------
section : str
Unique name of the output section.
kind : str
Marker kind, either ``begin`` or ``end``.
Returns
-------
str
Complete marker line expected in debugger output.
"""
return f"{_MARKER_EDGE} {section} {kind} {_MARKER_EDGE}"
class DebuggerAdapter(ABC):
"""Provide debugger-specific command and transcript hooks.
Parameters
----------
executable : Path
Debugger executable path.
formatter_init : Path
Pretty-printer entry-point path.
program : Path
Scenario executable path.
"""
kind: Debugger
def __init__(self, executable: Path, formatter_init: Path, program: Path) -> None:
"""Store paths shared by debugger-specific operations.
Parameters
----------
executable : Path
Debugger executable path.
formatter_init : Path
Pretty-printer entry-point path.
program : Path
Scenario executable path.
"""
self.executable = executable
self.formatter_init = formatter_init
self.program = program
def generate_commands(self, cases: Sequence[Case]) -> str:
"""Generate commands for an ordered list of cases.
Parameters
----------
cases : Sequence[Case]
Cases to execute in order.
Returns
-------
str
Complete debugger command-file contents.
Raises
------
HarnessError
If a completed breakpoint group is reopened later.
"""
closed_stops: set[tuple[str, int]] = set()
previous_stop: tuple[str, int] | None = None
for case in cases:
stop = (case.breakpoint, case.frame)
if stop == previous_stop:
continue
if stop in closed_stops:
raise HarnessError(
f"breakpoint group {case.breakpoint!r} at frame {case.frame} was reopened"
)
if previous_stop is not None:
closed_stops.add(previous_stop)
previous_stop = stop
return self._generate_commands(cases)
@abstractmethod
def _generate_commands(self, cases: Sequence[Case]) -> str:
"""Generate commands for validated cases.
Parameters
----------
cases : Sequence[Case]
Cases to execute in order.
Returns
-------
str
Complete debugger command-file contents.
"""
raise NotImplementedError
@abstractmethod
def command(self, command_file: Path) -> list[str]:
"""Build the debugger subprocess argument list.
Parameters
----------
command_file : Path
Generated debugger command-file path.
Returns
-------
list[str]
Subprocess arguments for this debugger.
"""
raise NotImplementedError
def include_transcript_line(self, line: str) -> bool:
"""Return whether a line inside a marked section should be retained.
Parameters
----------
line : str
Transcript line inside a marked section.
Returns
-------
bool
``True`` when the line belongs in normalized output.
"""
return True
def normalize_line(self, line: str) -> str:
"""Apply debugger-specific normalization to one output line.
Parameters
----------
line : str
Extracted debugger output line.
Returns
-------
str
Line after debugger-specific normalization.
"""
return line
class GDB(DebuggerAdapter):
"""Provide GDB-specific pretty-printer test behavior."""
kind = Debugger.GDB
def _generate_commands(self, cases: Sequence[Case]) -> str:
"""Generate a GDB command file for ordered cases.
Parameters
----------
cases : Sequence[Case]
Cases to execute in order.
Returns
-------
str
Complete GDB command-file contents.
"""
lines = [
"set pagination off",
"set print pretty on",
"set print array-indexes on",
"set debuginfod enabled off",
f"source {self.formatter_init}",
]
seen_breakpoints: set[str] = set()
for case in cases:
if case.breakpoint in seen_breakpoints:
continue
lines.append(f"break {case.breakpoint}")
seen_breakpoints.add(case.breakpoint)
lines.append("run")
previous_stop: tuple[str, int] | None = None
for case in cases:
stop = (case.breakpoint, case.frame)
if previous_stop is not None and stop != previous_stop:
lines.append("continue")
if stop != previous_stop:
lines.append(f"frame {case.frame}")
previous_stop = stop
begin = marker(case.section, "begin")
end = marker(case.section, "end")
expression = repr(case.expression)
lines.extend(
[
f"python print({begin!r})",
"python",
"try:",
f" print(gdb.execute('print ' + {expression}, from_tty=True, to_string=True), end='')",
"except Exception as error:",
" print(error)",
"end",
f"python print({end!r})",
]
)
return "\n".join(lines) + "\n"
def command(self, command_file: Path) -> list[str]:
"""Build the GDB subprocess argument list.
Parameters
----------
command_file : Path
Generated GDB command-file path.
Returns
-------
list[str]
GDB subprocess arguments.
"""
return [
str(self.executable),
"--quiet",
"--batch",
"--nx",
"--command",
str(command_file),
str(self.program),
]
def normalize_line(self, line: str) -> str:
"""Remove GDB value-history prefixes from an output line.
Parameters
----------
line : str
Extracted GDB output line.
Returns
-------
str
Line without a leading ``$N =`` prefix.
"""
return _GDB_VALUE_PREFIX_PATTERN.sub("", line)
class LLDB(DebuggerAdapter):
"""Provide LLDB-specific pretty-printer test behavior."""
kind = Debugger.LLDB
def _generate_commands(self, cases: Sequence[Case]) -> str:
"""Generate an LLDB command file for ordered cases.
Parameters
----------
cases : Sequence[Case]
Cases to execute in order.
Returns
-------
str
Complete LLDB command-file contents.
"""
lines = [f'command script import "{self.formatter_init}"']
seen_breakpoints: set[str] = set()
for case in cases:
if case.breakpoint in seen_breakpoints:
continue
lines.append(f"breakpoint set --name {case.breakpoint}")
seen_breakpoints.add(case.breakpoint)
lines.append("run")
previous_stop: tuple[str, int] | None = None
for case in cases:
stop = (case.breakpoint, case.frame)
if previous_stop is not None and stop != previous_stop:
lines.append("continue")
if stop != previous_stop:
lines.append(f"frame select {case.frame}")
previous_stop = stop
begin = marker(case.section, "begin")
end = marker(case.section, "end")
debugger_command = f"dwim-print -- {case.expression}"
lines.extend(
[
f"script print({begin!r})",
"script result = lldb.SBCommandReturnObject(); "
f"status = lldb.debugger.GetCommandInterpreter().HandleCommand({debugger_command!r}, result); "
"print(result.GetOutput(), end=''); print(result.GetError(), end='')",
f"script print({end!r})",
]
)
return "\n".join(lines) + "\n"
def command(self, command_file: Path) -> list[str]:
"""Build the LLDB subprocess argument list.
Parameters
----------
command_file : Path
Generated LLDB command-file path.
Returns
-------
list[str]
LLDB subprocess arguments.
"""
return [
str(self.executable),
"--batch",
"--no-lldbinit",
"--source",
str(command_file),
str(self.program),
]
def include_transcript_line(self, line: str) -> bool:
"""Exclude LLDB prompt and command-echo lines from marked output.
Parameters
----------
line : str
Transcript line inside a marked section.
Returns
-------
bool
``False`` for LLDB prompt or command-echo lines.
"""
return _LLDB_ECHO_PATTERN.match(line) is None
def extract_sections(
transcript: str, section_order: Sequence[str], debugger: DebuggerAdapter
) -> str:
"""Extract and validate marked sections from a debugger transcript.
Parameters
----------
transcript : str
Complete combined debugger output.
section_order : Sequence[str]
Expected section names in manifest order.
debugger : DebuggerAdapter
Adapter for the debugger that produced the transcript.
Returns
-------
str
Marked sections concatenated in manifest order.
Raises
------
HarnessError
If markers are unexpected, missing, duplicated, nested, mismatched, or
unterminated.
"""
expected_sections = set(section_order)
captured: dict[str, list[str]] = {}
active_section: str | None = None
for line in transcript.splitlines():
match = _MARKER_PATTERN.fullmatch(line)
if not match:
if active_section is None:
continue
if not debugger.include_transcript_line(line):
continue
captured[active_section].append(line)
continue
section = match.group("section")
if section not in expected_sections:
raise HarnessError(f"unexpected marked section: {section}")
kind = match.group("kind")
match kind:
case "begin":
if active_section is not None:
raise HarnessError(
f"nested section {section!r} inside {active_section!r}"
)
if section in captured:
raise HarnessError(f"duplicate marked section: {section}")
captured[section] = [line]
active_section = section
case "end":
if active_section is None:
raise HarnessError(f"end marker without begin marker: {section}")
if active_section != section:
raise HarnessError(
f"mismatched end marker for {section!r}; expected {active_section!r}"
)
captured[section].append(line)
active_section = None
case _:
raise HarnessError(f"invalid marker kind: {kind}")
if active_section is not None:
raise HarnessError(f"unterminated marked section: {active_section}")
missing = [section for section in section_order if section not in captured]
if missing:
raise HarnessError(f"missing marked sections: {', '.join(missing)}")
lines: list[str] = []
for section in section_order:
lines.extend(captured[section])
return "\n".join(lines) + "\n"
def normalize_output(output: str, debugger: DebuggerAdapter) -> str:
"""Normalize unstable values while preserving output structure.
Parameters
----------
output : str
Extracted marked output.
debugger : DebuggerAdapter
Adapter that applies debugger-specific line normalization.
Returns
-------
str
Output with unstable addresses and debugger prefixes normalized.
"""
normalized_lines: list[str] = []
for line in output.splitlines():
line = debugger.normalize_line(line.rstrip())
# Some debuggers may print C++98 style > > for multiple templates.
line = re.sub(r">\s+>", ">>", line)
line = _NONZERO_HEX_PATTERN.sub("<address>", line)
normalized_lines.append(line)
return "\n".join(normalized_lines) + "\n"
def compare_expected(
actual: str, expected: str, debugger: DebuggerAdapter, scenario: str
) -> None:
"""Compare normalized output with its checked-in golden text.
Parameters
----------
actual : str
Normalized debugger output.
expected : str
Checked-in golden output.
debugger : DebuggerAdapter
Adapter for the debugger that produced the output.
scenario : str
Scenario name used in diagnostics.
Raises
------
HarnessError
If actual and expected output differ.
"""
if actual == expected:
return
difference = "".join(
difflib.unified_diff(
expected.splitlines(keepends=True),
actual.splitlines(keepends=True),
fromfile=f"{scenario}/{debugger.kind}.expected",
tofile=f"{scenario}/{debugger.kind}.actual",
)
)
raise HarnessError(
f"{debugger.kind} pretty-printer output mismatch for {scenario}:\n{difference}"
)
def _parse_arguments(arguments: Sequence[str] | None) -> argparse.Namespace:
"""Parse debugger configuration and ordered case definitions.
Parameters
----------
arguments : Sequence[str] or None
Command-line arguments, or ``None`` to use ``sys.argv``.
Returns
-------
argparse.Namespace
Parsed command-line namespace.
"""
parser = argparse.ArgumentParser()
parser.add_argument("--debugger", type=Debugger, choices=Debugger, required=True)
parser.add_argument("--debugger-executable", type=Path, required=True)
parser.add_argument("--program", type=Path, required=True)
parser.add_argument("--formatter-init", type=Path, required=True)
parser.add_argument("--expected", type=Path, required=True)
parser.add_argument("--output-log", type=Path, required=True)
parser.add_argument("--timeout", type=float, default=90.0)
parser.add_argument("--update-expected", action="store_true")
parser.add_argument(
"--case",
dest="cases",
nargs=4,
action=CaseAction,
default=None,
required=True,
)
return parser.parse_args(arguments)
def _create_debugger(args: argparse.Namespace) -> DebuggerAdapter:
"""Create the debugger adapter selected by command-line arguments.
Parameters
----------
args : argparse.Namespace
Parsed runner arguments.
Returns
-------
DebuggerAdapter
Configured adapter for the selected debugger.
Raises
------
HarnessError
If the selected debugger is unsupported.
"""
match args.debugger:
case Debugger.LLDB:
return LLDB(args.debugger_executable, args.formatter_init, args.program)
case Debugger.GDB:
return GDB(args.debugger_executable, args.formatter_init, args.program)
case _:
raise HarnessError(f"unsupported debugger: {args.debugger}")
def _run_debugger(
args: argparse.Namespace, debugger: DebuggerAdapter, command_file: Path
) -> str:
"""Run the debugger and persist its complete transcript.
Parameters
----------
args : argparse.Namespace
Parsed runner arguments.
debugger : DebuggerAdapter
Configured debugger adapter.
command_file : Path
Generated debugger command-file path.
Returns
-------
str
Complete combined debugger output.
Raises
------
DebuggerError
If the debugger cannot launch, times out, or exits with a nonzero status.
OSError
If the transcript cannot be written.
"""
try:
completed = subprocess.run(
debugger.command(command_file),
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
timeout=args.timeout,
)
except subprocess.TimeoutExpired as error:
if isinstance(error.stdout, str):
args.output_log.write_text(error.stdout)
raise DebuggerError(
f"{debugger.kind} timed out after {args.timeout:g} seconds"
) from error
except OSError as error:
raise DebuggerError(f"failed to launch {debugger.kind}: {error}") from error
args.output_log.write_text(completed.stdout)
if completed.returncode != 0:
raise DebuggerError(
f"{debugger.kind} exited with status {completed.returncode}"
)
return completed.stdout
def _match_output(
args: argparse.Namespace,
debugger: DebuggerAdapter,
cases: Sequence[Case],
transcript: str,
) -> None:
"""Extract, normalize, and compare or update debugger output.
Parameters
----------
args : argparse.Namespace
Parsed runner arguments.
debugger : DebuggerAdapter
Configured debugger adapter.
cases : Sequence[Case]
Validated cases in manifest order.
transcript : str
Complete combined debugger output.
Raises
------
HarnessError
If marked output is invalid or differs from the golden.
OSError
If the golden file cannot be read or updated.
"""
extracted = extract_sections(transcript, [case.section for case in cases], debugger)
actual = normalize_output(extracted, debugger)
if args.update_expected:
args.expected.write_text(actual)
return
compare_expected(
actual,
args.expected.read_text(),
debugger,
args.expected.parent.name,
)
def _report_error(
args: argparse.Namespace,
debugger: DebuggerAdapter,
command_file: Path,
error: Exception,
) -> None:
"""Report a debugger or output-matching failure with artifact paths.
Parameters
----------
args : argparse.Namespace
Parsed runner arguments.
debugger : DebuggerAdapter
Configured debugger adapter.
command_file : Path
Generated debugger command-file path.
error : Exception
Failure being reported.
"""
scenario = args.expected.parent.name
print(
f"error: {debugger.kind} pretty-printer test for {scenario}: {error}",
file=sys.stderr,
)
print(f"debugger commands: {command_file}", file=sys.stderr)
if args.output_log.exists():
print(f"complete transcript: {args.output_log}", file=sys.stderr)
def main(arguments: Sequence[str] | None = None) -> int:
"""Run one debugger pretty-printer test from command-line arguments.
Parameters
----------
arguments : Sequence[str] or None
Command-line arguments, or ``None`` to use ``sys.argv``.
Returns
-------
int
Zero on success and one for handled debugger or matching failures.
Raises
------
HarnessError
If case setup is invalid.
OSError
If setup artifacts or golden files cannot be accessed.
"""
args = _parse_arguments(arguments)
debugger = _create_debugger(args)
commands = debugger.generate_commands(args.cases)
command_file = args.output_log.with_suffix(".commands")
args.output_log.parent.mkdir(parents=True, exist_ok=True)
args.output_log.unlink(missing_ok=True)
command_file.write_text(commands)
try:
transcript = _run_debugger(args, debugger, command_file)
except DebuggerError as error:
_report_error(args, debugger, command_file, error)
return 1
try:
_match_output(args, debugger, args.cases, transcript)
except HarnessError as error:
_report_error(args, debugger, command_file, error)
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,302 @@
option(
LIBCUDACXX_TEST_WITH_NVRTC
"Test libcu++ with runtime compilation instead of offline compilation. Only runs device side tests."
OFF
)
###############################################################################
### C2H tests:
cccl_get_c2h()
set(c2h_all_target "libcudacxx.test.c2h_all")
add_custom_target(${c2h_all_target})
if (NOT LIBCUDACXX_TEST_WITH_NVRTC AND NOT CCCL_ENABLE_TILE)
file(
GLOB_RECURSE test_srcs
RELATIVE "${CMAKE_CURRENT_LIST_DIR}"
CONFIGURE_DEPENDS
*.cu
)
function(libcudacxx_add_test target_name_var source)
string(REPLACE "/" "." target_name "${source}")
string(PREPEND target_name "libcudacxx.test.")
string(REGEX REPLACE "\\.[^.]+$" "" target_name "${target_name}")
set(${target_name_var} ${target_name} PARENT_SCOPE)
cccl_add_executable(
${target_name}
ADD_CTEST
NO_METATARGETS
DIALECT ${CMAKE_CUDA_STANDARD}
SOURCES "${source}"
)
target_include_directories(
${target_name}
PRIVATE
"${libcudacxx_SOURCE_DIR}/test/libcudacxx/cuda/ccclrt/common"
"${libcudacxx_SOURCE_DIR}/test/support"
)
target_link_libraries(
${target_name}
PRIVATE #
libcudacxx.compiler_interface
cccl.c2h.main
)
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
target_compile_options(${target_name} PRIVATE "-Wno-attributes")
endif()
add_dependencies(${c2h_all_target} ${target_name})
endfunction()
foreach (test_src IN LISTS test_srcs)
libcudacxx_add_test(test_target "${test_src}")
endforeach()
endif()
###############################################################################
### Lit tests:
macro(pythonize_bool var)
if (${var})
set(${var} True)
else()
set(${var} False)
endif()
endmacro()
cccl_get_cudatoolkit()
cccl_get_dlpack()
get_target_property(CUDA_INCLUDE_DIR CUDA::cudart INTERFACE_INCLUDE_DIRECTORIES)
message(STATUS "Lit enabled CUDA architectures: ${CMAKE_CUDA_ARCHITECTURES}")
if (LIBCUDACXX_TEST_WITH_NVRTC)
# TODO: Use project properties to get path to binary.
# Should also set up dependency on the project when NVRTC is enabled
foreach (include IN ITEMS ${CUDA_INCLUDE_DIR})
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -I'${include}'")
endforeach()
set(
LIBCUDACXX_CUDA_COMPILER
"${CMAKE_BINARY_DIR}/libcudacxx/test/utils/nvidia/nvrtc/nvrtcc"
)
set(LIBCUDACXX_CUDA_COMPILER_ARG1 "")
set(LIBCUDACXX_CUDA_TEST_WITH_NVRTC "True")
# Use the NVRTCC utility to run the built test outputs
set(
LIBCUDACXX_EXECUTOR
"PrefixExecutor(['${LIBCUDACXX_CUDA_COMPILER}'], LocalExecutor())"
)
# Enable 128-bit types for NVRTC
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -device-int128")
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -device-float128")
else() # NOT LIBCUDACXX_TEST_WITH_NVRTC
set(
LIBCUDACXX_FORCE_INCLUDE
"-include ${libcudacxx_SOURCE_DIR}/test/libcudacxx/force_include.h"
)
set(LIBCUDACXX_CUDA_COMPILER "${CMAKE_CUDA_COMPILER}")
set(LIBCUDACXX_CUDA_TEST_WITH_NVRTC "False")
endif()
# enable exceptions and assertions in tests
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -DCCCL_ENABLE_ASSERTIONS")
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -I ${dlpack_SOURCE_DIR}/include")
# Disable dialect deprecation
string(
APPEND LIBCUDACXX_TEST_COMPILER_FLAGS
" -DCCCL_IGNORE_DEPRECATED_CPP_DIALECT"
)
string(
APPEND LIBCUDACXX_TEST_COMPILER_FLAGS
" -DLIBCUDACXX_IGNORE_DEPRECATED_ABI"
)
# enable tile support
if (CCCL_ENABLE_TILE)
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " --enable-tile")
set(LIBCUDACXX_ENABLE_TILE True)
else()
set(LIBCUDACXX_ENABLE_TILE False)
endif()
if ("${CMAKE_CXX_COMPILER_ID}" STREQUAL "GNU")
string(APPEND LIBCUDACXX_TEST_LINKER_FLAGS " -latomic")
endif()
if (NOT MSVC AND NOT ${CMAKE_CUDA_COMPILER_ID} STREQUAL "Clang")
set(
LIBCUDACXX_WARNING_LEVEL
"--compiler-options=-Wall --compiler-options=-Wextra"
)
endif()
if (MSVC)
# We want to use cudaLaunchKernelEx which is guarded by __cplusplus
if ("${CMAKE_CUDA_COMPILER_VERSION}" LESS "12.3.0")
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -Xcompiler=/Zc:__cplusplus")
endif()
# Require the conforming preprocessor
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -Xcompiler=/Zc:preprocessor")
if (MSVC_TOOLSET_VERSION LESS 143)
# winbase.h(9572): warning C5105: macro expansion producing 'defined' has undefined behavior
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -Xcompiler=/wd5105")
endif()
endif()
if (${CMAKE_CUDA_COMPILER_ID} STREQUAL "Clang")
string(
APPEND LIBCUDACXX_TEST_COMPILER_FLAGS
" ${CMAKE_CUDA_FLAGS}"
" -Xclang -fcuda-allow-variadic-functions"
" -Xclang -Wno-unused-parameter"
" -Wno-unknown-cuda-version"
" ${LIBCUDACXX_FORCE_INCLUDE}"
" -I${libcudacxx_SOURCE_DIR}/include"
" ${LIBCUDACXX_WARNING_LEVEL}"
)
string(
APPEND LIBCUDACXX_TEST_LINKER_FLAGS
" ${CMAKE_CUDA_FLAGS}"
" -L${CUDAToolkit_LIBRARY_DIR}"
" -lcuda"
" -lcudart"
)
elseif (${CMAKE_CUDA_COMPILER_ID} STREQUAL "NVIDIA")
string(
APPEND LIBCUDACXX_TEST_COMPILER_FLAGS
" ${LIBCUDACXX_FORCE_INCLUDE}"
" ${LIBCUDACXX_WARNING_LEVEL}"
" -Wno-deprecated-gpu-targets"
)
elseif (${CMAKE_CUDA_COMPILER_ID} STREQUAL "NVHPC")
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -stdpar")
string(APPEND LIBCUDACXX_TEST_LINKER_FLAGS " -stdpar")
endif()
include(AddLLVM)
set(LIBCUDACXX_BINARY_DIR "${CMAKE_CURRENT_BINARY_DIR}")
set(
LIBCUDACXX_TARGET_INFO
"libcudacxx.test.target_info.LocalTI"
CACHE STRING
"TargetInfo to use when setting up test environment."
)
set(
LIBCUDACXX_EXECUTOR
"None"
CACHE STRING
"Executor to use when running tests."
)
set(
LIBCUDACXX_TEST_TIMEOUT
"200"
CACHE STRING
"Enable test timeouts (Default = 200, Off = 0)"
)
set(
AUTO_GEN_COMMENT
"## Autogenerated by libcudacxx configuration.\n# Do not edit!"
)
set(LIBCUDACXX_TEST_STANDARD_VER "c++${CMAKE_CUDA_STANDARD}")
# Pedantic = werror ON and system header pragma OFF (both at their cmake defaults)
if (CCCL_ENABLE_WERROR AND NOT CCCL_ENABLE_PRAGMA_SYSTEM_HEADER)
set(LIBCUDACXX_ENABLE_PEDANTIC_WARNINGS True)
else()
set(LIBCUDACXX_ENABLE_PEDANTIC_WARNINGS False)
endif()
set(lit_site_cfg_path "${CMAKE_CURRENT_BINARY_DIR}/lit.site.cfg")
configure_lit_site_cfg(
"${CMAKE_CURRENT_SOURCE_DIR}/lit.site.cfg.in"
"${lit_site_cfg_path}"
)
add_lit_testsuite(check-cudacxx
"Running libcu++ tests"
"${CMAKE_CURRENT_BINARY_DIR}"
)
find_program(libcudacxx_LIT lit REQUIRED)
set(
libcudacxx_LIT_FLAGS
""
CACHE STRING
"Semi-colon separated list of flags passed to the invocation of lit."
)
message(STATUS "libcudacxx_LIT_FLAGS: ${libcudacxx_LIT_FLAGS}")
if (NOT LIBCUDACXX_TEST_WITH_NVRTC)
# Build but don't run the tests. Used by CI to pre-seed sccache for the test machines.
# Only executed if explicitly requested.
add_custom_target(
libcudacxx.test.lit.precompile
# HACK: There is no way to tell CMake/ninja to always build a target serially,
# so we make this target depend on all other libcudacxx targets to avoid oversubscribing
# the build machine.
# FIXME: This has nasty side effects:
# - It's fragile and must be updated every time we add new targets to libcudacxx
# - It oversubs `-dev` presets that configure libcudacxx alongside other CCCL projects
# - It makes it impossible to just build this target alone since it brings in the world
# See related issue https://github.com/NVIDIA/cccl/issues/6163.
DEPENDS
libcudacxx.test.public_headers
libcudacxx.test.internal_headers
libcudacxx.test.public_headers_host_only
libcudacxx.test.c2h_all
# gersemi: off
COMMAND
"${CMAKE_COMMAND}" -E env "LIBCUDACXX_SITE_CONFIG=${lit_site_cfg_path}"
"${libcudacxx_LIT}"
-vv --no-progress-bar --time-tests
${libcudacxx_LIT_FLAGS}
"-Dexecutor=\"NoopExecutor()\""
"${libcudacxx_SOURCE_DIR}/test/libcudacxx"
# gersemi: on
USES_TERMINAL
)
endif()
# Restricted to avoid oversubscribing the GPU:
set(
libcudacxx_LIT_PARALLEL_LEVEL
8
CACHE STRING
"Parallelism used to run libcudacxx's lit test suite."
)
add_test(
NAME libcudacxx.test.lit
# gersemi: off
COMMAND
"${CMAKE_COMMAND}" -E env "LIBCUDACXX_SITE_CONFIG=${lit_site_cfg_path}"
"${libcudacxx_LIT}"
-vv --no-progress-bar --time-tests
${libcudacxx_LIT_FLAGS}
-j "${libcudacxx_LIT_PARALLEL_LEVEL}"
"${libcudacxx_SOURCE_DIR}/test/libcudacxx"
# gersemi: on
USES_TERMINAL
)
set_tests_properties(
libcudacxx.test.lit
PROPERTIES
# 6hr to match CI timeout
TIMEOUT 21600
RUN_SERIAL TRUE
)

View File

@@ -0,0 +1,70 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "test_macros.h"
#include "utils.h"
template <typename T>
TEST_FUNC __noinline__ void test_global_implicit_property(T ap, cudaAccessProperty cp)
{
// Test implicit conversions
cudaAccessProperty v = ap;
assert(cp == v);
// Test default, copy constructor, and copy-assignent
cuda::access_property o(ap);
cuda::access_property d;
d = ap;
// Test explicit conversion to i64
uint64_t x = (uint64_t) o;
uint64_t y = (uint64_t) d;
assert(x == y);
}
TEST_FUNC __noinline__ void test_global()
{
cuda::access_property o(cuda::access_property::global{});
uint64_t x = (uint64_t) o;
unused(x);
}
TEST_FUNC __noinline__ void test_shared()
{
(void) cuda::access_property::shared{};
}
static_assert(sizeof(cuda::access_property::shared) == 1);
static_assert(sizeof(cuda::access_property::global) == 1);
static_assert(sizeof(cuda::access_property::persisting) == 1);
static_assert(sizeof(cuda::access_property::normal) == 1);
static_assert(sizeof(cuda::access_property::streaming) == 1);
static_assert(sizeof(cuda::access_property) == 8);
static_assert(alignof(cuda::access_property::shared) == 1);
static_assert(alignof(cuda::access_property::global) == 1);
static_assert(alignof(cuda::access_property::persisting) == 1);
static_assert(alignof(cuda::access_property::normal) == 1);
static_assert(alignof(cuda::access_property::streaming) == 1);
static_assert(alignof(cuda::access_property) == 8);
int main(int argc, char** argv)
{
test_global_implicit_property(cuda::access_property::normal{}, cudaAccessProperty::cudaAccessPropertyNormal);
test_global_implicit_property(cuda::access_property::streaming{}, cudaAccessProperty::cudaAccessPropertyStreaming);
test_global_implicit_property(cuda::access_property::persisting{}, cudaAccessProperty::cudaAccessPropertyPersisting);
test_global();
test_shared();
return 0;
}

View File

@@ -0,0 +1,54 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: nvrtc
// error: expression must have a constant value annotated_ptr.h: note #2701-D: attempt to access run-time storage
// UNSUPPORTED: clang-14, gcc-11, gcc-10, gcc-9, gcc-8, gcc-7, msvc-19.29
#include <cuda/annotated_ptr>
#include "test_macros.h"
TEST_FUNC constexpr bool test_constexpr()
{
using namespace cuda;
access_property a{}; // default constructor
access_property b{a}; // copy constructor
access_property c{cuda::std::move(a)}; // move constructor
// user-declared ctor
access_property d1{access_property::global{}};
access_property d2{access_property::normal{}};
access_property d3{access_property::streaming{}};
access_property d4{access_property::persisting{}};
auto p1 = static_cast<cudaAccessProperty>(access_property::normal{});
auto p2 = static_cast<cudaAccessProperty>(access_property::streaming{});
auto p3 = static_cast<cudaAccessProperty>(access_property::persisting{});
// fraction ctor
access_property e1{access_property::normal{}, 1.0f};
access_property e2{access_property::streaming{}, 1.0f};
access_property e3{access_property::persisting{}, 1.0f};
access_property e4{access_property::normal{}, 1.0f, access_property::streaming{}};
access_property e5{access_property::persisting{}, 1.0f, access_property::streaming{}};
b = a; // copy assignment
b = cuda::std::move(a); // move assignment
auto value = static_cast<uint64_t>(a);
unused(p1, p2, p3, b, c, d1, d2, d3, d4, e1, e2, e3, e4, e5, value);
return true;
}
int main(int, char**)
{
static_assert(test_constexpr());
return 0;
}

View File

@@ -0,0 +1,26 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "utils.h"
TEST_FUNC __noinline__ void test_access_property_fail()
{
cuda::access_property o = cuda::access_property::normal{};
// Test implicit conversion fails
std::uint64_t x;
x = o;
unused(o);
}
int main(int argc, char** argv)
{
test_access_property_fail();
return 0;
}

View File

@@ -0,0 +1,148 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: nvrtc
// UNSUPPORTED: pre-sm-80
#include <cuda/annotated_ptr>
#include <cuda/cmath>
#include <cuda/std/type_traits>
#include "test_macros.h"
template <typename Prop>
TEST_DEVICE_FUNC constexpr cuda::__l2_evict_t to_enum()
{
if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::normal>)
{
return cuda::__l2_evict_t::_L2_Evict_Normal_Demote;
}
else if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::streaming>)
{
return cuda::__l2_evict_t::_L2_Evict_First;
}
else if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::persisting>)
{
return cuda::__l2_evict_t::_L2_Evict_Last;
}
else // if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::global>)
{
return cuda::__l2_evict_t::_L2_Evict_Unchanged;
}
}
//----------------------------------------------------------------------------------------------------------------------
// test range
template <typename Primary, typename Secondary = void, int I = 1>
TEST_DEVICE_FUNC void test_fraction_constexpr()
{
if constexpr (I > 16)
{
return;
}
else
{
constexpr auto fraction = static_cast<float>(I) * (1.0f / 16.0f);
auto policy = cuda::__createpolicy_fraction(to_enum<Primary>(), to_enum<Secondary>(), fraction);
if constexpr (cuda::std::is_void_v<Secondary>)
{
constexpr cuda::access_property property{Primary{}, fraction};
assert(static_cast<uint64_t>(property) == policy);
}
else
{
constexpr cuda::access_property property{Primary{}, fraction, Secondary{}};
assert(static_cast<uint64_t>(property) == policy);
}
test_fraction_constexpr<Primary, Secondary, I + 1>();
}
}
__global__ void test_fraction()
{
test_fraction_constexpr<cuda::access_property::normal>();
test_fraction_constexpr<cuda::access_property::streaming>();
test_fraction_constexpr<cuda::access_property::persisting>();
test_fraction_constexpr<cuda::access_property::normal, cuda::access_property::streaming>();
test_fraction_constexpr<cuda::access_property::persisting, cuda::access_property::streaming>();
}
//----------------------------------------------------------------------------------------------------------------------
// test range
template <typename Primary, typename Secondary>
__global__ void test_range_kernel(void* ptr, uint64_t property, uint32_t primary_size, uint32_t total_size)
{
auto policy = __createpolicy_range(to_enum<Primary>(), to_enum<Secondary>(), ptr, primary_size, total_size);
if (static_cast<uint64_t>(property) != policy)
{
printf(" primary_size = %u, total_size = %u\n", primary_size, total_size);
printf(" primary = %u, secondary = %u\n", (int) to_enum<Primary>(), (int) to_enum<Secondary>());
printf(" 0x%llX vs 0x%llX\n", static_cast<unsigned long long>(policy), static_cast<unsigned long long>(property));
}
assert(static_cast<uint64_t>(property) == policy);
}
template <typename Primary, typename Secondary = void>
void test_range_launch(void* ptr, uint32_t primary_size, uint32_t total_size)
{
cuda::access_property property;
if constexpr (cuda::std::is_void_v<Secondary>)
{
property = cuda::access_property{ptr, primary_size, total_size, Primary{}};
}
else
{
property = cuda::access_property{ptr, primary_size, total_size, Primary{}, Secondary{}};
}
test_range_kernel<Primary, Secondary><<<1, 1>>>(ptr, static_cast<uint64_t>(property), primary_size, total_size);
}
void test_range()
{
int* ptr = nullptr;
ptr++;
for (uint32_t total_size = 1, i = 0; i <= 31; i++, total_size <<= 1)
{
for (uint32_t primary_size = 1, j = 0; j <= i; j++, primary_size <<= 1)
{
test_range_launch<cuda::access_property::normal>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::streaming>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::persisting>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::global, cuda::access_property::streaming>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::normal, cuda::access_property::streaming>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::streaming, cuda::access_property::streaming>(
ptr, primary_size, total_size);
test_range_launch<cuda::access_property::persisting, cuda::access_property::streaming>(
ptr, primary_size, total_size);
}
}
// PTX createpolicy_range and access_property behaviors don't match (for now)
// uint32_t primary_size = 0xFFFFFFFF;
// uint32_t total_size = 0xFFFFFFFF;
// test_range_launch<cuda::access_property::normal>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::streaming>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::persisting>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::global, cuda::access_property::streaming>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::normal, cuda::access_property::streaming>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::persisting, cuda::access_property::streaming>(ptr, primary_size,
// total_size);
}
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST, (test_range();))
NV_IF_TARGET(NV_IS_HOST, (test_fraction<<<1, 1>>>();))
return 0;
}

View File

@@ -0,0 +1,164 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "utils.h"
static_assert(sizeof(cuda::annotated_ptr<int, cuda::access_property::global>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<char, cuda::access_property::global>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::global>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::persisting>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::normal>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::streaming>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property>) == 2 * sizeof(uintptr_t),
"annotated_ptr<T,access_property> must be 2 * pointer size");
// NOTE: we could make these smaller in the future (e.g. 32-bit) but that would be an ABI breaking change:
static_assert(sizeof(cuda::annotated_ptr<int, cuda::access_property::shared>) == sizeof(uintptr_t),
"annotated_ptr<T, shared> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<char, cuda::access_property::shared>) == sizeof(uintptr_t),
"annotated_ptr<T, shared> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::shared>) == sizeof(uintptr_t),
"annotated_ptr<T, shared> must be pointer size");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::global>) == alignof(int*),
"annotated_ptr must align with int*");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::persisting>) == alignof(int*),
"annotated_ptr must align with int*");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::normal>) == alignof(int*),
"annotated_ptr must align with int*");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::streaming>) == alignof(int*),
"annotated_ptr must align with int*");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property>) == alignof(int*),
"annotated_ptr must align with int*");
// NOTE: we could lower the alignment in the future but that would be an ABI breaking change:
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::shared>) == alignof(int*),
"annotated_ptr must align with int*");
#define N 128
struct S
{
int x;
TEST_FUNC S& operator=(int o)
{
this->x = o;
return *this;
}
};
template <typename In, typename T>
TEST_FUNC __noinline__ void test_read_access(In i, T* r)
{
assert(i);
assert(i - i == 0);
assert((bool) i);
const In o = i;
// assert(i->x == 0); // FAILS with shmem
// assert(o->x == 0); // FAILS with shmem
for (int n = 0; n < N; ++n)
{
assert(i[n].x == n);
assert(&i[n] == &i[n]);
assert(&i[n] == &r[n]);
assert(o[n].x == n);
assert(&o[n] == &o[n]);
assert(&o[n] == &r[n]);
}
}
template <typename In>
TEST_FUNC __noinline__ void test_write_access(In i)
{
assert(i);
assert((bool) i);
const In o = i;
for (int n = 0; n < N; ++n)
{
i[n].x = 2 * n;
assert(i[n].x == 2 * n);
assert(i[n].x == 2 * n);
i[n].x = n;
o[n].x = 2 * n;
assert(o[n].x == 2 * n);
assert(o[n].x == 2 * n);
o[n].x = n;
}
}
TEST_FUNC __noinline__ void all_tests()
{
S* arr = global_alloc<S, N>();
test_read_access(cuda::annotated_ptr<S, cuda::access_property::normal>(arr), arr);
test_read_access(cuda::annotated_ptr<S, cuda::access_property::streaming>(arr), arr);
test_read_access(cuda::annotated_ptr<S, cuda::access_property::persisting>(arr), arr);
test_read_access(cuda::annotated_ptr<S, cuda::access_property::global>(arr), arr);
test_read_access(cuda::annotated_ptr<S, cuda::access_property>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::normal>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::streaming>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::persisting>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::global>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::normal>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::streaming>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::persisting>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::global>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::normal>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::streaming>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::persisting>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::global>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property>(arr), arr);
test_write_access(cuda::annotated_ptr<S, cuda::access_property::normal>(arr));
test_write_access(cuda::annotated_ptr<S, cuda::access_property::streaming>(arr));
test_write_access(cuda::annotated_ptr<S, cuda::access_property::persisting>(arr));
test_write_access(cuda::annotated_ptr<S, cuda::access_property::global>(arr));
test_write_access(cuda::annotated_ptr<S, cuda::access_property>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::normal>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::streaming>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::persisting>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::global>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property>(arr));
NV_IF_TARGET(
NV_IS_DEVICE,
(S* sarr = shared_alloc<S, N>(); // Allocating shared memory is only supported on device
test_read_access(cuda::annotated_ptr<S, cuda::access_property::shared>(sarr), sarr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::shared>(sarr), sarr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::shared>(sarr), sarr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::shared>(sarr), sarr);
test_write_access(cuda::annotated_ptr<S, cuda::access_property::shared>(sarr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::shared>(sarr));))
}
int main(int argc, char** argv)
{
all_tests();
return 0;
}

View File

@@ -0,0 +1,157 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "test_macros.h"
#include "utils.h"
TEST_DEVICE_FUNC void annotated_ptr_timing_dev(int* in, int* out)
{
cuda::access_property ap(cuda::access_property::persisting{});
// Retrieve global id
int i = blockIdx.x * blockDim.x + threadIdx.x;
cuda::annotated_ptr<int, cuda::access_property> in_ann{in, ap};
cuda::annotated_ptr<int, cuda::access_property> out_ann{out, ap};
DPRINTF("&out[i]:%p = &in[i]:%p for i = %d\n", &out[i], &in[i], i);
DPRINTF("&out[i]:%p = &in_ann[i]:%p for i = %d\n", &out_ann[i], &in_ann[i], i);
out_ann[i] = in_ann[i];
};
__global__ void annotated_ptr_timing(int* in, int* out)
{
annotated_ptr_timing_dev(in, out);
}
TEST_DEVICE_FUNC void ptr_timing_dev(int* in, int* out)
{
// Retrieve global id
int i = blockIdx.x * blockDim.x + threadIdx.x;
DPRINTF("&out[i]:%p = &in[i]:%p for i = %d\n", &out[i], &in[i], i);
out[i] = in[i];
};
__global__ void ptr_timing(int* in, int* out)
{
ptr_timing_dev(in, out);
};
TEST_FUNC __noinline__ void bench()
{
#ifndef __CUDA_ARCH__
static const size_t ARR_SZ = 1 << 22;
static const size_t THREAD_CNT = 128;
static const size_t BLOCK_CNT = ARR_SZ / THREAD_CNT;
const dim3 threads(THREAD_CNT, 1, 1), blocks(BLOCK_CNT, 1, 1);
cudaEvent_t start, stop;
#else
static const size_t ARR_SZ = 1 << 10;
#endif
int* arr0 = nullptr;
int* arr1 = nullptr;
float annotated_time = 0.f, pointer_time = 0.f;
#ifdef __CUDA_ARCH__
arr0 = (int*) malloc(ARR_SZ * sizeof(int));
arr1 = (int*) malloc(ARR_SZ * sizeof(int));
#else
assert_rt(cudaMallocManaged((void**) &arr0, ARR_SZ * sizeof(int)));
assert_rt(cudaMallocManaged((void**) &arr1, ARR_SZ * sizeof(int)));
assert_rt(cudaDeviceSynchronize());
#endif
#ifdef __CUDA_ARCH__
ptr_timing_dev(arr0, arr1);
#else
ptr_timing<<<blocks, threads>>>(arr0, arr1);
assert_rt(cudaDeviceSynchronize());
#endif
for (size_t i = 0; i < ARR_SZ; ++i)
{
arr0[i] = static_cast<int>(i);
arr1[i] = 0;
}
#ifdef __CUDA_ARCH__
ptr_timing_dev(arr0, arr1);
#else
assert_rt(cudaDeviceSynchronize());
assert_rt(cudaEventCreate(&start));
assert_rt(cudaEventCreate(&stop));
assert_rt(cudaEventRecord(start));
ptr_timing<<<blocks, threads>>>(arr0, arr1);
assert_rt(cudaEventRecord(stop));
assert_rt(cudaEventSynchronize(stop));
assert_rt(cudaEventElapsedTime(&pointer_time, start, stop));
assert_rt(cudaEventDestroy(start));
assert_rt(cudaEventDestroy(stop));
assert_rt(cudaDeviceSynchronize());
for (size_t i = 0; i < ARR_SZ; ++i)
{
if (arr1[i] != (int) i)
{
DPRINTF("arr1[%d] == %d, should be:%d\n", i, arr1[i], i);
assert(arr1[i] == static_cast<int>(i));
}
arr1[i] = 0;
}
#endif
NV_IF_ELSE_TARGET(NV_IS_DEVICE,
(annotated_ptr_timing_dev(arr0, arr1);),
(assert_rt(cudaDeviceSynchronize()); annotated_ptr_timing<<<blocks, threads>>>(arr0, arr1);
assert_rt(cudaDeviceSynchronize());))
for (size_t i = 0; i < ARR_SZ; ++i)
{
arr0[i] = static_cast<int>(i);
arr1[i] = 0;
}
NV_IF_ELSE_TARGET(
NV_IS_DEVICE,
(annotated_ptr_timing_dev(arr0, arr1);),
(assert_rt(cudaDeviceSynchronize()); assert_rt(cudaEventCreate(&start)); assert_rt(cudaEventCreate(&stop));
assert_rt(cudaEventRecord(start));
annotated_ptr_timing<<<blocks, threads>>>(arr0, arr1);
assert_rt(cudaEventRecord(stop));
assert_rt(cudaEventSynchronize(stop));
assert_rt(cudaEventElapsedTime(&annotated_time, start, stop));
assert_rt(cudaEventDestroy(start));
assert_rt(cudaEventDestroy(stop));
assert_rt(cudaDeviceSynchronize());
for (size_t i = 0; i < ARR_SZ; ++i) {
if (arr1[i] != (int) i)
{
DPRINTF("arr1[%d] == %d, should be:%d\n", i, arr1[i], i);
assert(arr1[i] == static_cast<int>(i));
}
arr1[i] = 0;
}))
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr0); free(arr1);), (assert_rt(cudaFree(arr0)); assert_rt(cudaFree(arr1));))
printf("array(ms):%f, arrotated_ptr(ms):%f\n", pointer_time, annotated_time);
}
int main(int argc, char** argv)
{
NV_IF_TARGET(NV_IS_DEVICE, (bench();))
return 0;
}

View File

@@ -0,0 +1,65 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: nvrtc
// error: expression must have a constant value annotated_ptr.h: note #2701-D: attempt to access run-time storage
// UNSUPPORTED: clang-14, gcc-12, gcc-11, gcc-10, gcc-9, gcc-8, gcc-7, msvc-19.29
// UNSUPPORTED: msvc && nvcc-12.0
#include <cuda/annotated_ptr>
#include "test_macros.h"
TEST_FUNC constexpr bool test_public_methods()
{
using namespace cuda;
using annotated_ptr = cuda::annotated_ptr<const int, access_property::persisting>;
using annotated_smem_ptr [[maybe_unused]] = cuda::annotated_ptr<const int, access_property::shared>;
annotated_ptr a{}; // default constructor
annotated_ptr b{a}; // copy constructor
annotated_ptr c{cuda::std::move(a)}; // move constructor
NV_IF_TARGET(NV_IS_DEVICE, (annotated_smem_ptr d{nullptr};)) // pointer constructor
b = a; // copy assignment
b = cuda::std::move(a); // move assignment
auto diff = a - b;
auto pred = static_cast<bool>(a);
auto prop = a.__property();
unused(c);
unused(diff);
unused(pred);
unused(prop);
return true;
}
TEST_FUNC constexpr bool test_interleave_values()
{
using namespace cuda;
constexpr auto normal = __l2_interleave(__l2_evict_t::_L2_Evict_Unchanged, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
constexpr auto streaming = __l2_interleave(__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
constexpr auto persisting = __l2_interleave(__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
constexpr auto normal_demote =
__l2_interleave(__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
static_assert(normal == __l2_interleave_normal);
static_assert(streaming == __l2_interleave_streaming);
static_assert(persisting == __l2_interleave_persisting);
static_assert(normal_demote == __l2_interleave_normal_demote);
return true;
}
int main(int, char**)
{
static_assert(test_interleave_values());
static_assert(test_public_methods());
return 0;
}

View File

@@ -0,0 +1,84 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "utils.h"
template <typename T, typename P>
TEST_FUNC __noinline__ void test_ctor(T* ptr)
{
// default ctor, cpy and cpy assignment
cuda::annotated_ptr<T, P> def;
{
cuda::annotated_ptr<T, P> temp;
temp = def;
unused(temp);
}
cuda::annotated_ptr<T, P> other(def);
unused(other);
// from ptr
cuda::annotated_ptr<T, P> a(ptr);
assert(a);
// cpy ctor & assign to cv
cuda::annotated_ptr<const T, P> c(def);
cuda::annotated_ptr<volatile T, P> d(def);
cuda::annotated_ptr<const volatile T, P> e(def);
c = def;
d = def;
e = def;
// from c|v to c|v|cv
cuda::annotated_ptr<const T, P> f(c);
cuda::annotated_ptr<volatile T, P> g(d);
cuda::annotated_ptr<const volatile T, P> h(e);
f = c;
g = d;
h = e;
unused(f, g, h);
// to cv
cuda::annotated_ptr<const volatile T, P> i(c);
cuda::annotated_ptr<const volatile T, P> j(d);
i = c;
j = d;
}
template <typename T, typename P>
TEST_FUNC __noinline__ void test_global_ctor()
{
T* rp = nullptr;
rp++;
test_ctor<T, P>(rp);
// from ptr + prop
P p;
cuda::annotated_ptr<T, cuda::access_property> a(rp, p);
cuda::annotated_ptr<const T, cuda::access_property> b(rp, p);
cuda::annotated_ptr<volatile T, cuda::access_property> c(rp, p);
cuda::annotated_ptr<const volatile T, cuda::access_property> d(rp, p);
}
TEST_FUNC __noinline__ void test_global_ctors()
{
test_global_ctor<int, cuda::access_property::normal>();
test_global_ctor<int, cuda::access_property::streaming>();
test_global_ctor<int, cuda::access_property::persisting>();
test_global_ctor<int, cuda::access_property::global>();
test_global_ctor<int, cuda::access_property>();
NV_IF_TARGET(NV_IS_DEVICE, (__shared__ int smem_value; test_ctor<int, cuda::access_property::shared>(&smem_value);))
}
int main(int argc, char** argv)
{
test_global_ctors();
return 0;
}

View File

@@ -0,0 +1,49 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "utils.h"
template <typename T, typename P>
TEST_FUNC __noinline__ void test_ctor()
{
// default ctor, cpy and cpy assignment
cuda::annotated_ptr<T, P> def;
def = def;
cuda::annotated_ptr<T, P> other(def);
// from ptr
T* rp = nullptr;
cuda::annotated_ptr<T, P> a(rp);
assert(!a);
// cpy ctor & assign to cv
cuda::annotated_ptr<const T, P> c(def);
cuda::annotated_ptr<volatile T, P> d(def);
cuda::annotated_ptr<const volatile T, P> e(def);
c = e; // FAIL
d = d; // FAIL
}
template <typename T, typename P>
TEST_FUNC __noinline__ void test_global_ctor()
{
test_ctor<T, P>();
}
TEST_FUNC __noinline__ void test_global_ctors()
{
test_global_ctor<int, cuda::access_property::normal>();
}
int main(int argc, char** argv)
{
test_global_ctors();
return 0;
}

View File

@@ -0,0 +1,23 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "utils.h"
int main(int argc, char** argv)
{
cuda::access_property ap(cuda::access_property::persisting{});
int* array0 = new int[9];
cuda::annotated_ptr<int, cuda::access_property> array_anno_ptr{array0, ap};
cuda::annotated_ptr<int, cuda::access_property::shared> shared_ptr;
array_anno_ptr = shared_ptr; // fail to compile, as expected
return 0;
}

View File

@@ -0,0 +1,23 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "utils.h"
int main(int argc, char** argv)
{
cuda::access_property ap(cuda::access_property::persisting{});
int* array0 = new int[9];
cuda::annotated_ptr<int, cuda::access_property> array_anno_ptr{array0, ap};
cuda::annotated_ptr<int, cuda::access_property::shared> shared_ptr;
array_anno_ptr = shared_ptr; // fail to compile, as expected
return 0;
}

View File

@@ -0,0 +1,26 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// NVRTC does not do host side testing
// UNSUPPORTED: nvrtc
#include "utils.h"
TEST_FUNC static void fails_from_host()
{
int a;
__nv_associate_access_property(&a, uint64_t{0});
}
int main(int argc, char** argv)
{
// calling from host needs to fail and kill the app
fails_from_host();
return 0;
}

View File

@@ -0,0 +1,40 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "test_macros.h"
#include "utils.h"
template <typename T, typename U>
TEST_DEVICE_FUNC __noinline__ void shared_mem_test_dev()
{
T* smem = shared_alloc<T, 128>();
smem[10] = 42;
cuda::annotated_ptr<U, cuda::access_property::shared> p{smem + 10};
assert(*p == 42);
}
TEST_DEVICE_FUNC __noinline__ void test_all()
{
shared_mem_test_dev<int, int>();
shared_mem_test_dev<int, const int>();
shared_mem_test_dev<int, volatile int>();
shared_mem_test_dev<int, const volatile int>();
}
int main(int argc, char** argv)
{
NV_IF_TARGET(NV_IS_DEVICE, (test_all();))
return 0;
}

View File

@@ -0,0 +1,60 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "utils.h"
constexpr size_t array_size = 128;
template <typename T, typename P>
TEST_FUNC __noinline__ void test(P ap)
{
T* arr = global_alloc<T, array_size>();
cuda::apply_access_property(arr, array_size * sizeof(T), ap);
for (size_t i = 0; i < array_size; ++i)
{
assert(static_cast<size_t>(arr[i]) == i);
}
dealloc<T>(arr);
}
template <typename T, typename P>
TEST_FUNC __noinline__ void test_aligned(P ap)
{
T* arr = global_alloc<T, array_size>();
cuda::apply_access_property(arr, cuda::aligned_size_t<sizeof(T)>(array_size * sizeof(T)), ap);
for (size_t i = 0; i < array_size; ++i)
{
assert(static_cast<size_t>(arr[i]) == i);
}
dealloc<T>(arr);
}
TEST_FUNC __noinline__ void test_all()
{
test<int>(cuda::access_property::normal{});
test<int>(cuda::access_property::persisting{});
test_aligned<int>(cuda::access_property::normal{});
test_aligned<int>(cuda::access_property::persisting{});
}
int main(int argc, char** argv)
{
test_all();
return 0;
}

View File

@@ -0,0 +1,59 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "utils.h"
#define ARR_SZ 128
template <typename T, typename P>
TEST_FUNC __noinline__ void test(P ap)
{
T* arr = global_alloc<T, ARR_SZ>();
arr = cuda::associate_access_property(arr, ap);
for (int i = 0; i < ARR_SZ; ++i)
{
assert(arr[i] == i);
}
dealloc<T>(arr);
}
template <typename T, typename P>
TEST_FUNC __noinline__ void test_shared(P ap)
{
T* arr = shared_alloc<T, ARR_SZ>();
arr = cuda::associate_access_property(arr, ap);
for (int i = 0; i < ARR_SZ; ++i)
{
assert(arr[i] == i);
}
}
TEST_FUNC __noinline__ void test_all()
{
test<int>(cuda::access_property::normal{});
test<int>(cuda::access_property::persisting{});
test<int>(cuda::access_property::streaming{});
test<int>(cuda::access_property::global{});
test<int>(cuda::access_property{});
NV_IF_TARGET(NV_IS_DEVICE, (test_shared<int>(cuda::access_property::shared{});))
}
int main(int argc, char** argv)
{
test_all();
return 0;
}

View File

@@ -0,0 +1,121 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: pre-sm-70
#include <cooperative_groups.h>
#include "utils.h"
// TODO: global-shared
// TODO: read const
TEST_FUNC __noinline__ void test_memcpy_async()
{
size_t ARR_SZ = 1 << 10;
int* arr0 = nullptr;
int* arr1 = nullptr;
cuda::access_property ap(cuda::access_property::persisting{});
cuda::barrier<cuda::thread_scope_system> bar0, bar1, bar2, bar3;
init(&bar0, 1);
init(&bar1, 1);
init(&bar2, 1);
init(&bar3, 1);
NV_IF_ELSE_TARGET(
NV_IS_DEVICE,
(arr0 = (int*) malloc(ARR_SZ * sizeof(int)); arr1 = (int*) malloc(ARR_SZ * sizeof(int));),
(assert_rt(cudaMallocManaged((void**) &arr0, ARR_SZ * sizeof(int)));
assert_rt(cudaMallocManaged((void**) &arr1, ARR_SZ * sizeof(int)));
assert_rt(cudaDeviceSynchronize());))
cuda::annotated_ptr<int, cuda::access_property> ann0{arr0, ap};
cuda::annotated_ptr<int, cuda::access_property> ann1{arr1, ap};
// cuda::annotated_ptr<const int, cuda::access_property> cann0{arr0, ap};
for (size_t i = 0; i < ARR_SZ; ++i)
{
arr0[i] = static_cast<int>(i);
arr1[i] = 0;
}
cuda::memcpy_async(ann1, ann0, ARR_SZ * sizeof(int), bar0);
// cuda::memcpy_async(ann1, cann0, ARR_SZ * sizeof(int), bar0);
bar0.arrive_and_wait();
for (size_t i = 0; i < ARR_SZ; ++i)
{
if (arr1[i] != (int) i)
{
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
assert(arr1[i] == static_cast<int>(i));
}
arr1[i] = 0;
}
cuda::memcpy_async(arr1, ann0, ARR_SZ * sizeof(int), bar1);
// cuda::memcpy_async(arr1, cann0, ARR_SZ * sizeof(int), bar1);
bar1.arrive_and_wait();
for (size_t i = 0; i < ARR_SZ; ++i)
{
if (arr1[i] != (int) i)
{
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
assert(arr1[i] == static_cast<int>(i));
}
arr1[i] = 0;
}
NV_IF_TARGET(
NV_IS_DEVICE,
(
auto group = cooperative_groups::this_thread_block();
cuda::memcpy_async(group, ann1, ann0, ARR_SZ * sizeof(int), bar2);
// cuda::memcpy_async(group, ann1, cann0, ARR_SZ * sizeof(int), bar2);
bar2.arrive_and_wait();
for (size_t i = 0; i < ARR_SZ; ++i) {
if (arr1[i] != (int) i)
{
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
assert(arr1[i] == (int) i);
}
arr1[i] = 0;
}
cuda::memcpy_async(group, arr1, ann0, ARR_SZ * sizeof(int), bar3);
// cuda::memcpy_async(group, arr1, cann0, ARR_SZ * sizeof(int), bar3);
bar3.arrive_and_wait();
for (size_t i = 0; i < ARR_SZ; ++i) {
if (arr1[i] != (int) i)
{
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
assert(arr1[i] == (int) i);
}
arr1[i] = 0;
}))
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr0); free(arr1);), (assert_rt(cudaFree(arr0)); assert_rt(cudaFree(arr1));))
}
int main(int argc, char** argv)
{
test_memcpy_async();
return 0;
}

View File

@@ -0,0 +1,80 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "test_macros.h"
TEST_DIAG_SUPPRESS_MSVC(4505)
#include <cuda/annotated_ptr>
#include <cuda/std/cassert>
#include <nv/target>
#if defined(DEBUG)
# define DPRINTF(...) \
{ \
printf(__VA_ARGS__); \
}
#else
# define DPRINTF(...) \
do \
{ \
} while (false)
#endif
TEST_FUNC void assert_rt_wrap(cudaError_t code, const char* file, int line)
{
if (code != cudaSuccess)
{
#if !TEST_COMPILER(NVRTC)
NV_IF_ELSE_TARGET(NV_IS_HOST,
(printf("assert: %s %s %d\n", cudaGetErrorString(code), file, line);),
(printf("assert: error=%d %s %d\n", code, file, line);))
#endif // !TEST_COMPILER(NVRTC)
assert(code == cudaSuccess);
}
}
#define assert_rt(ret) \
{ \
assert_rt_wrap((ret), __FILE__, __LINE__); \
}
template <typename T, int N>
TEST_FUNC __noinline__ T* global_alloc()
{
T* arr = nullptr;
NV_IF_ELSE_TARGET(
NV_IS_DEVICE, (arr = (T*) malloc(N * sizeof(T));), (assert_rt(cudaMallocManaged((void**) &arr, N * sizeof(T)));))
for (int i = 0; i < N; ++i)
{
arr[i] = i;
}
return arr;
}
template <typename T, int N>
TEST_DEVICE_FUNC __noinline__ T* shared_alloc()
{
__shared__ T data[N];
for (int i = 0; i < N; ++i)
{
data[i] = i;
}
return data;
}
template <typename T>
TEST_FUNC __noinline__ void dealloc(T* arr)
{
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr);), assert_rt(cudaFree(arr));)
}

View File

@@ -0,0 +1,198 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/std/cassert>
#include <cuda/std/limits>
#include <cuda/std/span>
#include <cuda/std/type_traits>
#include "test_macros.h"
struct minimal_comparable_value
{
int value;
};
TEST_FUNC constexpr bool operator<(minimal_comparable_value lhs, minimal_comparable_value rhs)
{
return lhs.value < rhs.value;
}
TEST_FUNC constexpr bool operator==(minimal_comparable_value lhs, minimal_comparable_value rhs)
{
return lhs.value == rhs.value;
}
namespace cuda::std
{
template <>
class numeric_limits<minimal_comparable_value>
{
public:
static constexpr bool is_specialized = true;
TEST_FUNC static constexpr minimal_comparable_value lowest() noexcept
{
return {0};
}
TEST_FUNC static constexpr minimal_comparable_value max() noexcept
{
return {100};
}
};
} // namespace cuda::std
TEST_FUNC constexpr bool test()
{
// --- static_bounds ---
// Basic static bounds
{
constexpr auto b = cuda::args::static_bounds<1, 4096>{};
static_assert(b.lower() == 1);
static_assert(b.upper() == 4096);
}
// Exact static bounds
{
constexpr auto b = cuda::args::static_bounds<42, 42>{};
static_assert(b.lower() == 42);
static_assert(b.upper() == 42);
}
// Long type deduced from NTTPs
{
static_assert(cuda::std::is_same_v<decltype(cuda::args::static_bounds<0L, 1000L>::lower()), long>);
}
#if TEST_HAS_CLASS_NTTP
// Static bounds preserve their original NTTP types
{
constexpr auto b = cuda::args::bounds<1.0f, 8.0f>();
static_assert(b.lower() == 1.0f);
static_assert(b.upper() == 8);
static_assert(cuda::std::is_same_v<decltype(b.lower()), float>);
static_assert(cuda::std::is_same_v<decltype(b.upper()), float>);
}
#endif // TEST_HAS_CLASS_NTTP
// --- runtime_bounds ---
// Basic runtime bounds
{
auto b = cuda::args::runtime_bounds{10, 100};
assert(b.lower() == 10);
assert(b.upper() == 100);
static_assert(cuda::std::is_same_v<decltype(b.lower()), int>);
}
// Default runtime bounds span the element type's numeric_limits range
{
constexpr cuda::args::runtime_bounds<int> b{};
static_assert(b.lower() == cuda::std::numeric_limits<int>::lowest());
static_assert(b.upper() == (cuda::std::numeric_limits<int>::max)());
}
// --- argument_bounds factory functions ---
// Static via factory
{
constexpr auto b = cuda::args::bounds<1, 8>();
static_assert(b.lower() == 1);
static_assert(b.upper() == 8);
static_assert(cuda::args::__is_static_bounds_cv_v<decltype(b)>);
static_assert(!cuda::args::__is_runtime_bounds_cv_v<decltype(b)>);
static_assert(cuda::args::__is_bounds_v<decltype(b)>);
}
// Runtime via factory
{
auto b = cuda::args::bounds(10, 100);
assert(b.lower() == 10);
assert(b.upper() == 100);
static_assert(!cuda::args::__is_static_bounds_cv_v<decltype(b)>);
static_assert(cuda::args::__is_runtime_bounds_cv_v<decltype(b)>);
static_assert(cuda::args::__is_bounds_v<decltype(b)>);
}
// Runtime bounds only require operator< and operator==.
{
constexpr auto b = cuda::args::bounds(minimal_comparable_value{10}, minimal_comparable_value{20});
static_assert(b.lower() == minimal_comparable_value{10});
static_assert(b.upper() == minimal_comparable_value{20});
}
// Static and runtime bounds intersection
{
static_assert(cuda::args::__has_bounds_intersection<int, cuda::args::static_bounds<1, 100>>(
cuda::args::runtime_bounds<int>{50, 200}));
static_assert(!cuda::args::__has_bounds_intersection<int, cuda::args::static_bounds<100, 200>>(
cuda::args::runtime_bounds<int>{0, 50}));
}
// Runtime bounds validation with no static bounds only requires operator< and operator==.
{
minimal_comparable_value values[] = {{10}, {20}};
[[maybe_unused]] auto arg = cuda::args::deferred_sequence{
cuda::std::span<minimal_comparable_value>{values, 2},
cuda::args::bounds(minimal_comparable_value{5}, minimal_comparable_value{50})};
}
// Unsigned no-bounds arguments must not instantiate a pointless `value < 0` comparison.
{
unsigned int value = 0;
[[maybe_unused]] auto arg = cuda::args::deferred{&value};
}
#if TEST_HAS_CLASS_NTTP
// Static/runtime bounds intersection only requires operator< and operator==.
{
using static_bounds_t = cuda::args::static_bounds<minimal_comparable_value{10}, minimal_comparable_value{50}>;
constexpr auto runtime_bounds = cuda::args::bounds(minimal_comparable_value{20}, minimal_comparable_value{40});
static_assert(cuda::args::__has_bounds_intersection<minimal_comparable_value, static_bounds_t>(runtime_bounds));
static_assert(!cuda::args::__has_bounds_intersection<minimal_comparable_value, static_bounds_t>(
cuda::args::bounds(minimal_comparable_value{60}, minimal_comparable_value{70})));
cuda::args::__validate_static_element_bounds<minimal_comparable_value, static_bounds_t>(
minimal_comparable_value{30});
cuda::args::__validate_runtime_element_bounds(minimal_comparable_value{30}, runtime_bounds);
minimal_comparable_value values[] = {{20}, {30}};
[[maybe_unused]] auto arg = cuda::args::__immediate_sequence{
cuda::std::span<minimal_comparable_value>{values, 2}, static_bounds_t{}, runtime_bounds};
}
#endif // TEST_HAS_CLASS_NTTP
// Non-bounds type
{
static_assert(!cuda::args::__is_bounds_v<int>);
}
// Bounds types accepted by argument wrapper template parameters
{
static_assert(cuda::args::__valid_static_bounds_v<int, cuda::args::no_bounds>);
static_assert(cuda::args::__valid_static_bounds_v<int, cuda::args::static_bounds<1, 8>>);
static_assert(!cuda::args::__valid_static_bounds_v<int, cuda::args::runtime_bounds<int>>);
static_assert(!cuda::args::__valid_static_bounds_v<int, int>);
}
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,185 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/iterator>
#include <cuda/std/array>
#include <cuda/std/complex>
#include <cuda/std/expected>
#include <cuda/std/limits>
#include <cuda/std/mdspan>
#include <cuda/std/optional>
#include <cuda/std/span>
#include <cuda/std/tuple>
#include <cuda/std/type_traits>
#include <cuda/std/utility>
#include "test_iterators.h"
#include "test_macros.h"
enum class color
{
red,
green,
blue
};
template <class _Tp>
struct element_type_like
{
using element_type = _Tp;
};
template <class _Tp>
struct range_like
{
using iterator = _Tp*;
};
template <class _Tp>
struct value_type_like
{
using value_type = _Tp;
};
struct non_sequence_value
{};
TEST_FUNC void test()
{
// --- __is_sequence_v ---
// builtin and class type are not sequences
static_assert(!cuda::args::__is_sequence_v<int>);
static_assert(!cuda::args::__is_sequence_v<color>);
static_assert(!cuda::args::__is_sequence_v<non_sequence_value>);
static_assert(!cuda::args::__is_sequence_v<range_like<int>>);
static_assert(!cuda::args::__is_sequence_v<element_type_like<int>>);
static_assert(!cuda::args::__is_sequence_v<value_type_like<int>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::complex<float>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::pair<float, int>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::tuple<float, int>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::optional<int>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::expected<int, int>>);
// iterators and pointers can be sequences if they are at least random access
static_assert(cuda::args::__is_sequence_v<int*>);
static_assert(cuda::args::__is_sequence_v<const int*>);
static_assert(cuda::args::__is_sequence_v<cuda::counting_iterator<int>>);
static_assert(!cuda::args::__is_sequence_v<bidirectional_iterator<int*>>);
// ranges and arrays are sequences
static_assert(cuda::args::__is_sequence_v<int[]>);
static_assert(cuda::args::__is_sequence_v<const int[]>);
static_assert(cuda::args::__is_sequence_v<int[42]>);
static_assert(cuda::args::__is_sequence_v<const int[42]>);
static_assert(cuda::args::__is_sequence_v<cuda::std::span<int, 1>>);
static_assert(cuda::args::__is_sequence_v<const cuda::std::span<int, 1>&>);
static_assert(cuda::args::__is_sequence_v<cuda::std::span<int>>);
static_assert(cuda::args::__is_sequence_v<cuda::std::array<int, 3>>);
// --- __element_type_of_t ---
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<const cuda::std::span<int, 1>&>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<int*>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::counting_iterator<int>>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::std::array<int, 3>>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<range_like<int>>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<element_type_like<int>>, int>);
static_assert(
cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::std::mdspan<const int, cuda::std::extents<int, 1>>>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<value_type_like<int>>, int>);
// --- argument_traits: is_deferred ---
static_assert(!cuda::args::__traits<int>::is_deferred);
static_assert(!cuda::args::__traits<cuda::args::immediate<int>>::is_deferred);
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_deferred);
static_assert(!cuda::args::__traits<cuda::args::constant<42>>::is_deferred);
#if TEST_HAS_CLASS_NTTP
static_assert(!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_deferred);
#endif // TEST_HAS_CLASS_NTTP
static_assert(cuda::args::__traits<cuda::args::deferred<cuda::std::span<int, 1>>>::is_deferred);
static_assert(cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>::is_deferred);
// --- argument_traits: is_single_value ---
static_assert(cuda::args::__traits<int>::is_single_value);
static_assert(cuda::args::__traits<int*>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::immediate<int>>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::immediate<int*>>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::immediate<cuda::counting_iterator<int>>>::is_single_value);
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::constant<42>>::is_single_value);
#if TEST_HAS_CLASS_NTTP
static_assert(
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
#endif // TEST_HAS_CLASS_NTTP
static_assert(cuda::args::__traits<cuda::args::deferred<int*>>::is_single_value);
static_assert(!cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>::is_single_value);
// --- argument_traits: value_type ---
static_assert(cuda::std::is_same_v<cuda::args::__traits<int>::value_type, int>);
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::immediate<int>>::value_type, int>);
static_assert(
cuda::std::is_same_v<cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::value_type,
cuda::std::span<int>>);
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::constant<42>>::value_type, int>);
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::constant<10, float>>::value_type, float>);
#if TEST_HAS_CLASS_NTTP
static_assert(cuda::std::is_same_v<
cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::value_type,
cuda::std::array<int, 3>>);
#endif // TEST_HAS_CLASS_NTTP
// --- argument_traits: lowest / highest ---
static_assert(cuda::args::__traits<int>::lowest == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__traits<int>::highest == (cuda::std::numeric_limits<int>::max)());
static_assert(cuda::args::__traits<const int>::lowest == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__traits<int&>::highest == (cuda::std::numeric_limits<int>::max)());
static_assert(cuda::args::__traits<float>::lowest == cuda::std::numeric_limits<float>::lowest());
static_assert(cuda::args::__traits<float>::highest == (cuda::std::numeric_limits<float>::max)());
static_assert(cuda::args::__traits<const cuda::args::immediate<int, cuda::args::static_bounds<1, 8>>>::lowest == 1);
static_assert(cuda::args::__traits<cuda::args::immediate<int, cuda::args::static_bounds<1, 8>>&>::highest == 8);
static_assert(
cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>, cuda::args::static_bounds<1, 8>>>::highest
== 8);
static_assert(cuda::args::__traits<cuda::args::constant<10, float>>::lowest == 10.0f);
static_assert(cuda::args::__traits<cuda::args::constant<10, float>>::highest == 10.0f);
#if TEST_HAS_CLASS_NTTP
static_assert(cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{3, 1, 2}>>::lowest == 1);
static_assert(cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{3, 1, 2}>>::highest == 3);
#endif // TEST_HAS_CLASS_NTTP
// --- Free function bounds on plain values ---
static_assert(cuda::args::__lowest_(42) == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__highest_(42) == (cuda::std::numeric_limits<int>::max)());
static_assert(cuda::args::__lowest_(1.0f) == cuda::std::numeric_limits<float>::lowest());
static_assert(cuda::args::__highest_(1.0f) == (cuda::std::numeric_limits<float>::max)());
// --- Scalar and sequence wrappers expose distinct single-value traits ---
static_assert(cuda::args::__traits<cuda::args::constant<42>>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::immediate<int>>::is_single_value);
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
#if TEST_HAS_CLASS_NTTP
static_assert(
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
#endif // TEST_HAS_CLASS_NTTP
}
int main(int, char**)
{
test();
return 0;
}

View File

@@ -0,0 +1,174 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/iterator>
#include <cuda/std/cassert>
#include <cuda/std/limits>
#include <cuda/std/span>
#include <cuda/std/type_traits>
#include "test_macros.h"
TEST_FUNC constexpr bool test()
{
// Deferred single value via span<T, 1>
{
int val = 42;
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}};
assert(cuda::args::__unwrap(def)[0] == 42);
assert(cuda::args::__access::__arg(def)[0] == 42);
static_assert(cuda::args::__traits<decltype(def)>::lowest == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__traits<decltype(def)>::highest == (cuda::std::numeric_limits<int>::max)());
}
// Deferred single value with static bounds
{
int val = 42;
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds<1, 1000>()};
assert(cuda::args::__unwrap(def)[0] == 42);
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(def)>::highest == 1000);
}
// Deferred single value via pointer
{
int val = 42;
using def_t = cuda::args::deferred<int*, cuda::args::static_bounds<0, 100>>;
static_assert(cuda::args::__traits<def_t>::lowest == 0);
static_assert(cuda::args::__traits<def_t>::highest == 100);
// Also verify construction works
auto def = cuda::args::deferred{&val, cuda::args::bounds<0, 100>()};
assert(cuda::args::__unwrap(def) == &val);
}
// Deferred single value via fancy iterator
{
auto it = cuda::counting_iterator<int>{42};
auto def = cuda::args::deferred{it, cuda::args::bounds<0, 100>()};
assert(cuda::args::__unwrap(def)[0] == 42);
static_assert(cuda::args::__traits<decltype(def)>::lowest == 0);
static_assert(cuda::args::__traits<decltype(def)>::highest == 100);
static_assert(cuda::args::__traits<decltype(def)>::is_single_value);
}
// Deferred single value with both bounds, runtime bounds first
{
int val = 42;
auto def =
cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds(5, 100), cuda::args::bounds<1, 256>()};
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(def)>::highest == 256);
assert(cuda::args::__access::__runtime_bounds(def).lower() == 5);
assert(cuda::args::__access::__runtime_bounds(def).upper() == 100);
assert(cuda::args::__lowest_(def) == 5);
assert(cuda::args::__highest_(def) == 100);
cuda::args::__access::__runtime_bounds(def) = cuda::args::bounds(5, 90);
assert(cuda::args::__highest_(def) == 90);
}
// Deferred sequence via fancy iterator
{
auto it = cuda::counting_iterator<int>{10};
auto def = cuda::args::deferred_sequence{it, cuda::args::bounds<0, 100>()};
assert(cuda::args::__unwrap(def)[0] == 10);
assert(cuda::args::__unwrap(def)[2] == 12);
static_assert(cuda::args::__traits<decltype(def)>::lowest == 0);
static_assert(cuda::args::__traits<decltype(def)>::highest == 100);
static_assert(!cuda::args::__traits<decltype(def)>::is_single_value);
}
// Deferred sequence with both bounds
{
int arr[4] = {10, 20, 30, 40};
auto def = cuda::args::deferred_sequence{
cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 4096>(), cuda::args::bounds(5, 100)};
assert(cuda::args::__access::__arg(def).size() == 4);
assert(cuda::args::__access::__runtime_bounds(def).lower() == 5);
assert(cuda::args::__access::__runtime_bounds(def).upper() == 100);
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
assert(cuda::args::__lowest_(def) == 5);
assert(cuda::args::__highest_(def) == 100);
}
// Deferred sequence with both bounds, runtime bounds first
{
int arr[4] = {10, 20, 30, 40};
auto def = cuda::args::deferred_sequence{
cuda::std::span<int>{arr, 4}, cuda::args::bounds(5, 100), cuda::args::bounds<1, 4096>()};
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(def)>::highest == 4096);
assert(cuda::args::__lowest_(def) == 5);
assert(cuda::args::__highest_(def) == 100);
}
// Traits: deferred is single value
{
using traits = cuda::args::__traits<cuda::args::deferred<cuda::std::span<int, 1>>>;
static_assert(traits::is_deferred);
static_assert(traits::is_single_value);
}
// Traits: deferred with pointer is also single value
{
using traits = cuda::args::__traits<cuda::args::deferred<int*>>;
static_assert(traits::is_deferred);
static_assert(traits::is_single_value);
}
// Traits: deferred_sequence is not single value
{
using traits = cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>;
static_assert(traits::is_deferred);
static_assert(!traits::is_single_value);
}
// Unwrap: deferred
{
int val = 99;
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}};
auto& v = cuda::args::__unwrap(def);
assert(v[0] == 99);
}
// Unwrap: deferred_sequence
{
int arr[3] = {10, 20, 30};
auto def = cuda::args::deferred_sequence{cuda::std::span<int>{arr, 3}};
const auto& v = cuda::args::__unwrap(def);
assert(v.size() == 3);
assert(v[1] == 20);
}
// Unwrap: rvalue deferred returns by value
{
int val = 99;
auto v = cuda::args::__unwrap(cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}});
assert(v[0] == 99);
}
// Unwrap: rvalue deferred_sequence returns by value
{
int arr[3] = {10, 20, 30};
auto v = cuda::args::__unwrap(cuda::args::deferred_sequence{cuda::std::span<int>{arr, 3}});
assert(v.size() == 3);
assert(v[2] == 30);
}
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,18 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
[[maybe_unused]] cuda::args::deferred_sequence<int> invalid_arg{0};
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,20 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
using traits = cuda::args::__traits<cuda::args::deferred_sequence<int>>;
[[maybe_unused]] constexpr bool invalid_traits = traits::is_deferred;
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,181 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/std/cassert>
#include <cuda/std/limits>
#include <cuda/std/span>
#include <cuda/std/type_traits>
#include "test_macros.h"
struct non_sequence_value
{
int payload;
};
TEST_FUNC constexpr bool test()
{
// Uniform scalar via CTAD
{
auto da = cuda::args::immediate{5};
assert(cuda::args::__unwrap(da) == 5);
assert(cuda::args::__access::__arg(da) == 5);
static_assert(cuda::args::__traits<decltype(da)>::lowest == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__traits<decltype(da)>::highest == (cuda::std::numeric_limits<int>::max)());
assert(cuda::args::__lowest_(da) == 5);
assert(cuda::args::__highest_(da) == 5);
cuda::args::__access::__arg(da) = 6;
assert(cuda::args::__unwrap(da) == 6);
}
// Uniform scalar with static bounds
{
auto da = cuda::args::immediate{5, cuda::args::bounds<1, 8>()};
assert(cuda::args::__unwrap(da) == 5);
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(da)>::highest == 8);
assert(cuda::args::__lowest_(da) == 5);
assert(cuda::args::__highest_(da) == 5);
}
// Non-sequence values are accepted without scalar-only restrictions
{
auto da = cuda::args::immediate{non_sequence_value{7}};
assert(cuda::args::__unwrap(da).payload == 7);
}
// Pointer-like types can still represent a single value when explicitly wrapped that way
{
int value = 11;
auto da = cuda::args::immediate{&value};
static_assert(cuda::args::__traits<decltype(da)>::is_single_value);
assert(*cuda::args::__unwrap(da) == 11);
}
// Per-segment span with runtime bounds
{
int arr[4] = {10, 20, 30, 40};
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}, cuda::args::bounds(1L, 100L)};
assert(cuda::args::__unwrap(da).size() == 4);
assert(cuda::args::__access::__arg(da).size() == 4);
assert(cuda::args::__access::__runtime_bounds(da).lower() == 1);
assert(cuda::args::__access::__runtime_bounds(da).upper() == 100);
assert(cuda::args::__lowest_(da) == 1);
assert(cuda::args::__highest_(da) == 100);
cuda::args::__access::__runtime_bounds(da) = cuda::args::bounds(1, 90);
assert(cuda::args::__highest_(da) == 90);
}
// Per-segment span with both bounds
{
int arr[4] = {10, 20, 30, 40};
auto da = cuda::args::__immediate_sequence{
cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 256>(), cuda::args::bounds(10, 200)};
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(da)>::highest == 256);
assert(cuda::args::__lowest_(da) == 10);
assert(cuda::args::__highest_(da) == 200);
}
// Per-segment span with both bounds, runtime bounds first
{
int arr[4] = {10, 20, 30, 40};
auto da = cuda::args::__immediate_sequence{
cuda::std::span<int>{arr, 4}, cuda::args::bounds(10, 200), cuda::args::bounds<1, 256>()};
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(da)>::highest == 256);
assert(cuda::args::__lowest_(da) == 10);
assert(cuda::args::__highest_(da) == 200);
}
// Per-segment via span
{
int arr[4] = {1, 2, 3, 4};
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}};
assert(cuda::args::__unwrap(da).size() == 4);
assert(cuda::args::__unwrap(da)[0] == 1);
assert(cuda::args::__unwrap(da)[3] == 4);
}
// Per-segment with static bounds
{
int arr[4] = {10, 20, 30, 40};
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 100>()};
assert(cuda::args::__unwrap(da).size() == 4);
assert(cuda::args::__unwrap(da)[2] == 30);
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(da)>::highest == 100);
}
// Traits
{
using traits = cuda::args::__traits<cuda::args::immediate<int>>;
static_assert(!traits::is_deferred);
static_assert(traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, int>);
}
// Sequence traits
{
using traits = cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>;
static_assert(!traits::is_deferred);
static_assert(!traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, cuda::std::span<int>>);
}
// __is_sequence_v on unwrapped types
{
static_assert(!cuda::args::__is_sequence_v<cuda::args::__traits<cuda::args::immediate<int>>::value_type>);
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
}
// Unwrap: scalar
{
auto da = cuda::args::immediate{7};
auto& v = cuda::args::__unwrap(da);
assert(v == 7);
v = 8;
assert(cuda::args::__unwrap(da) == 8);
}
// Unwrap: span
{
int arr[3] = {10, 20, 30};
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 3}};
const auto& v = cuda::args::__unwrap(da);
assert(v.size() == 3);
assert(v[1] == 20);
}
// Unwrap: rvalue scalar returns by value
{
const auto& v = cuda::args::__unwrap(cuda::args::immediate{7});
assert(v == 7);
}
// Unwrap: rvalue span returns by value
{
int arr[3] = {10, 20, 30};
auto v = cuda::args::__unwrap(cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 3}});
assert(v.size() == 3);
assert(v[2] == 30);
}
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,23 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
// A type without a cuda::std::numeric_limits specialization has no meaningful implicit bounds. Default-constructing
// runtime_bounds for such a type must be rejected at compile time instead of silently producing a degenerate range.
struct unspecialized_type
{};
[[maybe_unused]] cuda::args::runtime_bounds<unspecialized_type> invalid_bounds{};
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,197 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/std/array>
#include <cuda/std/limits>
#include <cuda/std/type_traits>
#include "test_macros.h"
struct non_sequence_value
{
int payload;
};
enum class dependent_direction
{
min,
max
};
template <dependent_direction Value>
struct dependent_direction_tag
{
static constexpr auto value = Value;
};
template <class Tag>
TEST_FUNC void test_dependent_constant_type()
{
constexpr auto direction = Tag::value;
using constant_t = cuda::args::constant<direction>;
// Regression: NVCC bug generated a host stub using a cv/ref-qualified constant type while device registration used
// the unqualified type, causing cudaErrorInvalidDeviceFunction when launching the kernel.
static_assert(cuda::std::is_same_v<typename constant_t::value_type, dependent_direction>);
static_assert(cuda::std::is_same_v<constant_t, cuda::args::constant<Tag::value, dependent_direction>>);
}
TEST_FUNC void test()
{
// Basic value
{
constexpr auto sa = cuda::args::constant<42>{};
static_assert(cuda::args::__unwrap(sa) == 42);
static_assert(cuda::std::is_same_v<decltype(sa)::value_type, int>);
}
// Different types
{
constexpr auto sa_long = cuda::args::constant<100L>{};
static_assert(cuda::args::__unwrap(sa_long) == 100L);
static_assert(cuda::std::is_same_v<decltype(sa_long)::value_type, long>);
constexpr auto sa_float = cuda::args::constant<10, float>{};
static_assert(cuda::args::__unwrap(sa_float) == 10.0f);
static_assert(cuda::std::is_same_v<decltype(sa_float)::value_type, float>);
static_assert(cuda::std::is_same_v<decltype(cuda::args::__unwrap(sa_float)), float>);
}
// Negative value
{
constexpr auto sa_neg = cuda::args::constant<-1>{};
static_assert(cuda::args::__unwrap(sa_neg) == -1);
}
// Dependent value
{
test_dependent_constant_type<dependent_direction_tag<dependent_direction::max>>();
}
#if TEST_HAS_CLASS_NTTP
// Non-sequence values are accepted without scalar-only restrictions
{
constexpr auto sa = cuda::args::constant<non_sequence_value{7}>{};
static_assert(cuda::args::__unwrap(sa).payload == 7);
}
#endif // TEST_HAS_CLASS_NTTP
#if TEST_HAS_CLASS_NTTP
// Array sequence
{
constexpr auto sa_arr = cuda::args::__constant_sequence<cuda::std::array<int, 3>{128, 256, 512}>{};
static_assert(cuda::args::__unwrap(sa_arr)[0] == 128);
static_assert(cuda::args::__unwrap(sa_arr)[1] == 256);
static_assert(cuda::args::__unwrap(sa_arr)[2] == 512);
static_assert(cuda::std::is_same_v<decltype(sa_arr)::value_type, cuda::std::array<int, 3>>);
}
#endif // TEST_HAS_CLASS_NTTP
// Bounds: scalar
{
constexpr auto sa = cuda::args::constant<42>{};
static_assert(cuda::args::__lowest_(sa) == 42);
static_assert(cuda::args::__highest_(sa) == 42);
}
#if TEST_HAS_CLASS_NTTP
// Bounds: array sequence computes lowest/highest of elements
{
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 3>{128, 256, 512}>{};
static_assert(cuda::args::__lowest_(sa) == 128);
static_assert(cuda::args::__highest_(sa) == 512);
}
#endif // TEST_HAS_CLASS_NTTP
#if TEST_HAS_CLASS_NTTP
// Bounds: empty array sequence has unconstrained element bounds
{
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 0>{}>{};
static_assert(cuda::args::__lowest_(sa) == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__highest_(sa) == (cuda::std::numeric_limits<int>::max)());
}
#endif // TEST_HAS_CLASS_NTTP
// Traits
{
using traits = cuda::args::__traits<cuda::args::constant<42>>;
static_assert(!traits::is_deferred);
static_assert(traits::is_constant);
static_assert(traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, int>);
static_assert(traits::lowest == 42);
static_assert(traits::highest == 42);
}
// Traits: explicit constant value type
{
using traits = cuda::args::__traits<cuda::args::constant<10, float>>;
static_assert(!traits::is_deferred);
static_assert(traits::is_constant);
static_assert(traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, float>);
static_assert(cuda::std::is_same_v<traits::element_type, float>);
static_assert(traits::lowest == 10.0f);
static_assert(traits::highest == 10.0f);
}
#if TEST_HAS_CLASS_NTTP
// Sequence traits
{
using traits = cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>;
static_assert(traits::is_constant);
static_assert(!traits::is_deferred);
static_assert(!traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, cuda::std::array<int, 3>>);
static_assert(cuda::std::is_same_v<traits::element_type, int>);
}
#endif // TEST_HAS_CLASS_NTTP
// Single value: scalar is single, sequence is not
{
static_assert(!cuda::args::__is_sequence_v<cuda::args::__traits<cuda::args::constant<42>>::value_type>);
#if TEST_HAS_CLASS_NTTP
static_assert(
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
#endif // TEST_HAS_CLASS_NTTP
}
// Unwrap: scalar
{
constexpr auto sa = cuda::args::constant<42>{};
constexpr auto val = cuda::args::__unwrap(sa);
static_assert(val == 42);
}
// Unwrap: scalar with explicit value type
{
constexpr auto sa = cuda::args::constant<10, float>{};
constexpr auto val = cuda::args::__unwrap(sa);
static_assert(val == 10.0f);
static_assert(cuda::std::is_same_v<decltype(val), const float>);
}
#if TEST_HAS_CLASS_NTTP
// Unwrap: sequence
{
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 3>{10, 20, 30}>{};
constexpr auto val = cuda::args::__unwrap(sa);
static_assert(val[0] == 10);
static_assert(val[2] == 30);
}
#endif // TEST_HAS_CLASS_NTTP
}
int main(int, char**)
{
test();
return 0;
}

View File

@@ -0,0 +1,20 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
using arg_t = cuda::args::immediate<int, cuda::args::runtime_bounds<int>>;
[[maybe_unused]] arg_t invalid_arg{0};
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,20 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
using arg_t = cuda::args::immediate<unsigned char, cuda::args::static_bounds<0, 1000>>;
[[maybe_unused]] constexpr auto invalid_highest = cuda::args::__traits<arg_t>::highest;
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,18 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
[[maybe_unused]] constexpr auto invalid_bounds = cuda::args::static_bounds<0, 1L>{};
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,24 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
// Reading the implicit bounds of __traits for an element type without a cuda::std::numeric_limits specialization must
// fail to compile rather than silently yielding a value-initialized (and therefore meaningless) bound. This exercises
// the __traits_impl primary-template path, which is the bound surface read by generic consumers.
struct unspecialized_type
{};
[[maybe_unused]] constexpr auto invalid_lowest = cuda::args::__traits<unspecialized_type>::lowest;
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,214 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// Integration test: demonstrates how an algorithm consumes argument wrappers
// to make compile-time and runtime resource decisions.
// All argument types (plain values, constants, immediate values, deferred values) work uniformly
// through the free functions.
#include <cuda/argument>
#include <cuda/std/algorithm>
#include <cuda/std/array>
#include <cuda/std/cassert>
#include <cuda/std/span>
#include "test_macros.h"
constexpr int shared_memory_capacity = 256;
constexpr int default_max_segment_size = 1024;
enum class algorithm_variant
{
shared_memory,
global_memory
};
// Static scaling: choose algorithm variant at compile time.
template <class _SegSizeArg>
TEST_FUNC constexpr algorithm_variant select_variant(_SegSizeArg)
{
if constexpr (cuda::args::__traits<_SegSizeArg>::highest <= shared_memory_capacity)
{
return algorithm_variant::shared_memory;
}
else
{
return algorithm_variant::global_memory;
}
}
// Dynamic scaling: compute buffer size at runtime, clamped to default.
template <class _SegSizeArg>
TEST_FUNC constexpr int compute_buffer_size(_SegSizeArg __seg_size, int __num_segments)
{
auto __highest = cuda::std::min(default_max_segment_size, static_cast<int>(cuda::args::__highest_(__seg_size)));
return __highest * __num_segments;
}
// Process: use the actual unwrapped value.
template <class _SegSizeArg>
TEST_FUNC constexpr int process_segments(_SegSizeArg __seg_size)
{
const auto& __val = cuda::args::__unwrap(__seg_size);
if constexpr (cuda::args::__traits<_SegSizeArg>::is_single_value)
{
return static_cast<int>(__val);
}
else
{
int __total = 0;
for (size_t __i = 0; __i < __val.size(); ++__i)
{
__total += static_cast<int>(__val[__i]);
}
return __total;
}
}
TEST_FUNC constexpr bool test()
{
// Plain scalar: no bounds, global memory, buffer clamped to default
{
static_assert(select_variant(100) == algorithm_variant::global_memory);
assert(compute_buffer_size(100, 4) == default_max_segment_size * 4);
assert(process_segments(100) == 100);
}
#if 0 // FIXME(miscco): This should not work
// Plain span: per-segment, no bounds, global memory
{
int sizes[3] = {64, 128, 96};
auto seg = cuda::std::span<int>{sizes, 3};
assert(select_variant(seg) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg, 3) == default_max_segment_size * 3);
assert(process_segments(seg) == 64 + 128 + 96);
}
#endif
// constant: scalar, fits in shared memory, buffer = value
{
constexpr auto seg_size = cuda::args::constant<128>{};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_size, 4) == 128 * 4);
assert(process_segments(seg_size) == 128);
}
#if TEST_HAS_CLASS_NTTP
// __constant_sequence: array sequence, highest fits in shared memory
{
constexpr auto seg_sizes = cuda::args::__constant_sequence<cuda::std::array{64, 128, 256}>{};
static_assert(select_variant(seg_sizes) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_sizes, 3) == 256 * 3);
assert(process_segments(seg_sizes) == 64 + 128 + 256);
}
// __constant_sequence: array sequence, highest exceeds shared memory, buffer clamped
{
constexpr auto seg_sizes = cuda::args::__constant_sequence<cuda::std::array{64, 128, 512}>{};
static_assert(select_variant(seg_sizes) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg_sizes, 3) == 512 * 3);
assert(process_segments(seg_sizes) == 64 + 128 + 512);
}
#endif // TEST_HAS_CLASS_NTTP
// immediate: tight static bounds, shared memory, buffer = value
{
constexpr auto seg_size = cuda::args::immediate{100, cuda::args::bounds<1, 256>()};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
assert(process_segments(seg_size) == 100);
}
// immediate: wide static bounds, global memory, buffer = value
{
constexpr auto seg_size = cuda::args::immediate{100, cuda::args::bounds<1, 4096>()};
static_assert(select_variant(seg_size) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
assert(process_segments(seg_size) == 100);
}
// immediate: no bounds, global memory, buffer = value
{
constexpr auto seg_size = cuda::args::immediate{100};
static_assert(select_variant(seg_size) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
assert(process_segments(seg_size) == 100);
}
// __immediate_sequence: per-segment span with runtime bounds only
{
int sizes[3] = {64, 128, 96};
auto seg_sizes = cuda::args::__immediate_sequence{cuda::std::span<int>{sizes, 3}, cuda::args::bounds(1, 200)};
assert(select_variant(seg_sizes) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg_sizes, 3) == 200 * 3);
assert(process_segments(seg_sizes) == 64 + 128 + 96);
}
// __immediate_sequence: per-segment span with both bounds
{
int sizes[3] = {64, 128, 96};
auto seg_sizes = cuda::args::__immediate_sequence{
cuda::std::span<int>{sizes, 3}, cuda::args::bounds<1, 256>(), cuda::args::bounds(1, 200)};
static_assert(cuda::args::__traits<decltype(seg_sizes)>::highest <= shared_memory_capacity);
assert(select_variant(seg_sizes) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_sizes, 3) == 200 * 3);
assert(process_segments(seg_sizes) == 64 + 128 + 96);
}
// deferred: uniform, bounds for decisions only
{
int val = 100;
auto seg_size =
cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds<1, 256>(), cuda::args::bounds(1, 200)};
static_assert(cuda::args::__traits<decltype(seg_size)>::highest <= shared_memory_capacity);
assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_size, 4) == 200 * 4);
}
// --- Floating point cases ---
// Plain float: no bounds
{
static_assert(select_variant(1.0f) == algorithm_variant::global_memory);
assert(process_segments(1.0f) == 1);
}
// constant float using an integer NTTP and explicit value type
{
constexpr auto seg_size = cuda::args::constant<128, float>{};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(process_segments(seg_size) == 128);
}
#if TEST_HAS_CLASS_NTTP
// constant float (float NTTPs require C++20)
{
constexpr auto seg_size = cuda::args::constant<128.0f>{};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(process_segments(seg_size) == 128);
}
// immediate float with static bounds
{
constexpr auto seg_size = cuda::args::immediate{100.0f, cuda::args::bounds<1.0f, 256.0f>()};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(process_segments(seg_size) == 100);
}
#endif // TEST_HAS_CLASS_NTTP
return true;
}
int main(int, char**)
{
test();
return 0;
}

View File

@@ -0,0 +1,115 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// UNSUPPORTED: nvcc-11, nvcc-12
// <cuda/atomic>
// TODO: Add support for new half
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "atomic_helpers.h"
#include "cuda_space_selector.h"
#include "test_macros.h"
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFn
{
TEST_FUNC void operator()() const
{
// Fetch min
{
using A = cuda::atomic<T, ThreadScope>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(T(-5)) == T(-1));
printf("%i == %i\n", (int) t.load(), (int) T(-5));
NV_IF_TARGET(NV_IS_HOST, (fflush(stdout);))
assert(t.load() == T(-5));
}
{
using A = cuda::atomic<T, ThreadScope>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(T(-5)) == T(-1));
assert(t.load() == T(-5));
}
// Test not lesser
{
using A = cuda::atomic<T, ThreadScope>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(4) == T(-1));
assert(t.load() == T(-1));
}
{
using A = cuda::atomic<T, ThreadScope>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(4) == T(-1));
assert(t.load() == T(-1));
}
// Fetch max
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(1);
assert(t.fetch_max(2) == T(1));
assert(t.load() == T(2));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(1);
assert(t.fetch_max(2) == T(1));
assert(t.load() == T(2));
}
// Test not greater
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(3);
assert(t.fetch_max(2) == T(3));
assert(t.load() == T(3));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(3);
assert(t.fetch_max(2) == T(3));
assert(t.load() == T(3));
}
}
};
int main(int, char**)
{
NV_DISPATCH_TARGET(NV_IS_HOST,
(TestFn<__half, local_memory_selector, cuda::thread_scope::thread_scope_thread>()();),
NV_PROVIDES_SM_70,
(TestFn<__half, local_memory_selector, cuda::thread_scope::thread_scope_thread>()();))
NV_IF_TARGET(NV_IS_DEVICE,
(TestFn<__half, shared_memory_selector, cuda::thread_scope::thread_scope_thread>()();
TestFn<__half, global_memory_selector, cuda::thread_scope::thread_scope_thread>()();))
return 0;
}

View File

@@ -0,0 +1,140 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "atomic_helpers.h"
#include "cuda_space_selector.h"
#include "test_macros.h"
template <class T,
template <typename, typename> class Selector,
cuda::thread_scope ThreadScope,
bool Signed = cuda::std::is_signed<T>::value>
struct TestFn
{
TEST_FUNC void operator()() const
{
// Test greater
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(1);
assert(t.fetch_max(2) == T(1));
assert(t.load() == T(2));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(1);
assert(t.fetch_max(2) == T(1));
assert(t.load() == T(2));
}
// Test not greater
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(3);
assert(t.fetch_max(2) == T(3));
assert(t.load() == T(3));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(3);
assert(t.fetch_max(2) == T(3));
assert(t.load() == T(3));
}
}
};
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFn<T, Selector, ThreadScope, true>
{
TEST_FUNC void operator()() const
{
// Call unsigned tests
TestFn<T, Selector, ThreadScope, false>()();
// Test greater, but with signed math
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-5);
assert(t.fetch_max(-1) == T(-5));
assert(t.load() == T(-1));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-5);
assert(t.fetch_max(-1) == T(-5));
assert(t.load() == T(-1));
}
// Test not greater
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_max(-5) == T(-1));
assert(t.load() == T(-1));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_max(-5) == T(-1));
assert(t.load() == T(-1));
}
}
};
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFnDispatch
{
TEST_FUNC void operator()() const
{
TestFn<T, Selector, ThreadScope>()();
}
};
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();),
NV_PROVIDES_SM_70,
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();))
NV_IF_TARGET(NV_IS_DEVICE,
(TestEachIntegralType<TestFnDispatch, shared_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, shared_memory_selector>()();
TestEachIntegralType<TestFnDispatch, global_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, global_memory_selector>()();))
return 0;
}

View File

@@ -0,0 +1,140 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "atomic_helpers.h"
#include "cuda_space_selector.h"
#include "test_macros.h"
template <class T,
template <typename, typename> class Selector,
cuda::thread_scope ThreadScope,
bool Signed = cuda::std::is_signed<T>::value>
struct TestFn
{
TEST_FUNC void operator()() const
{
// Test lesser
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(5);
assert(t.fetch_min(4) == T(5));
assert(t.load() == T(4));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(5);
assert(t.fetch_min(4) == T(5));
assert(t.load() == T(4));
}
// Test not lesser
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(3);
assert(t.fetch_min(4) == T(3));
assert(t.load() == T(3));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(3);
assert(t.fetch_min(4) == T(3));
assert(t.load() == T(3));
}
}
};
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFn<T, Selector, ThreadScope, true>
{
TEST_FUNC void operator()() const
{
// Call unsigned tests
TestFn<T, Selector, ThreadScope, false>()();
// Test lesser, but with signed math
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(-5) == T(-1));
assert(t.load() == T(-5));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(-5) == T(-1));
assert(t.load() == T(-5));
}
// Test not lesser
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(4) == T(-1));
assert(t.load() == T(-1));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(4) == T(-1));
assert(t.load() == T(-1));
}
}
};
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFnDispatch
{
TEST_FUNC void operator()() const
{
TestFn<T, Selector, ThreadScope>()();
}
};
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();),
NV_PROVIDES_SM_70,
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();))
NV_IF_TARGET(NV_IS_DEVICE,
(TestEachIntegralType<TestFnDispatch, shared_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, shared_memory_selector>()();
TestEachIntegralType<TestFnDispatch, global_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, global_memory_selector>()();))
return 0;
}

View File

@@ -0,0 +1,102 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
#ifndef ATOMIC_HELPERS_H
#define ATOMIC_HELPERS_H
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
struct UserAtomicType
{
int i;
TEST_FUNC explicit UserAtomicType(int d = 0) noexcept
: i(d)
{}
TEST_FUNC friend bool operator==(const UserAtomicType& x, const UserAtomicType& y)
{
return x.i == y.i;
}
};
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
template <typename, typename> class Selector,
cuda::thread_scope Scope
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
= cuda::thread_scope_system
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
>
struct TestEachIntegralType
{
TEST_FUNC void operator()() const
{
TestFunctor<char, Selector, Scope>()();
TestFunctor<signed char, Selector, Scope>()();
TestFunctor<unsigned char, Selector, Scope>()();
TestFunctor<short, Selector, Scope>()();
TestFunctor<unsigned short, Selector, Scope>()();
TestFunctor<int, Selector, Scope>()();
TestFunctor<unsigned int, Selector, Scope>()();
TestFunctor<long, Selector, Scope>()();
TestFunctor<unsigned long, Selector, Scope>()();
TestFunctor<long long, Selector, Scope>()();
TestFunctor<unsigned long long, Selector, Scope>()();
TestFunctor<wchar_t, Selector, Scope>();
TestFunctor<char16_t, Selector, Scope>()();
TestFunctor<char32_t, Selector, Scope>()();
TestFunctor<int8_t, Selector, Scope>()();
TestFunctor<uint8_t, Selector, Scope>()();
TestFunctor<int16_t, Selector, Scope>()();
TestFunctor<uint16_t, Selector, Scope>()();
TestFunctor<int32_t, Selector, Scope>()();
TestFunctor<uint32_t, Selector, Scope>()();
TestFunctor<int64_t, Selector, Scope>()();
TestFunctor<uint64_t, Selector, Scope>()();
}
};
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
template <typename, typename> class Selector,
cuda::thread_scope Scope
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
= cuda::thread_scope_system
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
>
struct TestEachFloatingPointType
{
TEST_FUNC void operator()() const
{
TestFunctor<float, Selector, Scope>()();
TestFunctor<double, Selector, Scope>()();
}
};
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
template <typename, typename> class Selector,
cuda::thread_scope Scope
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
= cuda::thread_scope_system
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
>
struct TestEachAtomicType
{
TEST_FUNC void operator()() const
{
TestEachIntegralType<TestFunctor, Selector, Scope>()();
TestEachFloatingPointType<TestFunctor, Selector, Scope>()();
TestFunctor<UserAtomicType, Selector, Scope>()();
TestFunctor<int*, Selector, Scope>()();
TestFunctor<const int*, Selector, Scope>()();
}
};
#endif // ATOMIC_HELPER_H

View File

@@ -0,0 +1,13 @@
// -*- C++ -*-
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,137 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: windows && pre-sm-70
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
template <typename T>
TEST_DEVICE_FUNC T store(T in)
{
cuda::atomic<T> x(in);
x.store(in + 1, cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T compare_exchange_weak(T in)
{
cuda::atomic<T> x(in);
T old = T(7);
x.compare_exchange_weak(old, T(42), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T compare_exchange_strong(T in)
{
cuda::atomic<T> x(in);
T old = T(7);
x.compare_exchange_strong(old, T(42), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T exchange(T in)
{
cuda::atomic<T> x(in);
T out = x.exchange(T(1), cuda::memory_order_relaxed);
return out + x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_add(T in)
{
cuda::atomic<T> x(in);
x.fetch_add(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_sub(T in)
{
cuda::atomic<T> x(in);
x.fetch_sub(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_and(T in)
{
cuda::atomic<T> x(in);
x.fetch_and(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_or(T in)
{
cuda::atomic<T> x(in);
x.fetch_or(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_xor(T in)
{
cuda::atomic<T> x(in);
x.fetch_xor(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_min(T in)
{
cuda::atomic<T> x(in);
x.fetch_min(T(7), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_max(T in)
{
cuda::atomic<T> x(in);
x.fetch_max(T(7), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC inline void tests()
{
const T tid = threadIdx.x;
assert(tid + T(1) == store(tid));
assert(T(1) + tid == exchange(tid));
assert(tid == T(7) ? T(42) : tid == compare_exchange_weak(tid));
assert(tid == T(7) ? T(42) : tid == compare_exchange_strong(tid));
assert((tid + T(1)) == fetch_add(tid));
assert((tid & T(1)) == fetch_and(tid));
assert((tid | T(1)) == fetch_or(tid));
assert((tid ^ T(1)) == fetch_xor(tid));
assert(min(tid, T(7)) == fetch_min(tid));
assert(max(tid, T(7)) == fetch_max(tid));
assert(T(tid - T(1)) == fetch_sub(tid));
}
int main(int arg, char** argv)
{
#if !defined(_CCCL_ATOMIC_UNSAFE_AUTOMATIC_STORAGE)
NV_IF_ELSE_TARGET(
NV_IS_HOST,
(cuda_thread_count = 64;),
(tests<uint8_t>(); tests<uint16_t>(); tests<uint32_t>(); tests<uint64_t>(); tests<int8_t>(); tests<int16_t>();
tests<int32_t>();
tests<int64_t>();))
#endif
return 0;
}

View File

@@ -0,0 +1,119 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// UNSUPPORTED: nvrtc
// <cuda/atomic>
#define _LIBCUDACXX_FORCE_PTX_AUTOMATIC_STORAGE_PATH 1 // Force using the PTX is_local atomics path
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "test_macros.h"
/*
Test goals:
Pre-load registers with values that will be used to trigger the wrong codepath in local device atomics.
This test is architecture and driver dependent. It is not possible to reproduce this when compiled to SASS on 12.0, but
will repro on 12.8.
Compiled to SASS is an important point, compiling to PTX will show the failure to initialize the local test flag for
isspacep.local to 0, but that might be compiled out by the JIT compiler in the driver
*/
__global__ void __launch_bounds__(1024) device_test(char* gmem)
{
constexpr int threads = 1024;
__shared__ int hidx;
__shared__ int histogram[threads];
cuda::atomic<int, cuda::thread_scope_thread> xatom(0);
constexpr int passes = 16;
constexpr int ops = 32;
constexpr int expected = passes * ops;
if (threadIdx.x == 0)
{
hidx = 0;
memset(histogram, sizeof(histogram), 0);
}
__syncthreads();
for (xatom = 0; xatom.load() < passes; xatom++)
{
using A = cuda::atomic_ref<int, cuda::std::thread_scope_block>;
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 0]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 1]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 2]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 3]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 4]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 5]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 6]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 7]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 8]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 9]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 10]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 11]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 12]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 13]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 14]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 15]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 16]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 17]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 18]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 19]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 20]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 21]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 22]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 23]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 24]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 25]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 26]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 27]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 28]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 29]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 30]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 31]);
}
__syncthreads();
if (histogram[threadIdx.x] != expected)
{
printf("[%i] = %i\r\n", threadIdx.x, histogram[threadIdx.x]);
}
assert(histogram[threadIdx.x] == expected);
}
void launch_kernel()
{
cudaError_t err;
char* inptr = nullptr;
CUDA_CALL(err, cudaGetLastError());
CUDA_CALL(err, cudaMalloc(&inptr, 1024));
CUDA_CALL(err, cudaMemset(inptr, 1, 1024));
device_test<<<1, 1024>>>(inptr);
CUDA_CALL(err, cudaGetLastError());
CUDA_CALL(err, cudaDeviceSynchronize());
}
int main(int arg, char** argv)
{
NV_IF_TARGET(NV_IS_HOST, (launch_kernel();))
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: windows
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
// Check that atomics on host may be constructed
template <class T>
TEST_FUNC void do_test()
{
T v(0);
cuda::atomic_ref<T> a(v);
}
int main(int, char**)
{
do_test<__int128_t>();
do_test<__uint128_t>();
return 0;
}

View File

@@ -0,0 +1,33 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: nvrtc
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
// Check that host atomics fail to build
template <class T>
TEST_FUNC void do_test()
{
T v(0);
cuda::atomic_ref<T> a(v);
a.store(1);
assert(a++ == 1);
assert(a.load() == 2);
}
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST, (do_test<__int128_t>(); do_test<__uint128_t>();))
return 0;
}

View File

@@ -0,0 +1,67 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: pre-sm-70
// UNSUPPORTED: windows
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "cuda_space_selector.h"
#include "test_macros.h"
template <typename T>
TEST_FUNC constexpr T combine_literal(uint64_t lower, uint64_t upper)
{
return T(lower) | (T(upper) << 64);
}
template <template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
TEST_FUNC void test()
{
{
using T = __int128_t;
using A = cuda::atomic_ref<T, ThreadScope>;
Selector<T, constructor_initializer> sel;
T& t = *sel.construct();
t = T(0);
A atom(t);
auto test_v = combine_literal<T>(0x01234567DEADBEEF, 0x1337B33701234567);
atom.store(test_v, cuda::std::memory_order_release);
assert(atom.load() == test_v);
}
{
using T = __uint128_t;
using A = cuda::atomic_ref<T, ThreadScope>;
Selector<T, constructor_initializer> sel;
T& t = *sel.construct();
t = T(0);
A atom(t);
auto test_v = combine_literal<T>(0x01234567DEADBEEF, 0x1337B33701234567);
atom.store(test_v);
assert(atom.load() == test_v);
}
}
int main(int, char**)
{
#if __cccl_ptx_isa >= 840
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70,
(test<local_memory_selector, cuda::thread_scope_thread>(); test<shared_memory_selector, cuda::thread_scope_block>();
test<global_memory_selector, cuda::thread_scope_block>();
test<global_memory_selector, cuda::thread_scope_device>();))
#endif
return 0;
}

View File

@@ -0,0 +1,99 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "test_macros.h"
/*
Test goals:
Interleaved 8b/16b access to a 32b window while there is thread contention.
for 8b:
Launch 1024 threads, fetch_add(1) each window, value at end of kernel should be 0xFF..FF. This checks for corruption
caused by interleaved access to different parts of the window.
for 16b:
Launch 1024 threads, fetch_add(1), checking for 0x01FF01FF.
*/
template <class T, int Inc>
TEST_FUNC void fetch_add_into_window(T* window, uint16_t* atomHistory)
{
using Atom = cuda::atomic_ref<T, cuda::thread_scope_block>;
Atom a(*window);
*atomHistory = a.fetch_add(Inc);
}
template <class T>
TEST_DEVICE_FUNC void device_do_test(uint32_t expected)
{
constexpr uint32_t threadCount = 1024;
constexpr uint32_t histogramResultCount = 256 * sizeof(T);
constexpr uint32_t histogramEntriesPerThread = 4 / sizeof(T);
__shared__ uint16_t atomHistory[threadCount];
__shared__ uint8_t atomHistogram[histogramResultCount];
__shared__ uint32_t atomicStorage;
cuda::atomic_ref<uint32_t, cuda::thread_scope_block> bucket(atomicStorage);
constexpr uint32_t offsetMask = ((4 / sizeof(T)) - 1);
// Access offset is interleaved meaning threads 4, 5, 6, 7 access window 0, 1, 2, 3 and so on.
const uint32_t threadOffset = threadIdx.x & offsetMask;
if (threadIdx.x == 0)
{
memset(atomHistogram, 0, histogramResultCount);
bucket.store(0);
}
__syncthreads();
T* window = reinterpret_cast<T*>(&atomicStorage) + threadOffset;
fetch_add_into_window<T, 1>(window, atomHistory + threadIdx.x);
__syncthreads();
if (threadIdx.x == 0)
{
// For each thread, add its atomic result into the corresponding bucket
for (uint32_t i = 0; i < threadCount; i++)
{
atomHistogram[atomHistory[i]]++;
}
// Check that each bucket has exactly (4 / sizeof(T)) entries
// This checks that atomic fetch operations were sequential. i.e. 4xfetch_add(1) returns [0, 1, 2, 3]
for (uint32_t i = 0; i < histogramResultCount; i++)
{
assert(atomHistogram[i] == histogramEntriesPerThread);
}
printf("expected: 0x%X\r\n", expected);
printf("result: 0x%X\r\n", bucket.load());
assert(bucket.load() == expected);
}
}
int main(int, char**)
{
NV_DISPATCH_TARGET(NV_IS_HOST,
(cuda_thread_count = 1024;),
NV_IS_DEVICE,
(device_do_test<uint8_t>(0); device_do_test<uint16_t>(0x02000200);));
return 0;
}

View File

@@ -0,0 +1,77 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// <cuda/atomic>
// cuda::atomic<key>
// Original test issue:
// https://github.com/NVIDIA/libcudacxx/issues/160
#include <cuda/atomic>
#include "cuda_space_selector.h"
#include "test_macros.h"
template <template <typename, typename> class Selector>
struct TestFn
{
TEST_FUNC void operator()() const
{
{
struct key
{
int32_t a;
int32_t b;
};
using A = cuda::std::atomic<key>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
cuda::std::atomic_init(&t, key{1, 2});
auto r = t.load();
auto d = key{5, 5};
t.store(r);
(void) t.exchange(r);
(void) t.compare_exchange_weak(r, d, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
(void) t.compare_exchange_strong(d, r, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
}
{
struct alignas(8) key
{
int32_t a;
int32_t b;
};
using A = cuda::std::atomic<key>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
cuda::std::atomic_init(&t, key{1, 2});
auto r = t.load();
auto d = key{5, 5};
t.store(r);
(void) t.exchange(r);
(void) t.compare_exchange_weak(r, d, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
(void) t.compare_exchange_strong(d, r, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
}
}
};
int main(int, char**)
{
NV_DISPATCH_TARGET(NV_IS_HOST, TestFn<local_memory_selector>()();
, NV_PROVIDES_SM_70, TestFn<local_memory_selector>()();)
NV_IF_TARGET(NV_IS_DEVICE, (TestFn<shared_memory_selector>()(); TestFn<global_memory_selector>()();))
return 0;
}

View File

@@ -0,0 +1,87 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef TEST_ARRIVE_TX_H_
#define TEST_ARRIVE_TX_H_
#include <cuda/barrier>
#include <cuda/memory>
#include <cuda/std/utility>
#include "concurrent_agents.h"
#include "cuda_space_selector.h"
#include "test_macros.h"
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
template <typename Barrier>
inline TEST_DEVICE_FUNC void mbarrier_complete_tx(Barrier& b, int transaction_count)
{
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_90,
(
if (cuda::device::is_address_from(cuda::device::barrier_native_handle(b), cuda::device::address_space::shared)) {
asm volatile(
"mbarrier.complete_tx.relaxed.cta.shared::cta.b64 [%0], %1;"
:
: "r"((unsigned int) __cvta_generic_to_shared(cuda::device::barrier_native_handle(b))), "r"(transaction_count)
: "memory");
} else { __trap(); }),
NV_ANY_TARGET,
(
// On architectures pre-SM90 (and on host), we drop the transaction count
// update. The barriers do not keep track of transaction counts.
__trap();));
}
template <bool split_arrive_and_expect>
TEST_DEVICE_FUNC void thread(cuda::barrier<cuda::thread_scope_block>& b, int arrives_per_thread)
{
constexpr int tx_count = 1;
typename cuda::barrier<cuda::thread_scope_block>::arrival_token tok;
if _CCCL_CONSTEXPR_CXX20 (split_arrive_and_expect)
{
cuda::device::barrier_expect_tx(b, tx_count);
tok = b.arrive(arrives_per_thread);
}
else
{
tok = cuda::device::barrier_arrive_tx(b, arrives_per_thread, tx_count);
}
// Manually increase the transaction count of the barrier.
mbarrier_complete_tx(b, tx_count);
b.wait(cuda::std::move(tok));
}
template <bool split_arrive_and_expect>
TEST_DEVICE_FUNC void test()
{
NV_DISPATCH_TARGET(
NV_IS_DEVICE,
(
// Run all threads, each arriving with arrival count 1
using barrier_t = cuda::barrier<cuda::thread_scope_block>;
shared_memory_selector<barrier_t, constructor_initializer> sel_1;
barrier_t* bar_1 = sel_1.construct(blockDim.x);
__syncthreads();
thread<split_arrive_and_expect>(*bar_1, 1);
// Run all threads, each arriving with arrival count 2
shared_memory_selector<barrier_t, constructor_initializer> sel_2;
barrier_t* bar_2 = sel_2.construct(2 * blockDim.x);
__syncthreads();
thread<split_arrive_and_expect>(*bar_2, 2);));
}
#endif // TEST_ARRIVE_TX_H_

View File

@@ -0,0 +1,56 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: clang && !nvcc
// UNSUPPORTED: no_execute
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cooperative_groups.h>
#include "test_macros.h"
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// When PR #416 is merged, uncomment this line:
// cuda_cluster_size = 2;
),
NV_IS_DEVICE,
(__shared__ cuda::barrier<cuda::thread_scope_block> bar;
if (threadIdx.x == 0) { init(&bar, blockDim.x); } namespace cg = cooperative_groups;
auto cluster = cg::this_cluster();
cluster.sync();
// This test currently fails at this point because support for
// clusters has not yet been added.
cuda::barrier<cuda::thread_scope_block> * remote_bar;
remote_bar = cluster.map_shared_rank(&bar, cluster.block_rank() ^ 1);
// When PR #416 is merged, this should fail here because the barrier
// is in device memory.
auto token = cuda::device::barrier_arrive_tx(*remote_bar, 1, 0);));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 256;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,42 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: no_execute
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include "test_macros.h"
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
TEST_DEVICE_FUNC uint64_t bar_storage;
int main(int, char**)
{
NV_IF_TARGET(
NV_IS_DEVICE,
(cuda::barrier<cuda::thread_scope_block> * bar_ptr;
bar_ptr = reinterpret_cast<cuda::barrier<cuda::thread_scope_block>*>(bar_storage);
if (threadIdx.x == 0) { init(bar_ptr, blockDim.x); } __syncthreads();
// Should fail because the barrier is in device memory.
[[maybe_unused]] auto token = cuda::device::barrier_arrive_tx(*bar_ptr, 1, 0);));
return 0;
}

View File

@@ -0,0 +1,28 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#ifndef __cccl_lib_local_barrier_arrive_tx
static_assert(false, "should define __cccl_lib_local_barrier_arrive_tx");
#endif // __cccl_lib_local_barrier_arrive_tx
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-70
// <cuda/barrier>
#include <cuda/barrier>
int main(int, char**)
{
NV_IF_TARGET(
NV_IS_DEVICE,
(__shared__ cuda::barrier<cuda::thread_scope_block> bar;
if (threadIdx.x == 0) { init(&bar, blockDim.x); } __syncthreads();
// barrier_arrive_tx should fail on SM70 and SM80, because it is hidden.
auto token = cuda::device::barrier_arrive_tx(bar, 1, 0);
#ifdef __cccl_lib_local_barrier_arrive_tx
static_assert(false, "Fail manually for SM90 and up.");
#endif // __cccl_lib_local_barrier_arrive_tx
));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 2;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 32;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,107 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/utility> // cuda::std::move
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
using barrier = cuda::barrier<cuda::thread_scope_block>;
namespace cde = cuda::device::experimental;
static constexpr int buf_len = 1024;
alignas(128) TEST_GLOBAL_VARIABLE int gmem_buffer[buf_len];
TEST_DEVICE_FUNC void test()
{
// SETUP: fill global memory buffer
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
{
gmem_buffer[i] = i;
}
// Ensure that writes to global memory are visible to others, including
// those in the async proxy.
__threadfence();
__syncthreads();
// TEST: Add i to buffer[i]
alignas(16) __shared__ int smem_buffer[buf_len];
#if _CCCL_CUDA_COMPILER(CLANG)
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
__shared__ barrier bar;
#endif // !_CCCL_CUDA_COMPILER(CLANG)
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
// Load data:
uint64_t token;
if (threadIdx.x == 0)
{
cde::cp_async_bulk_global_to_shared(smem_buffer, gmem_buffer, sizeof(smem_buffer), bar);
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
}
else
{
token = bar.arrive();
}
bar.wait(cuda::std::move(token));
// Update in shared memory
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
{
smem_buffer[i] += i;
}
cde::fence_proxy_async_shared_cta();
__syncthreads();
// Write back to global memory:
if (threadIdx.x == 0)
{
cde::cp_async_bulk_shared_to_global(gmem_buffer, smem_buffer, sizeof(smem_buffer));
cde::cp_async_bulk_commit_group();
cde::cp_async_bulk_wait_group_read<0>();
}
__threadfence();
__syncthreads();
// TEAR-DOWN: check that global memory is correct
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
{
assert(gmem_buffer[i] == 2 * i);
}
}
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512;));
NV_DISPATCH_TARGET(NV_IS_DEVICE, (test();));
return 0;
}

View File

@@ -0,0 +1,28 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#ifndef __cccl_lib_experimental_ctk12_cp_async_exposure
static_assert(false, "should define __cccl_lib_experimental_ctk12_cp_async_exposure");
#endif
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,95 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
using barrier = cuda::barrier<cuda::thread_scope_block>;
namespace cde = cuda::device::experimental;
// Kernels below are intended to be compiled, but not run. This is to check if
// all generated PTX is valid.
__global__ void test_bulk_tensor(CUtensorMap* map)
{
__shared__ int smem;
#if _CCCL_CUDA_COMPILER(CLANG)
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
__shared__ barrier bar;
#endif // !_CCCL_CUDA_COMPILER(CLANG)
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
cde::cp_async_bulk_tensor_1d_global_to_shared(&smem, map, 0, bar);
cde::cp_async_bulk_tensor_2d_global_to_shared(&smem, map, 0, 0, bar);
cde::cp_async_bulk_tensor_3d_global_to_shared(&smem, map, 0, 0, 0, bar);
cde::cp_async_bulk_tensor_4d_global_to_shared(&smem, map, 0, 0, 0, 0, bar);
cde::cp_async_bulk_tensor_5d_global_to_shared(&smem, map, 0, 0, 0, 0, 0, bar);
cde::cp_async_bulk_tensor_1d_shared_to_global(map, 0, &smem);
cde::cp_async_bulk_tensor_2d_shared_to_global(map, 0, 0, &smem);
cde::cp_async_bulk_tensor_3d_shared_to_global(map, 0, 0, 0, &smem);
cde::cp_async_bulk_tensor_4d_shared_to_global(map, 0, 0, 0, 0, &smem);
cde::cp_async_bulk_tensor_5d_shared_to_global(map, 0, 0, 0, 0, 0, &smem);
}
__global__ void test_bulk(void* gmem)
{
__shared__ int smem;
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
cde::cp_async_bulk_global_to_shared(&smem, gmem, 1024, bar);
cde::cp_async_bulk_shared_to_global(gmem, &smem, 1024);
}
__global__ void test_fences_async_group(void*)
{
cde::fence_proxy_async_shared_cta();
cde::cp_async_bulk_commit_group();
// Wait for up to 8 groups
cde::cp_async_bulk_wait_group_read<0>();
cde::cp_async_bulk_wait_group_read<1>();
cde::cp_async_bulk_wait_group_read<2>();
cde::cp_async_bulk_wait_group_read<3>();
cde::cp_async_bulk_wait_group_read<4>();
cde::cp_async_bulk_wait_group_read<5>();
cde::cp_async_bulk_wait_group_read<6>();
cde::cp_async_bulk_wait_group_read<7>();
cde::cp_async_bulk_wait_group_read<8>();
}
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,221 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: clang && !nvcc
// UNSUPPORTED: nvrtc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/utility> // cuda::std::move
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
// NVRTC does not support cuda.h (due to import of stdlib.h)
#if !TEST_COMPILER(NVRTC)
# include <cudaTypedefs.h> // PFN_cuTensorMapEncodeTiled, CUtensorMap
#endif // !TEST_COMPILER(NVRTC)
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
using barrier = cuda::barrier<cuda::thread_scope_block>;
namespace cde = cuda::device::experimental;
constexpr size_t GMEM_WIDTH = 1024; // Width of tensor (in # elements)
constexpr size_t GMEM_HEIGHT = 1024; // Height of tensor (in # elements)
constexpr size_t gmem_len = GMEM_WIDTH * GMEM_HEIGHT;
constexpr int SMEM_WIDTH = 32; // Width of shared memory buffer (in # elements)
constexpr int SMEM_HEIGHT = 8; // Height of shared memory buffer (in # elements)
static constexpr int buf_len = SMEM_HEIGHT * SMEM_WIDTH;
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
// We need a type with a size. On NVRTC, cuda.h cannot be imported, so we don't
// have access to the definition of CUTensorMap (only to the declaration of CUtensorMap inside
// cuda/barrier). So we use this type instead and reinterpret_cast in the
// kernel.
struct fake_cutensormap
{
alignas(64) uint64_t opaque[16];
};
__constant__ fake_cutensormap global_fake_tensor_map;
TEST_DEVICE_FUNC void test(int base_i, int base_j)
{
CUtensorMap* global_tensor_map = reinterpret_cast<CUtensorMap*>(&global_fake_tensor_map);
// SETUP: fill global memory buffer
for (int i = threadIdx.x; i < static_cast<int>(gmem_len); i += blockDim.x)
{
gmem_tensor[i] = i;
}
// Ensure that writes to global memory are visible to others, including
// those in the async proxy.
__threadfence();
__syncthreads();
// TEST: Add i to buffer[i]
alignas(128) __shared__ int smem_buffer[buf_len];
#if _CCCL_CUDA_COMPILER(CLANG)
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
__shared__ barrier bar;
#endif // !_CCCL_CUDA_COMPILER(CLANG)
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
// Load data:
uint64_t token;
if (threadIdx.x == 0)
{
// Fastest moving coordinate first.
cde::cp_async_bulk_tensor_2d_global_to_shared(smem_buffer, global_tensor_map, base_j, base_i, bar);
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
}
else
{
token = bar.arrive();
}
bar.wait(cuda::std::move(token));
// Check smem
for (int i = 0; i < SMEM_HEIGHT; ++i)
{
for (int j = 0; j < SMEM_HEIGHT; ++j)
{
const int gmem_lin_idx = (base_i + i) * GMEM_WIDTH + base_j + j;
const int smem_lin_idx = i * SMEM_WIDTH + j;
assert(smem_buffer[smem_lin_idx] == gmem_lin_idx);
}
}
__syncthreads();
// Update smem
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
{
smem_buffer[i] = 2 * smem_buffer[i] + 1;
}
cde::fence_proxy_async_shared_cta();
__syncthreads();
// Write back to global memory:
if (threadIdx.x == 0)
{
cde::cp_async_bulk_tensor_2d_shared_to_global(global_tensor_map, base_j, base_i, smem_buffer);
cde::cp_async_bulk_commit_group();
cde::cp_async_bulk_wait_group_read<0>();
}
__threadfence();
__syncthreads();
// TEAR-DOWN: check that global memory is correct
for (int i = 0; i < SMEM_HEIGHT; ++i)
{
for (int j = 0; j < SMEM_HEIGHT; ++j)
{
int gmem_lin_idx = (base_i + i) * GMEM_WIDTH + base_j + j;
assert(gmem_tensor[gmem_lin_idx] == 2 * gmem_lin_idx + 1);
}
}
__syncthreads();
}
#if !TEST_COMPILER(NVRTC)
# if _CCCL_CTK_BELOW(12, 5)
PFN_cuTensorMapEncodeTiled get_cuTensorMapEncodeTiled()
{
void* driver_ptr = nullptr;
cudaDriverEntryPointQueryResult driver_status;
auto code = cudaGetDriverEntryPoint("cuTensorMapEncodeTiled", &driver_ptr, cudaEnableDefault, &driver_status);
assert(code == cudaSuccess && "Could not get driver API");
return reinterpret_cast<PFN_cuTensorMapEncodeTiled>(driver_ptr);
}
# else // ^^^ _CCCL_CTK_BELOW(12, 5) ^^^ / vvv _CCCL_CTK_AT_LEAST(12, 5) vvv
PFN_cuTensorMapEncodeTiled_v12000 get_cuTensorMapEncodeTiled()
{
void* driver_ptr = nullptr;
cudaDriverEntryPointQueryResult driver_status;
auto code =
cudaGetDriverEntryPointByVersion("cuTensorMapEncodeTiled", &driver_ptr, 12000, cudaEnableDefault, &driver_status);
assert(code == cudaSuccess && "Could not get driver API");
return reinterpret_cast<PFN_cuTensorMapEncodeTiled_v12000>(driver_ptr);
}
# endif // _CCCL_CTK_AT_LEAST(12, 5)
#endif // !TEST_COMPILER(NVRTC)
int main(int, char**)
{
NV_IF_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512;
int* tensor_ptr = nullptr;
auto code = cudaGetSymbolAddress((void**) &tensor_ptr, gmem_tensor);
assert(code == cudaSuccess && "getsymboladdress failed.");
// https://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TENSOR__MEMORY.html
CUtensorMap local_tensor_map{};
// rank is the number of dimensions of the array.
constexpr uint32_t rank = 2;
uint64_t size[rank] = {GMEM_WIDTH, GMEM_HEIGHT};
// The stride is the number of bytes to traverse from the first element of one row to the next.
// It must be a multiple of 16.
uint64_t stride[rank - 1] = {GMEM_WIDTH * sizeof(int)};
// The box_size is the size of the shared memory buffer that is used as the
// destination of a TMA transfer.
uint32_t box_size[rank] = {SMEM_WIDTH, SMEM_HEIGHT};
// The distance between elements in units of sizeof(element). A stride of 2
// can be used to load only the real component of a complex-valued tensor, for instance.
uint32_t elem_stride[rank] = {1, 1};
// Get a function pointer to the cuTensorMapEncodeTiled driver API.
auto cuTensorMapEncodeTiled = get_cuTensorMapEncodeTiled();
// Create the tensor descriptor.
CUresult res = cuTensorMapEncodeTiled(
&local_tensor_map, // CUtensorMap *tensorMap,
CUtensorMapDataType::CU_TENSOR_MAP_DATA_TYPE_INT32,
rank, // cuuint32_t tensorRank,
tensor_ptr, // void *globalAddress,
size, // const cuuint64_t *globalDim,
stride, // const cuuint64_t *globalStrides,
box_size, // const cuuint32_t *boxDim,
elem_stride, // const cuuint32_t *elementStrides,
CUtensorMapInterleave::CU_TENSOR_MAP_INTERLEAVE_NONE,
CUtensorMapSwizzle::CU_TENSOR_MAP_SWIZZLE_NONE,
CUtensorMapL2promotion::CU_TENSOR_MAP_L2_PROMOTION_NONE,
CUtensorMapFloatOOBfill::CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE);
assert(res == CUDA_SUCCESS && "tensormap creation failed.");
code = cudaMemcpyToSymbol(global_fake_tensor_map, &local_tensor_map, sizeof(CUtensorMap));
assert(code == cudaSuccess && "memcpytosymbol failed.");));
NV_DISPATCH_TARGET(NV_IS_DEVICE, (test(0, 0); test(4, 0); test(4, 4);));
return 0;
}

View File

@@ -0,0 +1,64 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 1> GMEM_DIMS{256};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 1> GMEM_DIMS_DEV{256};
constexpr cuda::std::array<uint32_t, 1> SMEM_DIMS{32};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 1> SMEM_DIMS_DEV{32};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 1> TEST_SMEM_COORDS[] = {{0}, {4}, {8}};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,69 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 2> GMEM_DIMS{8, 11};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 2> GMEM_DIMS_DEV{8, 11};
constexpr cuda::std::array<uint32_t, 2> SMEM_DIMS{4, 2};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 2> SMEM_DIMS_DEV{4, 2};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 2> TEST_SMEM_COORDS[] = {
{0, 0},
{4, 1},
{4, 5},
{0, 5},
};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,64 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 3> GMEM_DIMS{8, 11, 13};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 3> GMEM_DIMS_DEV{8, 11, 13};
constexpr cuda::std::array<uint32_t, 3> SMEM_DIMS{4, 2, 4};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 3> SMEM_DIMS_DEV{4, 2, 4};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 3> TEST_SMEM_COORDS[] = {{0, 0, 0}, {4, 1, 3}, {4, 5, 1}};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,65 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 4> GMEM_DIMS{8, 11, 13, 3};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 4> GMEM_DIMS_DEV{8, 11, 13, 3};
constexpr cuda::std::array<uint32_t, 4> SMEM_DIMS{4, 2, 4, 1};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 4> SMEM_DIMS_DEV{4, 2, 4, 1};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 4> TEST_SMEM_COORDS[] = {
{0, 0, 0, 0}, {4, 1, 3, 0}, {4, 8, 7, 2}, {4, 5, 1, 1}};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,65 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 5> GMEM_DIMS{8, 11, 13, 3, 3};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 5> GMEM_DIMS_DEV{8, 11, 13, 3, 3};
constexpr cuda::std::array<uint32_t, 5> SMEM_DIMS{4, 2, 4, 1, 1};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 5> SMEM_DIMS_DEV{4, 2, 4, 1, 1};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 5> TEST_SMEM_COORDS[] = {
{0, 0, 0, 0, 0}, {4, 1, 3, 0, 1}, {4, 5, 1, 1, 2}};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,349 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// <cuda/barrier>
#ifndef TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
#define TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
#include <cuda/barrier>
#include <cuda/ptx>
#include <cuda/std/array>
#include <cuda/std/utility> // cuda::std::move
namespace ptx = cuda::ptx;
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
// NVRTC does not support cuda.h (due to import of stdlib.h)
#if !TEST_COMPILER(NVRTC)
# include <cstdio>
# include <cudaTypedefs.h> // PFN_cuTensorMapEncodeTiled, CUtensorMap
#endif // ! TEST_COMPILER(NVRTC)
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
using barrier = cuda::barrier<cuda::thread_scope_block>;
namespace cde = cuda::device::experimental;
/*
* This header supports the 1d, 2d, ..., 5d test of the TMA PTX wrappers.
*
* The functions below help convert Nd coordinates into something useful.
*
*/
// Compute the total number of elements in a tensor
template <class T, size_t num_dims>
constexpr TEST_FUNC int tensor_len(cuda::std::array<T, num_dims> dims)
{
T len = 1;
for (T d : dims)
{
len *= d;
}
return static_cast<int>(len);
}
// Function to convert:
// a linear index into a shared memory tensor
// into
// a linear index into a global memory tensor.
template <size_t num_dims>
inline TEST_DEVICE_FUNC int smem_lin_idx_to_gmem_lin_idx(
int smem_lin_idx,
cuda::std::array<uint32_t, num_dims> smem_coord,
cuda::std::array<uint32_t, num_dims> smem_dims,
cuda::std::array<uint64_t, num_dims> gmem_dims)
{
assert(smem_coord.size() == smem_dims.size());
assert(smem_coord.size() == gmem_dims.size());
int gmem_lin_idx = 0;
int gmem_stride = 1;
for (int i = 0; i < (int) smem_coord.size(); ++i)
{
int smem_i_idx = smem_lin_idx % smem_dims.begin()[i];
gmem_lin_idx += (smem_coord.begin()[i] + smem_i_idx) * gmem_stride;
smem_lin_idx /= smem_dims.begin()[i];
gmem_stride *= gmem_dims.begin()[i];
}
return gmem_lin_idx;
}
template <size_t num_dims>
TEST_DEVICE_FUNC inline void cp_tensor_global_to_shared(
CUtensorMap* tensor_map, cuda::std::array<uint32_t, num_dims> indices, void* smem, barrier& bar)
{
switch (indices.size())
{
case 1:
cde::cp_async_bulk_tensor_1d_global_to_shared(smem, tensor_map, indices[0], bar);
break;
case 2:
cde::cp_async_bulk_tensor_2d_global_to_shared(smem, tensor_map, indices[0], indices[1], bar);
break;
case 3:
cde::cp_async_bulk_tensor_3d_global_to_shared(smem, tensor_map, indices[0], indices[1], indices[2], bar);
break;
case 4:
cde::cp_async_bulk_tensor_4d_global_to_shared(
smem, tensor_map, indices[0], indices[1], indices[2], indices[3], bar);
break;
case 5:
cde::cp_async_bulk_tensor_5d_global_to_shared(
smem, tensor_map, indices[0], indices[1], indices[2], indices[3], indices[4], bar);
break;
default:
assert(false && "Wrong number of dimensions.");
}
}
template <size_t num_dims>
TEST_DEVICE_FUNC inline void
cp_tensor_shared_to_global(CUtensorMap* tensor_map, cuda::std::array<uint32_t, num_dims> indices, void* smem)
{
switch (indices.size())
{
case 1:
cde::cp_async_bulk_tensor_1d_shared_to_global(tensor_map, indices[0], smem);
break;
case 2:
cde::cp_async_bulk_tensor_2d_shared_to_global(tensor_map, indices[0], indices[1], smem);
break;
case 3:
cde::cp_async_bulk_tensor_3d_shared_to_global(tensor_map, indices[0], indices[1], indices[2], smem);
break;
case 4:
cde::cp_async_bulk_tensor_4d_shared_to_global(tensor_map, indices[0], indices[1], indices[2], indices[3], smem);
break;
case 5:
cde::cp_async_bulk_tensor_5d_shared_to_global(
tensor_map, indices[0], indices[1], indices[2], indices[3], indices[4], smem);
break;
default:
assert(false && "Wrong number of dimensions.");
}
}
// To define a tensor map in constant memory, we need a type with a size. On
// NVRTC, cuda.h cannot be imported, so we don't have access to the definition
// of CUTensorMap (only to the declaration of CUtensorMap inside cuda/barrier).
// So we use this type instead and reinterpret_cast in the kernel.
struct fake_cutensormap
{
alignas(64) uint64_t opaque[16];
};
__constant__ fake_cutensormap global_fake_tensor_map;
/*
* This test has as primary purpose to make sure that the indices in the mapping
* from C++ to PTX didn't get mixed up.
*
* How does it test this?
*
* 1. It fills a global memory tensor with linear coordinates 0, 1, ...
* 2. It loads a tile into shared memory at some coordinate (x, y, ... )
* 3. It checks that the coordinates that were received in shared memory match the expected.
* 4. It modifies the coordinates (c = 2 * c + 1)
* 5. It writes the tile back to global memory
* 6. It checks that all the values in global are properly modified.
*/
template <size_t smem_len, size_t num_dims>
TEST_DEVICE_FUNC void
test(cuda::std::array<uint32_t, num_dims> smem_coord,
cuda::std::array<uint32_t, num_dims> smem_dims,
cuda::std::array<uint64_t, num_dims> gmem_dims,
int* gmem_tensor,
int gmem_len)
{
CUtensorMap* global_tensor_map = reinterpret_cast<CUtensorMap*>(&global_fake_tensor_map);
// SETUP: fill global memory buffer
for (int i = threadIdx.x; i < gmem_len; i += blockDim.x)
{
gmem_tensor[i] = i;
}
// Ensure that writes to global memory are visible to others, including
// those in the async proxy.
// ahendriksen: Issuing threadfence and fence.proxy.async.global. The
// fence.proxy.async.global should suffice, but I am keeping the threadfence
// out of an abundance of caution.
__threadfence();
ptx::fence_proxy_async(ptx::space_global);
__syncthreads();
// TEST: Add i to buffer[i]
alignas(128) __shared__ int smem_buffer[smem_len];
#if _CCCL_CUDA_COMPILER(CLANG)
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
__shared__ barrier bar;
#endif // !_CCCL_CUDA_COMPILER(CLANG)
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
// Load data:
uint64_t token;
if (threadIdx.x == 0)
{
// Fastest moving coordinate first.
cp_tensor_global_to_shared(global_tensor_map, smem_coord, smem_buffer, bar);
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
}
else
{
token = bar.arrive();
}
bar.wait(cuda::std::move(token));
// Check smem
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
{
int gmem_lin_idx = smem_lin_idx_to_gmem_lin_idx(i, smem_coord, smem_dims, gmem_dims);
assert(smem_buffer[i] == gmem_lin_idx);
}
__syncthreads();
// Update smem
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
{
smem_buffer[i] = 2 * smem_buffer[i] + 1;
}
cde::fence_proxy_async_shared_cta();
__syncthreads();
// Write back to global memory:
if (threadIdx.x == 0)
{
cp_tensor_shared_to_global(global_tensor_map, smem_coord, smem_buffer);
cde::cp_async_bulk_commit_group();
cde::cp_async_bulk_wait_group_read<0>();
}
// ahendriksen: Issuing threadfence and fence.proxy.async.global. The
// fence.proxy.async.global should suffice, but I am keeping the threadfence
// out of an abundance of caution.
__threadfence();
ptx::fence_proxy_async(ptx::space_global);
__syncthreads();
// // TEAR-DOWN: check that global memory is correct
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
{
int gmem_lin_idx = smem_lin_idx_to_gmem_lin_idx(i, smem_coord, smem_dims, gmem_dims);
assert(gmem_tensor[gmem_lin_idx] == 2 * gmem_lin_idx + 1);
}
__syncthreads();
}
#if !TEST_COMPILER(NVRTC)
# if _CCCL_CTK_BELOW(12, 5)
PFN_cuTensorMapEncodeTiled get_cuTensorMapEncodeTiled()
{
void* driver_ptr = nullptr;
cudaDriverEntryPointQueryResult driver_status;
auto code = cudaGetDriverEntryPoint("cuTensorMapEncodeTiled", &driver_ptr, cudaEnableDefault, &driver_status);
assert(code == cudaSuccess && "Could not get driver API");
return reinterpret_cast<PFN_cuTensorMapEncodeTiled>(driver_ptr);
}
# else // ^^^ _CCCL_CTK_BELOW(12, 5) ^^^ / vvv _CCCL_CTK_AT_LEAST(12, 5) vvv
PFN_cuTensorMapEncodeTiled_v12000 get_cuTensorMapEncodeTiled()
{
void* driver_ptr = nullptr;
cudaDriverEntryPointQueryResult driver_status;
auto code =
cudaGetDriverEntryPointByVersion("cuTensorMapEncodeTiled", &driver_ptr, 12000, cudaEnableDefault, &driver_status);
assert(code == cudaSuccess && "Could not get driver API");
return reinterpret_cast<PFN_cuTensorMapEncodeTiled_v12000>(driver_ptr);
}
# endif // _CCCL_CTK_AT_LEAST(12, 5)
#endif // !TEST_COMPILER(NVRTC)
#if !TEST_COMPILER(NVRTC)
template <typename T, size_t num_dims>
CUtensorMap map_encode(T* tensor_ptr,
const cuda::std::array<uint64_t, num_dims>& gmem_dims,
const cuda::std::array<uint32_t, num_dims>& smem_dims)
{
// https://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TENSOR__MEMORY.html
CUtensorMap tensor_map{};
// The stride is the number of bytes to traverse from the first element of one row to the next.
// It must be a multiple of 16.
// cuTensorMapEncodeTiled requires that the stride array is a valid pointer, so we add one superfluous element
// This is necessary for num_dims == 1
cuda::std::array<uint64_t, num_dims> stride;
uint64_t base_stride = sizeof(T);
for (size_t i = 0; i < stride.size() - 1; ++i)
{
base_stride *= gmem_dims[i];
stride[i] = base_stride;
}
// The distance between elements in units of sizeof(element). A stride of 2
// can be used to load only the real component of a complex-valued tensor, for instance.
cuda::std::array<uint32_t, num_dims> elem_stride; // = {1, .., 1};
for (size_t i = 0; i < elem_stride.size(); ++i)
{
elem_stride[i] = 1;
}
// Get a function pointer to the cuTensorMapEncodeTiled driver API.
auto cuTensorMapEncodeTiled = get_cuTensorMapEncodeTiled();
// Create the tensor descriptor.
CUresult res = cuTensorMapEncodeTiled(
&tensor_map, // CUtensorMap *tensorMap,
CUtensorMapDataType::CU_TENSOR_MAP_DATA_TYPE_INT32,
num_dims, // cuuint32_t tensorRank,
tensor_ptr, // void *globalAddress,
gmem_dims.data(), // const cuuint64_t *globalDim,
stride.data(), // const cuuint64_t *globalStrides,
smem_dims.data(), // const cuuint32_t *boxDim,
elem_stride.data(), // const cuuint32_t *elementStrides,
CUtensorMapInterleave::CU_TENSOR_MAP_INTERLEAVE_NONE,
CUtensorMapSwizzle::CU_TENSOR_MAP_SWIZZLE_NONE,
CUtensorMapL2promotion::CU_TENSOR_MAP_L2_PROMOTION_NONE,
CUtensorMapFloatOOBfill::CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE);
assert(res == CUDA_SUCCESS && "tensormap creation failed.");
return tensor_map;
}
template <typename T, size_t num_dims>
void init_tensor_map(const T& gmem_tensor_symbol,
const cuda::std::array<uint64_t, num_dims>& gmem_dims,
const cuda::std::array<uint32_t, num_dims>& smem_dims)
{
// Get pointer to gmem_tensor to create tensor map.
int* tensor_ptr = nullptr;
auto code = cudaGetSymbolAddress((void**) &tensor_ptr, gmem_tensor_symbol);
assert(code == cudaSuccess && "Could not get symbol address.");
// Create tensor map
CUtensorMap local_tensor_map = map_encode(tensor_ptr, gmem_dims, smem_dims);
// Copy it to device
code = cudaMemcpyToSymbol(global_fake_tensor_map, &local_tensor_map, sizeof(CUtensorMap));
assert(code == cudaSuccess && "Could not copy symbol to device.");
}
#endif // ! TEST_COMPILER(NVRTC)
#endif // TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 256;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,42 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: no_execute
// <cuda/barrier>
#include <cuda/barrier>
#include "test_macros.h"
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
[[maybe_unused]] TEST_GLOBAL_VARIABLE uint64_t bar_storage;
int main(int, char**)
{
NV_IF_TARGET(
NV_IS_DEVICE,
(cuda::barrier<cuda::thread_scope_block> * bar_ptr;
bar_ptr = reinterpret_cast<cuda::barrier<cuda::thread_scope_block>*>(bar_storage);
if (threadIdx.x == 0) { init(bar_ptr, blockDim.x); } __syncthreads();
// Should fail because the barrier is in device memory.
cuda::device::barrier_expect_tx(*bar_ptr, 1);));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 2;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 32;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,47 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: pre-sm-70
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include <cuda/barrier>
#include "cuda_space_selector.h"
template <cuda::thread_scope Sco, template <typename, typename> class BarrierSelector>
TEST_FUNC void test()
{
cuda::barrier<Sco> b(3);
init(&b, 2);
auto token = b.arrive();
b.arrive_and_wait();
b.wait(std::move(token));
}
template <cuda::thread_scope Sco>
TEST_FUNC void test_select_barrier()
{
test<Sco, local_memory_selector>();
NV_IF_TARGET(NV_IS_DEVICE, (test<Sco, shared_memory_selector>(); test<Sco, global_memory_selector>();))
}
int main(int argc, char** argv)
{
test_select_barrier<cuda::thread_scope_system>();
test_select_barrier<cuda::thread_scope_device>();
test_select_barrier<cuda::thread_scope_block>();
test_select_barrier<cuda::thread_scope_thread>();
return 0;
}

View File

@@ -0,0 +1,41 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: pre-sm-80
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include <cuda/barrier>
#include "cuda_space_selector.h"
#include "test_macros.h"
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
TEST_NV_DIAG_SUPPRESS(set_but_not_used)
TEST_DEVICE_FUNC void test()
{
__shared__ cuda::barrier<cuda::thread_scope_block>* b;
shared_memory_selector<cuda::barrier<cuda::thread_scope_block>, constructor_initializer> sel;
b = sel.construct(2);
[[maybe_unused]] uint64_t token;
asm volatile("mbarrier.arrive.b64 %0, [%1];" : "=l"(token) : "l"(cuda::device::barrier_native_handle(*b)) : "memory");
b->arrive_and_wait();
}
int main(int argc, char** argv)
{
NV_IF_TARGET(NV_PROVIDES_SM_80, test();)
return 0;
}

View File

@@ -0,0 +1,62 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/bit>
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/type_traits>
#include "test_macros.h"
template <typename T>
TEST_FUNC constexpr bool test()
{
using nl = cuda::std::numeric_limits<T>;
constexpr T all_ones = static_cast<T>(~T{0});
constexpr T half_low = all_ones >> (nl::digits / 2u);
constexpr T half_high = static_cast<T>(all_ones << (nl::digits / 2u));
static_assert(cuda::bit_reverse(all_ones) == all_ones);
static_assert(cuda::bit_reverse(T{0}) == T{0});
static_assert(cuda::bit_reverse(half_low) == half_high);
static_assert(cuda::bit_reverse(T{0b11001001}) == (T{0b10010011} << (nl::digits - 8u)));
static_assert(cuda::bit_reverse(T{T{0b10010011} << (nl::digits - 8u)}) == T{0b11001001});
unused(all_ones);
unused(half_low);
unused(half_high);
return true;
}
TEST_FUNC constexpr bool test()
{
test<unsigned char>();
test<unsigned short>();
test<unsigned>();
test<unsigned long>();
test<unsigned long long>();
test<uint8_t>();
test<uint16_t>();
test<uint32_t>();
test<uint64_t>();
test<size_t>();
test<uintmax_t>();
test<uintptr_t>();
#if _CCCL_HAS_INT128()
test<__uint128_t>();
#endif // _CCCL_HAS_INT128()
return true;
}
int main(int, char**)
{
assert(test());
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,32 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/bit>
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/type_traits>
#include "test_macros.h"
int main(int, char**)
{
using T = uint32_t;
static_assert(cuda::bitfield_insert(T{0}, T{0}, -1, 1));
static_assert(cuda::bitfield_insert(T{0}, T{0}, 0, -1));
static_assert(cuda::bitfield_insert(T{0}, T{0}, 0, 33));
static_assert(cuda::bitfield_insert(T{0}, T{0}, 32, 1));
static_assert(cuda::bitfield_insert(T{0}, T{0}, 20, 20));
static_assert(cuda::bitfield_extract(T{0}, -1, 1));
static_assert(cuda::bitfield_extract(T{0}, 0, -1));
static_assert(cuda::bitfield_extract(T{0}, 0, 33));
static_assert(cuda::bitfield_extract(T{0}, 32, 1));
static_assert(cuda::bitfield_extract(T{0}, 20, 20));
return 0;
}

View File

@@ -0,0 +1,84 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/bit>
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/type_traits>
#include "test_macros.h"
template <typename T>
TEST_FUNC constexpr bool test()
{
using nl = cuda::std::numeric_limits<T>;
constexpr T all_ones = static_cast<T>(~T{0});
unused(all_ones);
assert(cuda::bitfield_insert(T{0}, all_ones, 0, 1) == 1);
assert(cuda::bitfield_insert(T{0}, all_ones, 1, 1) == 0b10);
assert(cuda::bitfield_insert(T{0b10}, all_ones, 0, 1) == 0b11);
assert(cuda::bitfield_insert(all_ones, all_ones, 0, 0) == all_ones);
assert(cuda::bitfield_insert(all_ones, all_ones, 0, 1) == all_ones);
assert(cuda::bitfield_insert(all_ones, all_ones, 2, 1) == all_ones);
assert(cuda::bitfield_insert(all_ones, T{0b1000}, 1, 2) == (all_ones & static_cast<T>(~T{0b110})));
assert(cuda::bitfield_insert(T{0}, all_ones, 0, 2) == 0b11);
assert(cuda::bitfield_insert(T{0}, all_ones, 3, 2) == 0b11000);
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, 3, 2) == 0b10111000);
assert(cuda::bitfield_insert(T{0b10100000}, T{0b11}, 3, 2) == 0b10111000);
assert(cuda::bitfield_insert(T{0}, all_ones, nl::digits - 1, 1) == (T{1} << (nl::digits - 1u)));
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, 0, nl::digits) == all_ones);
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, nl::digits, 0) == T{0b10100000});
assert(cuda::bitfield_extract(T{0}, 3, 4) == 0);
assert(cuda::bitfield_extract(T{0b1011}, 0, 1) == 1);
assert(cuda::bitfield_extract(T{0b1011}, 1, 1) == 1);
assert(cuda::bitfield_extract(T{0b1011}, 2, 2) == 0b10);
assert(cuda::bitfield_extract(all_ones, 0, 0) == 0);
assert(cuda::bitfield_extract(all_ones, 0, 4) == 0b1111);
assert(cuda::bitfield_extract(all_ones, 2, 4) == 0b1111);
assert(cuda::bitfield_extract(T{0b1010010}, 0, 2) == 0b10);
assert(cuda::bitfield_extract(T{0b10101100}, 3, 2) == 1);
assert(cuda::bitfield_extract(T{0b10100000}, 3, 3) == 0b100);
assert(cuda::bitfield_extract(T{all_ones}, nl::digits - 1, 1) == 1);
assert(cuda::bitfield_extract(T{0b10100000}, 0, nl::digits) == T{0b10100000});
assert(cuda::bitfield_extract(T{0b10100000}, nl::digits, 0) == 0);
return true;
}
TEST_FUNC constexpr bool test()
{
test<unsigned char>();
test<unsigned short>();
test<unsigned>();
test<unsigned long>();
test<unsigned long long>();
test<uint8_t>();
test<uint16_t>();
test<uint32_t>();
test<uint64_t>();
test<size_t>();
test<uintmax_t>();
test<uintptr_t>();
#if _CCCL_HAS_INT128()
test<__uint128_t>();
#endif // _CCCL_HAS_INT128()
return true;
}
int main(int, char**)
{
assert(test());
static_assert(test());
return 0;
}

Some files were not shown because too many files have changed in this diff Show More