[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
68
cccl_upstream/libcudacxx/test/.gitignore
vendored
Normal file
68
cccl_upstream/libcudacxx/test/.gitignore
vendored
Normal file
@@ -0,0 +1,68 @@
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
|
||||
# C extensions
|
||||
*.so
|
||||
|
||||
# Distribution / packaging
|
||||
.Python
|
||||
env/
|
||||
build/
|
||||
develop-eggs/
|
||||
dist/
|
||||
downloads/
|
||||
eggs/
|
||||
#lib/ # We actually have things checked in to lib/
|
||||
lib64/
|
||||
parts/
|
||||
sdist/
|
||||
var/
|
||||
*.egg-info/
|
||||
.installed.cfg
|
||||
*.egg
|
||||
|
||||
# PyInstaller
|
||||
# Usually these files are written by a python script from a template
|
||||
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
||||
*.manifest
|
||||
*.spec
|
||||
!*.spec/
|
||||
|
||||
# Installer logs
|
||||
pip-log.txt
|
||||
pip-delete-this-directory.txt
|
||||
|
||||
# Unit test / coverage reports
|
||||
htmlcov/
|
||||
.tox/
|
||||
.coverage
|
||||
.cache
|
||||
nosetests.xml
|
||||
coverage.xml
|
||||
|
||||
# Translations
|
||||
*.mo
|
||||
*.pot
|
||||
|
||||
# Django stuff:
|
||||
*.log
|
||||
|
||||
# Sphinx documentation
|
||||
docs/_build/
|
||||
|
||||
# PyBuilder
|
||||
target/
|
||||
|
||||
# MSVC libraries test harness
|
||||
env.lst
|
||||
keep.lst
|
||||
|
||||
# Editor by-products
|
||||
.vscode/
|
||||
|
||||
# Random things
|
||||
build-*
|
||||
|
||||
# Perforce files
|
||||
.p4config
|
||||
102
cccl_upstream/libcudacxx/test/CMakeLists.txt
Normal file
102
cccl_upstream/libcudacxx/test/CMakeLists.txt
Normal file
@@ -0,0 +1,102 @@
|
||||
find_package(Python COMPONENTS Interpreter)
|
||||
if (NOT Python_Interpreter_FOUND)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Failed to find python interpreter, which is required for running tests and building a libcu++ static library."
|
||||
)
|
||||
endif()
|
||||
|
||||
# Determine the host triple to avoid invoking `${CXX} -dumpmachine`.
|
||||
include(GetHostTriple)
|
||||
get_host_triple(LLVM_INFERRED_HOST_TRIPLE)
|
||||
|
||||
set(
|
||||
LLVM_HOST_TRIPLE
|
||||
"${LLVM_INFERRED_HOST_TRIPLE}"
|
||||
CACHE STRING
|
||||
"Host on which LLVM binaries will run"
|
||||
)
|
||||
|
||||
# By default, we target the host, but this can be overridden at CMake
|
||||
# invocation time.
|
||||
set(
|
||||
LLVM_DEFAULT_TARGET_TRIPLE
|
||||
"${LLVM_HOST_TRIPLE}"
|
||||
CACHE STRING
|
||||
"Default target for which LLVM will generate code."
|
||||
)
|
||||
set(TARGET_TRIPLE "${LLVM_DEFAULT_TARGET_TRIPLE}")
|
||||
message(STATUS "LLVM host triple: ${LLVM_HOST_TRIPLE}")
|
||||
message(STATUS "LLVM default target triple: ${LLVM_DEFAULT_TARGET_TRIPLE}")
|
||||
|
||||
set(LIT_EXTRA_ARGS "" CACHE STRING "Use for additional options (e.g. -j12)")
|
||||
find_program(LLVM_DEFAULT_EXTERNAL_LIT lit)
|
||||
set(LLVM_LIT_ARGS "-sv ${LIT_EXTRA_ARGS}")
|
||||
|
||||
# Libcudacxx's main lit tests
|
||||
add_subdirectory(libcudacxx)
|
||||
|
||||
add_subdirectory(cmake)
|
||||
|
||||
# Set appropriate warning levels for MSVC/sane
|
||||
if ("${CMAKE_CUDA_COMPILER_ID}" STREQUAL "NVIDIA")
|
||||
# CUDA 11.5 and down do not support '-use-local-env'
|
||||
if (MSVC)
|
||||
set(
|
||||
headertest_warning_levels_device
|
||||
-Xcompiler=/W4
|
||||
-Xcompiler=/WX
|
||||
-Wno-deprecated-gpu-targets
|
||||
)
|
||||
if ("${CMAKE_CUDA_COMPILER_VERSION}" GREATER_EQUAL "11.6.0")
|
||||
list(APPEND headertest_warning_levels_device --use-local-env)
|
||||
endif()
|
||||
else()
|
||||
set(
|
||||
headertest_warning_levels_device
|
||||
-Wall
|
||||
-Werror
|
||||
all-warnings
|
||||
-Wno-deprecated-gpu-targets
|
||||
)
|
||||
endif()
|
||||
|
||||
if (
|
||||
CCCL_ENABLE_TILE
|
||||
AND "${CMAKE_CUDA_COMPILER_VERSION}" VERSION_LESS_EQUAL "13.2"
|
||||
)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"tile programs require NVCC 13.3 or later; found NVCC ${CMAKE_CUDA_COMPILER_VERSION}"
|
||||
)
|
||||
endif()
|
||||
# Set warnings for Clang as device compiler
|
||||
elseif ("${CMAKE_CUDA_COMPILER_ID}" STREQUAL "Clang")
|
||||
set(
|
||||
headertest_warning_levels_device
|
||||
-Wall
|
||||
-Werror
|
||||
-Wno-unknown-cuda-version
|
||||
-Xclang=-fcuda-allow-variadic-functions
|
||||
)
|
||||
# If the CMAKE_CUDA_COMPILER is unknown, try to use gcc style warnings
|
||||
else()
|
||||
set(headertest_warning_levels_device -Wall -Werror)
|
||||
endif()
|
||||
|
||||
# Set raw host/device warnings
|
||||
if (MSVC)
|
||||
set(headertest_warning_levels_host /W4 /WX)
|
||||
else()
|
||||
set(headertest_warning_levels_host -Wall -Werror)
|
||||
endif()
|
||||
|
||||
# Enable building the nvrtcc project if NVRTC is enabled
|
||||
if (LIBCUDACXX_TEST_WITH_NVRTC)
|
||||
add_subdirectory(utils/nvidia/nvrtc)
|
||||
endif()
|
||||
|
||||
add_subdirectory(nvtarget)
|
||||
add_subdirectory(atomic_codegen)
|
||||
add_subdirectory(simd_codegen)
|
||||
add_subdirectory(debugging)
|
||||
150
cccl_upstream/libcudacxx/test/CREDITS.TXT
Normal file
150
cccl_upstream/libcudacxx/test/CREDITS.TXT
Normal file
@@ -0,0 +1,150 @@
|
||||
This file is a partial list of people who have contributed to the LLVM/libc++
|
||||
project. If you have contributed a patch or made some other contribution to
|
||||
LLVM/libc++, please submit a patch to this file to add yourself, and it will be
|
||||
done!
|
||||
|
||||
The list is sorted by surname and formatted to allow easy grepping and
|
||||
beautification by scripts. The fields are: name (N), email (E), web-address
|
||||
(W), PGP key ID and fingerprint (P), description (D), and snail-mail address
|
||||
(S).
|
||||
|
||||
N: Saleem Abdulrasool
|
||||
E: compnerd@compnerd.org
|
||||
D: Minor patches and Linux fixes.
|
||||
|
||||
N: Dan Albert
|
||||
E: danalbert@google.com
|
||||
D: Android support and test runner improvements.
|
||||
|
||||
N: Dimitry Andric
|
||||
E: dimitry@andric.com
|
||||
D: Visibility fixes, minor FreeBSD portability patches.
|
||||
|
||||
N: Holger Arnold
|
||||
E: holgerar@gmail.com
|
||||
D: Minor fix.
|
||||
|
||||
N: Ruben Van Boxem
|
||||
E: vanboxem dot ruben at gmail dot com
|
||||
D: Initial Windows patches.
|
||||
|
||||
N: David Chisnall
|
||||
E: theraven at theravensnest dot org
|
||||
D: FreeBSD and Solaris ports, libcxxrt support, some atomics work.
|
||||
|
||||
N: Marshall Clow
|
||||
E: mclow.lists@gmail.com
|
||||
E: marshall@idio.com
|
||||
D: C++14 support, patches and bug fixes.
|
||||
|
||||
N: Jonathan B Coe
|
||||
E: jbcoe@me.com
|
||||
D: Implementation of propagate_const.
|
||||
|
||||
N: Glen Joseph Fernandes
|
||||
E: glenjofe@gmail.com
|
||||
D: Implementation of to_address.
|
||||
|
||||
N: Eric Fiselier
|
||||
E: eric@efcs.ca
|
||||
D: LFTS support, patches and bug fixes.
|
||||
|
||||
N: Bill Fisher
|
||||
E: william.w.fisher@gmail.com
|
||||
D: Regex bug fixes.
|
||||
|
||||
N: Matthew Dempsky
|
||||
E: matthew@dempsky.org
|
||||
D: Minor patches and bug fixes.
|
||||
|
||||
N: Google Inc.
|
||||
D: Copyright owner and contributor of the CityHash algorithm
|
||||
|
||||
N: Howard Hinnant
|
||||
E: hhinnant@apple.com
|
||||
D: Architect and primary author of libc++
|
||||
|
||||
N: Hyeon-bin Jeong
|
||||
E: tuhertz@gmail.com
|
||||
D: Minor patches and bug fixes.
|
||||
|
||||
N: Argyrios Kyrtzidis
|
||||
E: kyrtzidis@apple.com
|
||||
D: Bug fixes.
|
||||
|
||||
N: Bruce Mitchener, Jr.
|
||||
E: bruce.mitchener@gmail.com
|
||||
D: Emscripten-related changes.
|
||||
|
||||
N: Michel Morin
|
||||
E: mimomorin@gmail.com
|
||||
D: Minor patches to is_convertible.
|
||||
|
||||
N: Andrew Morrow
|
||||
E: andrew.c.morrow@gmail.com
|
||||
D: Minor patches and Linux fixes.
|
||||
|
||||
N: Michael Park
|
||||
E: mcypark@gmail.com
|
||||
D: Implementation of <variant>.
|
||||
|
||||
N: Arvid Picciani
|
||||
E: aep at exys dot org
|
||||
D: Minor patches and musl port.
|
||||
|
||||
N: Bjorn Reese
|
||||
E: breese@users.sourceforge.net
|
||||
D: Initial regex prototype
|
||||
|
||||
N: Nico Rieck
|
||||
E: nico.rieck@gmail.com
|
||||
D: Windows fixes
|
||||
|
||||
N: Jon Roelofs
|
||||
E: jroelofS@jroelofs.com
|
||||
D: Remote testing, Newlib port, baremetal/single-threaded support.
|
||||
|
||||
N: Jonathan Sauer
|
||||
D: Minor patches, mostly related to constexpr
|
||||
|
||||
N: Craig Silverstein
|
||||
E: csilvers@google.com
|
||||
D: Implemented Cityhash as the string hash function on 64-bit machines
|
||||
|
||||
N: Richard Smith
|
||||
D: Minor patches.
|
||||
|
||||
N: Joerg Sonnenberger
|
||||
E: joerg@NetBSD.org
|
||||
D: NetBSD port.
|
||||
|
||||
N: Stephan Tolksdorf
|
||||
E: st@quanttec.com
|
||||
D: Minor <atomic> fix
|
||||
|
||||
N: Michael van der Westhuizen
|
||||
E: r1mikey at gmail dot com
|
||||
|
||||
N: Larisse Voufo
|
||||
D: Minor patches.
|
||||
|
||||
N: Klaas de Vries
|
||||
E: klaas at klaasgaaf dot nl
|
||||
D: Minor bug fix.
|
||||
|
||||
N: Zhang Xiongpang
|
||||
E: zhangxiongpang@gmail.com
|
||||
D: Minor patches and bug fixes.
|
||||
|
||||
N: Xing Xue
|
||||
E: xingxue@ca.ibm.com
|
||||
D: AIX port
|
||||
|
||||
N: Zhihao Yuan
|
||||
E: lichray@gmail.com
|
||||
D: Standard compatibility fixes.
|
||||
|
||||
N: Jeffrey Yasskin
|
||||
E: jyasskin@gmail.com
|
||||
E: jyasskin@google.com
|
||||
D: Linux fixes.
|
||||
311
cccl_upstream/libcudacxx/test/LICENSE.TXT
Normal file
311
cccl_upstream/libcudacxx/test/LICENSE.TXT
Normal file
@@ -0,0 +1,311 @@
|
||||
==============================================================================
|
||||
The LLVM Project is under the Apache License v2.0 with LLVM Exceptions:
|
||||
==============================================================================
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
|
||||
|
||||
---- LLVM Exceptions to the Apache 2.0 License ----
|
||||
|
||||
As an exception, if, as a result of your compiling your source code, portions
|
||||
of this Software are embedded into an Object form of such source code, you
|
||||
may redistribute such embedded portions in such Object form without complying
|
||||
with the conditions of Sections 4(a), 4(b) and 4(d) of the License.
|
||||
|
||||
In addition, if you combine or link compiled forms of this Software with
|
||||
software that is licensed under the GPLv2 ("Combined Software") and if a
|
||||
court of competent jurisdiction determines that the patent provision (Section
|
||||
3), the indemnity provision (Section 9) or other Section of the License
|
||||
conflicts with the conditions of the GPLv2, you may retroactively and
|
||||
prospectively choose to deem waived or otherwise exclude such Section(s) of
|
||||
the License, but only in their entirety and only with respect to the Combined
|
||||
Software.
|
||||
|
||||
==============================================================================
|
||||
Software from third parties included in the LLVM Project:
|
||||
==============================================================================
|
||||
The LLVM Project contains third party software which is under different license
|
||||
terms. All such code will be identified clearly using at least one of two
|
||||
mechanisms:
|
||||
1) It will be in a separate directory tree with its own `LICENSE.txt` or
|
||||
`LICENSE` file at the top containing the specific license and restrictions
|
||||
which apply to that software, or
|
||||
2) It will contain specific license and restriction terms at the top of every
|
||||
file.
|
||||
|
||||
==============================================================================
|
||||
Legacy LLVM License (https://llvm.org/docs/DeveloperPolicy.html#legacy):
|
||||
==============================================================================
|
||||
|
||||
The libc++ library is dual licensed under both the University of Illinois
|
||||
"BSD-Like" license and the MIT license. As a user of this code you may choose
|
||||
to use it under either license. As a contributor, you agree to allow your code
|
||||
to be used under both.
|
||||
|
||||
Full text of the relevant licenses is included below.
|
||||
|
||||
==============================================================================
|
||||
|
||||
University of Illinois/NCSA
|
||||
Open Source License
|
||||
|
||||
Copyright (c) 2009-2019 by the contributors listed in CREDITS.TXT
|
||||
|
||||
All rights reserved.
|
||||
|
||||
Developed by:
|
||||
|
||||
LLVM Team
|
||||
|
||||
University of Illinois at Urbana-Champaign
|
||||
|
||||
http://llvm.org
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal with
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
of the Software, and to permit persons to whom the Software is furnished to do
|
||||
so, subject to the following conditions:
|
||||
|
||||
* Redistributions of source code must retain the above copyright notice,
|
||||
this list of conditions and the following disclaimers.
|
||||
|
||||
* Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimers in the
|
||||
documentation and/or other materials provided with the distribution.
|
||||
|
||||
* Neither the names of the LLVM Team, University of Illinois at
|
||||
Urbana-Champaign, nor the names of its contributors may be used to
|
||||
endorse or promote products derived from this Software without specific
|
||||
prior written permission.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS WITH THE
|
||||
SOFTWARE.
|
||||
|
||||
==============================================================================
|
||||
|
||||
Copyright (c) 2009-2014 by the contributors listed in CREDITS.TXT
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
29
cccl_upstream/libcudacxx/test/NOTES.TXT
Normal file
29
cccl_upstream/libcudacxx/test/NOTES.TXT
Normal file
@@ -0,0 +1,29 @@
|
||||
//===---------------------------------------------------------------------===//
|
||||
// Notes relating to various libc++ tasks
|
||||
//===---------------------------------------------------------------------===//
|
||||
|
||||
This file contains notes about various libc++ tasks and processes.
|
||||
|
||||
//===---------------------------------------------------------------------===//
|
||||
// Post-Release TODO
|
||||
//===---------------------------------------------------------------------===//
|
||||
|
||||
These notes contain a list of things that must be done after branching for
|
||||
an LLVM release.
|
||||
|
||||
1. Update _LIBCUDACXX_VERSION in `__config`
|
||||
2. Update the __cccl_version file.
|
||||
3. Update the version number in `docs/conf.py`
|
||||
4. Create ABI lists for the previous release under `lib/abi`
|
||||
|
||||
//===---------------------------------------------------------------------===//
|
||||
// Adding a new header TODO
|
||||
//===---------------------------------------------------------------------===//
|
||||
|
||||
These notes contain a list of things that must be done upon adding a new header
|
||||
to libc++.
|
||||
|
||||
1. Add a test under `test/libcxx` that the header defines `_LIBCUDACXX_VERSION`.
|
||||
2. Update `test/libcxx/double_include.sh.cpp` to include the new header.
|
||||
3. Create a submodule in `include/module.modulemap` for the new header.
|
||||
4. Update the include/CMakeLists.txt file to include the new header.
|
||||
76
cccl_upstream/libcudacxx/test/TODO.TXT
Normal file
76
cccl_upstream/libcudacxx/test/TODO.TXT
Normal file
@@ -0,0 +1,76 @@
|
||||
This is meant to be a general place to list things that should be done "someday"
|
||||
|
||||
CXX Runtime Library Tasks
|
||||
=========================
|
||||
* Fix that CMake always link to /usr/lib/libc++abi.dylib on OS X.
|
||||
* Look into mirroring libsupc++'s typeinfo vtable layout when libsupc++/libstdc++
|
||||
is used as the runtime library.
|
||||
* Investigate and document interoperability between libc++ and libstdc++ on
|
||||
linux. Do this for every supported c++ runtime library.
|
||||
|
||||
Atomic Related Tasks
|
||||
====================
|
||||
* future should use <atomic> for synchronization.
|
||||
|
||||
Test Suite Tasks
|
||||
================
|
||||
* Improve the quality and portability of the locale test data.
|
||||
* Convert failure tests to use Clang Verify.
|
||||
|
||||
Filesystem Tasks
|
||||
================
|
||||
* P0492r2 - Implement National body comments for Filesystem
|
||||
* INCOMPLETE - US 25: has_filename() is equivalent to just !empty()
|
||||
* INCOMPLETE - US 31: Everything is defined in terms of one implicit host system
|
||||
* INCOMPLETE - US 32: Meaning of 27.10.2.1 unclear
|
||||
* INCOMPLETE - US 33: Definition of canonical path problematic
|
||||
* INCOMPLETE - US 34: Are there attributes of a file that are not an aspect of the file system?
|
||||
* INCOMPLETE - US 35: What synchronization is required to avoid a file system race?
|
||||
* INCOMPLETE - US 36: Symbolic links themselves are attached to a directory via (hard) links
|
||||
* INCOMPLETE - US 37: The term "redundant current directory (dot) elements" is not defined
|
||||
* INCOMPLETE - US 38: Duplicates §17.3.16
|
||||
* INCOMPLETE - US 39: Remove note: Dot and dot-dot are not directories
|
||||
* INCOMPLETE - US 40: Not all directories have a parent.
|
||||
* INCOMPLETE - US 41: The term "parent directory" for a (non-directory) file is unusual
|
||||
* INCOMPLETE - US 42: Pathname resolution does not always resolve a symlink
|
||||
* INCOMPLETE - US 43: Concerns about encoded character types
|
||||
* INCOMPLETE - US 44: Definition of path in terms of a string requires leaky abstraction
|
||||
* INCOMPLETE - US 45: Generic format portability compromised by unspecified root-name
|
||||
* INCOMPLETE - US 46: filename can be empty so productions for relative-path are redundant
|
||||
* INCOMPLETE - US 47: "." and ".." already match the name production
|
||||
* INCOMPLETE - US 48: Multiple separators are often meaningful in a root-name
|
||||
* INCOMPLETE - US 49: What does "method of conversion method" mean?
|
||||
* INCOMPLETE - US 50: 27.10.8.1 ¶ 1.4 largely redundant with ¶ 1.3
|
||||
* INCOMPLETE - US 51: Failing to add / when appending empty string prevents useful apps
|
||||
* INCOMPLETE - US 52: remove_filename() postcondition is not by itself a definition
|
||||
* INCOMPLETE - US 53: remove_filename()'s name does not correspond to its behavior
|
||||
* INCOMPLETE - US 54: remove_filename() is broken
|
||||
* INCOMPLETE - US 55: replace_extension()'s use of path as parameter is inappropriate
|
||||
* INCOMPLETE - US 56: Remove replace_extension()'s conditional addition of period
|
||||
* INCOMPLETE - US 57: On Windows, absolute paths will sort in among relative paths
|
||||
* INCOMPLETE - US 58: parent_path() behavior for root paths is useless
|
||||
* INCOMPLETE - US 59: filename() returning path for single path components is bizarre
|
||||
* INCOMPLETE - US 60: path("/foo/").filename()==path(".") is surprising
|
||||
* INCOMPLETE - US 61: Leading dots in filename() should not begin an extension
|
||||
* INCOMPLETE - US 62: It is important that stem()+extension()==filename()
|
||||
* INCOMPLETE - US 63: lexically_normal() inconsistently treats trailing "/" but not "/.." as directory
|
||||
* INCOMPLETE - US 73, CA 2: root-name is effectively implementation defined
|
||||
* INCOMPLETE - US 74, CA 3: The term "pathname" is ambiguous in some contexts
|
||||
* INCOMPLETE - US 75, CA 4: Extra flag in path constructors is needed
|
||||
* INCOMPLETE - US 76, CA 5: root-name definition is over-specified.
|
||||
* INCOMPLETE - US 77, CA 6: operator/ and other appends not useful if arg has root-name
|
||||
* INCOMPLETE - US 78, CA 7: Member absolute() in 27.10.4.1 is overspecified for non-POSIX-like O/S
|
||||
* INCOMPLETE - US 79, CA 8: Some operation functions are overspecified for implementation-defined file types
|
||||
* INCOMPLETE - US 185: Fold error_code and non-error_code signatures into one signature
|
||||
* INCOMPLETE - FI 14: directory_entry comparisons are members
|
||||
* INCOMPLETE - Late 36: permissions() error_code overload should be noexcept
|
||||
* INCOMPLETE - Late 37: permissions() actions should be separate parameter
|
||||
* INCOMPLETE - Late 42: resize_file() Postcondition missing argument
|
||||
|
||||
Misc Tasks
|
||||
==========
|
||||
* Find all sequences of >2 underscores and eradicate them.
|
||||
* run clang-tidy on libc++
|
||||
* Document the "conditionally-supported" bits of libc++
|
||||
* Look at basic_string's move assignment operator, re LWG 2063 and POCMA
|
||||
* Put a static_assert in std::allocator to deny const/volatile types (LWG 2447)
|
||||
67
cccl_upstream/libcudacxx/test/atomic_codegen/CMakeLists.txt
Normal file
67
cccl_upstream/libcudacxx/test/atomic_codegen/CMakeLists.txt
Normal file
@@ -0,0 +1,67 @@
|
||||
add_custom_target(libcudacxx.test.atomics.ptx)
|
||||
|
||||
find_program(filecheck "FileCheck")
|
||||
|
||||
if (filecheck)
|
||||
message("-- ${filecheck} found... building atomic codegen tests")
|
||||
else()
|
||||
return()
|
||||
endif()
|
||||
|
||||
find_program(cuobjdump "cuobjdump" REQUIRED)
|
||||
find_program(bash "bash" REQUIRED)
|
||||
|
||||
set(atomic_codegen_cuda_arch 80)
|
||||
|
||||
set(libcudacxx_atomic_codegen_tests)
|
||||
if (NOT "NVHPC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
|
||||
file(GLOB libcudacxx_atomic_codegen_tests "*.cu")
|
||||
endif()
|
||||
|
||||
# For every atomic API compile the TU and check if the SASS/PTX matches the expected result
|
||||
foreach (test_path IN LISTS libcudacxx_atomic_codegen_tests)
|
||||
cmake_path(GET test_path FILENAME test_file)
|
||||
cmake_path(REMOVE_EXTENSION test_file LAST_ONLY OUTPUT_VARIABLE test_name)
|
||||
|
||||
add_library(atomic_codegen_${test_name} STATIC "${test_path}")
|
||||
|
||||
set_target_properties(
|
||||
atomic_codegen_${test_name}
|
||||
PROPERTIES
|
||||
CUDA_ARCHITECTURES "${atomic_codegen_cuda_arch}"
|
||||
COMPILE_DEFINITIONS "_CCCL_ATOMIC_UNSAFE_AUTOMATIC_STORAGE=1"
|
||||
)
|
||||
|
||||
# Clang stopped emitting PTX in clang20. Add flags to re-enable it.
|
||||
if (
|
||||
CMAKE_CUDA_COMPILER_ID STREQUAL Clang
|
||||
AND CMAKE_CUDA_COMPILER_VERSION VERSION_GREATER_EQUAL 20
|
||||
)
|
||||
target_compile_options(
|
||||
atomic_codegen_${test_name}
|
||||
PRIVATE "--cuda-include-ptx=sm_${atomic_codegen_cuda_arch}"
|
||||
)
|
||||
endif()
|
||||
|
||||
target_compile_options(atomic_codegen_${test_name} PRIVATE "-Wno-comment")
|
||||
|
||||
## Important for testing the local headers
|
||||
target_include_directories(
|
||||
atomic_codegen_${test_name}
|
||||
PRIVATE "${libcudacxx_SOURCE_DIR}/include"
|
||||
)
|
||||
add_dependencies(libcudacxx.test.atomics.ptx atomic_codegen_${test_name})
|
||||
|
||||
# Add output path to object directory
|
||||
add_custom_command(
|
||||
TARGET libcudacxx.test.atomics.ptx
|
||||
POST_BUILD
|
||||
# gersemi: off
|
||||
COMMAND
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/dump_and_check.bash"
|
||||
$<TARGET_FILE:atomic_codegen_${test_name}>
|
||||
"${test_path}"
|
||||
SM8X
|
||||
# gersemi: on
|
||||
)
|
||||
endforeach()
|
||||
@@ -0,0 +1,21 @@
|
||||
#include <cuda/atomic>
|
||||
|
||||
__global__ void add_relaxed_device_non_volatile(int* data, int* out, int n)
|
||||
{
|
||||
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
|
||||
*out = ref.fetch_add(n, cuda::std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
/*
|
||||
|
||||
; SM8X-LABEL: .target sm_80
|
||||
; SM8X: .visible .entry [[FUNCTION:_.*add_relaxed_device_non_volatile.*]](
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#RESULT:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_2{{.*}}
|
||||
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#RESULT]];
|
||||
; SM8X-NEXT: {{/*[[:space:]] *}}atom.add.relaxed.gpu.s32 %r[[#DEST:]],[%rd[[#ATOM]]],%r[[#INPUT]];{{[[:space:]]/*}}
|
||||
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
|
||||
; SM8X-NEXT: ret;
|
||||
|
||||
*/
|
||||
@@ -0,0 +1,23 @@
|
||||
#include <cuda/atomic>
|
||||
|
||||
__global__ void cas_device_relaxed_non_volatile(int* data, int* out, int n)
|
||||
{
|
||||
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
|
||||
ref.compare_exchange_strong(*out, n, cuda::std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
// clang-format off
|
||||
/*
|
||||
|
||||
; SM8X-LABEL: .target sm_80
|
||||
; SM8X: .visible .entry [[FUNCTION:_.*cas_device_relaxed_non_volatile.*]](
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#EXPECTED:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_2{{.*}}
|
||||
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#EXPECTED]];
|
||||
; SM8X-DAG: ld.global.{{b|u}}32 %r[[#LOCALEXP:]], [%rd[[#INPUT]]];
|
||||
; SM8X-NEXT: {{/*[[:space:]] *}}atom.cas.relaxed.gpu.b32 %r[[#DEST:]],[%rd[[#ATOM]]],%r[[#LOCALEXP]],%r[[#INPUT]];{{[[:space:]]/*}}
|
||||
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
|
||||
; SM8X-NEXT: ret;
|
||||
|
||||
*/
|
||||
@@ -0,0 +1,21 @@
|
||||
#include <cuda/atomic>
|
||||
|
||||
__global__ void exch_device_relaxed_non_volatile(int* data, int* out, int n)
|
||||
{
|
||||
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
|
||||
*out = ref.exchange(n, cuda::std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
/*
|
||||
|
||||
; SM8X-LABEL: .target sm_80
|
||||
; SM8X: .visible .entry [[FUNCTION:_.*exch_device_relaxed_non_volatile.*]](
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#EXPECTED:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_2{{.*}}
|
||||
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#EXPECTED]];
|
||||
; SM8X-NEXT: {{/*[[:space:]] *}}atom.exch.relaxed.gpu.b32 %r[[#DEST:]],[%rd[[#ATOM]]],%r[[#INPUT]];{{[[:space:]]/*}}
|
||||
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
|
||||
; SM8X-NEXT: ret;
|
||||
|
||||
*/
|
||||
@@ -0,0 +1,20 @@
|
||||
#include <cuda/atomic>
|
||||
|
||||
__global__ void load_relaxed_device_non_volatile(int* data, int* out)
|
||||
{
|
||||
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
|
||||
*out = ref.load(cuda::std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
/*
|
||||
|
||||
; SM8X-LABEL: .target sm_80
|
||||
; SM8X: .visible .entry [[FUNCTION:_.*load_relaxed_device_non_volatile.*]](
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#EXPECTED:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
|
||||
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#EXPECTED]];
|
||||
; SM8X-NEXT: {{/*[[:space:]] *}}ld.relaxed.gpu.b32 %r[[#DEST:]],[%rd[[#ATOM]]];{{[[:space:]]/*}}
|
||||
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
|
||||
; SM8X-NEXT: ret;
|
||||
|
||||
*/
|
||||
@@ -0,0 +1,18 @@
|
||||
#include <cuda/atomic>
|
||||
|
||||
__global__ void store_relaxed_device_non_volatile(int* data, int in)
|
||||
{
|
||||
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
|
||||
ref.store(in, cuda::std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
/*
|
||||
|
||||
; SM8X-LABEL: .target sm_80
|
||||
; SM8X: .visible .entry [[FUNCTION:_.*store_relaxed_device_non_volatile.*]](
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
|
||||
; SM8X-NEXT: {{/*[[:space:]] *}}st.relaxed.gpu.b32 [%rd[[#ATOM]]],%r[[#INPUT]];{{[[:space:]]/*}}
|
||||
; SM8X-NEXT: ret;
|
||||
|
||||
*/
|
||||
@@ -0,0 +1,22 @@
|
||||
#include <cuda/atomic>
|
||||
|
||||
__global__ void sub_relaxed_device_non_volatile(int* data, int* out, int n)
|
||||
{
|
||||
auto ref = cuda::atomic_ref<int, cuda::thread_scope_device>{*(data)};
|
||||
*out = ref.fetch_sub(n, cuda::std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
/*
|
||||
|
||||
; SM8X-LABEL: .target sm_80
|
||||
; SM8X: .visible .entry [[FUNCTION:_.*sub_relaxed_device_non_volatile.*]](
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#ATOM:]], {{.*}}[[FUNCTION]]_param_0{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}64 %rd[[#RESULT:]], {{.*}}[[FUNCTION]]_param_1{{.*}}
|
||||
; SM8X-DAG: ld.param.{{b|u}}32 %r[[#INPUT:]], {{.*}}[[FUNCTION]]_param_2{{.*}}
|
||||
; SM8X-DAG: cvta.to.global.u64 %rd[[#GOUT:]], %rd[[#RESULT]];
|
||||
; SM8X-NEXT: neg.s32 %r[[#NEG:]], %r[[#INPUT]];
|
||||
; SM8X-NEXT: {{/*[[:space:]] *}}atom.add.relaxed.gpu.s32 %r[[#DEST:]],[%rd[[#ATOM]]],%r[[#NEG]];{{[[:space:]]/*}}
|
||||
; SM8X-NEXT: st.global.{{b|u}}32 [%rd[[#GOUT]]], %r[[#DEST]];
|
||||
; SM8X-NEXT: ret;
|
||||
|
||||
*/
|
||||
11
cccl_upstream/libcudacxx/test/atomic_codegen/dump_and_check.bash
Executable file
11
cccl_upstream/libcudacxx/test/atomic_codegen/dump_and_check.bash
Executable file
@@ -0,0 +1,11 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
## Usage: dump_and_check test.a test.cu PREFIXES [cuobjdump-mode]
|
||||
input_archive="${1}"
|
||||
input_testfile="${2}"
|
||||
input_prefix="${3}"
|
||||
dump_mode="${4:---dump-ptx}"
|
||||
filecheck="${FILECHECK:-FileCheck}"
|
||||
|
||||
cuobjdump "${dump_mode}" "${input_archive}" | "${filecheck}" --match-full-lines --check-prefixes="${input_prefix}" "${input_testfile}"
|
||||
10
cccl_upstream/libcudacxx/test/cmake/CMakeLists.txt
Normal file
10
cccl_upstream/libcudacxx/test/cmake/CMakeLists.txt
Normal file
@@ -0,0 +1,10 @@
|
||||
# Check source code for issues that can be found by pattern matching:
|
||||
add_test(
|
||||
NAME libcudacxx.test.cmake.check_source_files
|
||||
# gersemi: off
|
||||
COMMAND
|
||||
"${CMAKE_COMMAND}"
|
||||
-D "LIBCUDACXX_SOURCE_DIR=${libcudacxx_SOURCE_DIR}"
|
||||
-P "${CMAKE_CURRENT_LIST_DIR}/check_source_files.cmake"
|
||||
# gersemi: on
|
||||
)
|
||||
108
cccl_upstream/libcudacxx/test/cmake/check_source_files.cmake
Normal file
108
cccl_upstream/libcudacxx/test/cmake/check_source_files.cmake
Normal file
@@ -0,0 +1,108 @@
|
||||
# Check libcudacxx source files for issues that can be detected using pattern
|
||||
# matching.
|
||||
#
|
||||
# This is run as a ctest test named `libcudacxx.test.cmake.check_source_files`,
|
||||
# or manually with:
|
||||
# cmake -D "LIBCUDACXX_SOURCE_DIR=<libcudacxx project root>" -P check_source_files.cmake
|
||||
|
||||
cmake_minimum_required(VERSION 3.15)
|
||||
|
||||
function(count_substrings input search_regex output_var)
|
||||
string(REGEX MATCHALL "${search_regex}" matches "${input}")
|
||||
list(LENGTH matches num_matches)
|
||||
set(${output_var} ${num_matches} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
set(found_errors 0)
|
||||
file(
|
||||
GLOB_RECURSE libcudacxx_srcs
|
||||
RELATIVE "${LIBCUDACXX_SOURCE_DIR}"
|
||||
"${LIBCUDACXX_SOURCE_DIR}/include/cuda/*"
|
||||
"${LIBCUDACXX_SOURCE_DIR}/include/nv/*"
|
||||
)
|
||||
|
||||
# Exclude the imported libc++ headers from the scan. They are not under CCCL's
|
||||
# direct control and intentionally mirror the upstream libc++ implementation.
|
||||
list(FILTER libcudacxx_srcs EXCLUDE REGEX "^include/cuda/std/detail/libcxx/")
|
||||
|
||||
################################################################################
|
||||
# stdpar header checks.
|
||||
# Check all files in libcudacxx to make sure that they aren't including
|
||||
# <algorithm>, <memory>, or <numeric>, all of which can introduce circular
|
||||
# dependencies with compilers that integrate CCCL components deeply into
|
||||
# their C++ standard library implementations.
|
||||
#
|
||||
# The following headers should be used instead:
|
||||
# <algorithm> -> <cuda/std/__host_stdlib/algorithm>
|
||||
# <memory> -> <cuda/std/__host_stdlib/memory>
|
||||
# <numeric> -> <cuda/std/__host_stdlib/numeric>
|
||||
#
|
||||
set(
|
||||
stdpar_header_exclusions
|
||||
include/cuda/std/__host_stdlib/algorithm
|
||||
include/cuda/std/__host_stdlib/memory
|
||||
include/cuda/std/__host_stdlib/numeric
|
||||
)
|
||||
|
||||
set(algorithm_regex "#[ \t]*include[ \t]+<algorithm>")
|
||||
set(memory_regex "#[ \t]*include[ \t]+<memory>")
|
||||
set(numeric_regex "#[ \t]*include[ \t]+<numeric>")
|
||||
|
||||
# Validation check for the above regex pattern:
|
||||
count_substrings([=[
|
||||
#include <algorithm>
|
||||
# include <algorithm>
|
||||
#include <algorithm>
|
||||
# include <algorithm>
|
||||
# include <algorithm> // ...
|
||||
]=]
|
||||
${algorithm_regex} valid_count
|
||||
)
|
||||
if (NOT valid_count EQUAL 5)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Validation of stdpar header regex failed: "
|
||||
"Matched ${valid_count} times, expected 5."
|
||||
)
|
||||
endif()
|
||||
|
||||
################################################################################
|
||||
# Read source files:
|
||||
foreach (src ${libcudacxx_srcs})
|
||||
if (IS_DIRECTORY "${LIBCUDACXX_SOURCE_DIR}/${src}")
|
||||
continue()
|
||||
endif()
|
||||
|
||||
file(READ "${LIBCUDACXX_SOURCE_DIR}/${src}" src_contents)
|
||||
|
||||
if (NOT ${src} IN_LIST stdpar_header_exclusions)
|
||||
count_substrings("${src_contents}" "${algorithm_regex}" algorithm_count)
|
||||
count_substrings("${src_contents}" "${memory_regex}" memory_count)
|
||||
count_substrings("${src_contents}" "${numeric_regex}" numeric_count)
|
||||
|
||||
if (NOT algorithm_count EQUAL 0)
|
||||
message(
|
||||
"'${src}' includes the <algorithm> header. Replace with <cuda/std/__host_stdlib/algorithm>."
|
||||
)
|
||||
set(found_errors 1)
|
||||
endif()
|
||||
|
||||
if (NOT memory_count EQUAL 0)
|
||||
message(
|
||||
"'${src}' includes the <memory> header. Replace with <cuda/std/__host_stdlib/memory>."
|
||||
)
|
||||
set(found_errors 1)
|
||||
endif()
|
||||
|
||||
if (NOT numeric_count EQUAL 0)
|
||||
message(
|
||||
"'${src}' includes the <numeric> header. Replace with <cuda/std/__host_stdlib/numeric>."
|
||||
)
|
||||
set(found_errors 1)
|
||||
endif()
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if (NOT found_errors EQUAL 0)
|
||||
message(FATAL_ERROR "Errors detected.")
|
||||
endif()
|
||||
202
cccl_upstream/libcudacxx/test/debugging/CMakeLists.txt
Normal file
202
cccl_upstream/libcudacxx/test/debugging/CMakeLists.txt
Normal file
@@ -0,0 +1,202 @@
|
||||
set(LIBCUDACXX_SUPPORTED_DEBUGGERS lldb gdb)
|
||||
set(LIBCUDACXX_DEBUGGING_BASE_TARGET libcudacxx.test.debugging)
|
||||
|
||||
function(libcudacxx_init_debugger_testing enabled_var)
|
||||
# Unconditionally create the umbrella target to make CMakePresets easier to setup. If we
|
||||
# don't have the debuggers available, this target doesn't do anything
|
||||
add_custom_target(${LIBCUDACXX_DEBUGGING_BASE_TARGET})
|
||||
|
||||
# Windows lldb is broken sometimes (see
|
||||
# https://github.com/llvm/llvm-project/issues/74073) so merely finding it does not mean
|
||||
# it is functional. In any case, it is good to validate the binary since even a found
|
||||
# lldb/gdb on other platforms that ends up not working is annoying to work around
|
||||
function(validator result_var item)
|
||||
execute_process(
|
||||
COMMAND ${item} --version
|
||||
OUTPUT_QUIET
|
||||
ERROR_QUIET
|
||||
RESULT_VARIABLE result
|
||||
TIMEOUT 10
|
||||
)
|
||||
|
||||
if (result EQUAL 0)
|
||||
set(${result_var} TRUE PARENT_SCOPE)
|
||||
else()
|
||||
set(${result_var} FALSE PARENT_SCOPE)
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
set(enabled FALSE)
|
||||
foreach (debugger IN LISTS LIBCUDACXX_SUPPORTED_DEBUGGERS)
|
||||
string(TOUPPER "${debugger}" DEBUGGER_UPPER)
|
||||
|
||||
find_program(
|
||||
LIBCUDACXX_${DEBUGGER_UPPER}
|
||||
NAMES "${debugger}"
|
||||
VALIDATOR validator
|
||||
)
|
||||
|
||||
if (LIBCUDACXX_${DEBUGGER_UPPER})
|
||||
set(debugger_exe "${LIBCUDACXX_${DEBUGGER_UPPER}}")
|
||||
message(STATUS "Found ${debugger}: ${debugger_exe}")
|
||||
execute_process(
|
||||
COMMAND ${debugger_exe} --version
|
||||
OUTPUT_VARIABLE version
|
||||
ERROR_VARIABLE version
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
ERROR_STRIP_TRAILING_WHITESPACE
|
||||
COMMAND_ERROR_IS_FATAL ANY
|
||||
)
|
||||
message(STATUS "${debugger_exe} --version: ${version}")
|
||||
endif()
|
||||
|
||||
set(default OFF)
|
||||
if (LIBCUDACXX_${DEBUGGER_UPPER})
|
||||
set(default ON)
|
||||
endif()
|
||||
|
||||
option(
|
||||
LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING
|
||||
"Run libcudacxx pretty-printer tests with ${debugger}"
|
||||
${default}
|
||||
)
|
||||
|
||||
if (
|
||||
LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING
|
||||
AND NOT LIBCUDACXX_${DEBUGGER_UPPER}
|
||||
)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING is enabled, but ${debugger} was not found"
|
||||
)
|
||||
endif()
|
||||
|
||||
set(
|
||||
LIBCUDACXX_${DEBUGGER_UPPER}
|
||||
"${LIBCUDACXX_${DEBUGGER_UPPER}}"
|
||||
PARENT_SCOPE
|
||||
)
|
||||
set(
|
||||
LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING
|
||||
"${LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING}"
|
||||
PARENT_SCOPE
|
||||
)
|
||||
|
||||
if (LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING)
|
||||
set(enabled TRUE)
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
set(${enabled_var} ${enabled} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
libcudacxx_init_debugger_testing(testing_enabled)
|
||||
|
||||
if (NOT testing_enabled)
|
||||
return()
|
||||
endif()
|
||||
|
||||
#[=======================================================================[.rst:
|
||||
libcudacxx_add_pretty_printer_test
|
||||
---------------------------------
|
||||
|
||||
Build a CUDA pretty-printer scenario and register its enabled debugger tests.
|
||||
|
||||
The function creates the executable target
|
||||
``libcudacxx.test.debugging.<NAME>`` with host debug information and optimization
|
||||
disabled. For each enabled debugger, it registers a serial CTest named
|
||||
``libcudacxx.test.debugging.<debugger>.<NAME>``. The test uses the debugger's
|
||||
formatter entry point and the ``<debugger>.expected`` file in the caller's source
|
||||
directory.
|
||||
|
||||
Arguments
|
||||
^^^^^^^^^
|
||||
|
||||
``NAME``
|
||||
Scenario name used in the executable target, CTest names, and diagnostic output.
|
||||
|
||||
``SOURCES``
|
||||
Source files used to build the CUDA scenario executable. Relative paths are resolved
|
||||
against the caller's source directory.
|
||||
|
||||
``CASES``
|
||||
Ordered runner arguments describing the debugger stops and expressions. Each case
|
||||
has the form ``--case <breakpoint> <caller-frame-index> <section-name>
|
||||
<expression>``.
|
||||
|
||||
#]=======================================================================]
|
||||
function(libcudacxx_add_pretty_printer_test)
|
||||
set(options)
|
||||
set(one_value_arguments NAME)
|
||||
set(multi_value_arguments SOURCES CASES)
|
||||
cmake_parse_arguments(
|
||||
pretty_printer
|
||||
"${options}"
|
||||
"${one_value_arguments}"
|
||||
"${multi_value_arguments}"
|
||||
${ARGN}
|
||||
)
|
||||
|
||||
if (pretty_printer_UNPARSED_ARGUMENTS)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Unrecognized arguments: ${pretty_printer_UNPARSED_ARGUMENTS}"
|
||||
)
|
||||
endif()
|
||||
if (NOT pretty_printer_NAME)
|
||||
message(FATAL_ERROR "libcudacxx_add_pretty_printer_test requires NAME")
|
||||
endif()
|
||||
if (NOT pretty_printer_SOURCES)
|
||||
message(FATAL_ERROR "libcudacxx_add_pretty_printer_test requires SOURCES")
|
||||
endif()
|
||||
if (NOT pretty_printer_CASES)
|
||||
message(FATAL_ERROR "libcudacxx_add_pretty_printer_test requires CASES")
|
||||
endif()
|
||||
|
||||
set(target_name "${LIBCUDACXX_DEBUGGING_BASE_TARGET}.${pretty_printer_NAME}")
|
||||
cccl_add_executable(
|
||||
${target_name}
|
||||
DIALECT 17
|
||||
NO_CLANG_TIDY
|
||||
SOURCES ${pretty_printer_SOURCES}
|
||||
)
|
||||
target_compile_options(
|
||||
${target_name}
|
||||
PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:-g> $<$<COMPILE_LANGUAGE:CUDA>:-O0>
|
||||
)
|
||||
target_link_libraries(${target_name} PRIVATE libcudacxx.compiler_interface)
|
||||
|
||||
foreach (debugger IN LISTS LIBCUDACXX_SUPPORTED_DEBUGGERS)
|
||||
string(TOUPPER "${debugger}" DEBUGGER_UPPER)
|
||||
|
||||
if (NOT LIBCUDACXX_ENABLE_${DEBUGGER_UPPER}_TESTING)
|
||||
continue()
|
||||
endif()
|
||||
|
||||
set(executable "${LIBCUDACXX_${DEBUGGER_UPPER}}")
|
||||
set(
|
||||
formatter
|
||||
"${libcudacxx_SOURCE_DIR}/share/libcudacxx/${debugger}/__init__.py"
|
||||
)
|
||||
set(test_name "${target_name}.${debugger}")
|
||||
add_test(
|
||||
NAME ${test_name}
|
||||
COMMAND
|
||||
# gersemi: off
|
||||
"${Python_EXECUTABLE}" "${CMAKE_CURRENT_FUNCTION_LIST_DIR}/run_pretty_printer_test.py"
|
||||
--debugger "${debugger}"
|
||||
--debugger-executable "${executable}"
|
||||
--program "$<TARGET_FILE:${target_name}>"
|
||||
--formatter-init "${formatter}"
|
||||
--expected "${CMAKE_CURRENT_SOURCE_DIR}/${debugger}.expected"
|
||||
--output-log "${CMAKE_CURRENT_BINARY_DIR}/${debugger}.log"
|
||||
${pretty_printer_CASES}
|
||||
# gersemi: on
|
||||
)
|
||||
set_tests_properties(${test_name} PROPERTIES TIMEOUT 60)
|
||||
endforeach()
|
||||
endfunction()
|
||||
|
||||
add_subdirectory(array)
|
||||
add_subdirectory(buffer)
|
||||
add_subdirectory(memory_resource)
|
||||
13
cccl_upstream/libcudacxx/test/debugging/array/CMakeLists.txt
Normal file
13
cccl_upstream/libcudacxx/test/debugging/array/CMakeLists.txt
Normal file
@@ -0,0 +1,13 @@
|
||||
libcudacxx_add_pretty_printer_test(
|
||||
NAME array
|
||||
SOURCES source.cu
|
||||
CASES
|
||||
# gersemi: off
|
||||
--case inspect_normal 1 array.normal normal
|
||||
--case inspect_empty 1 array.empty empty
|
||||
--case inspect_nested 1 array.nested nested
|
||||
--case inspect_alias 1 array.alias alias
|
||||
--case inspect_before_update 1 array.update.before updated_values
|
||||
--case inspect_after_update 1 array.update.after updated_values
|
||||
# gersemi: on
|
||||
)
|
||||
44
cccl_upstream/libcudacxx/test/debugging/array/gdb.expected
Normal file
44
cccl_upstream/libcudacxx/test/debugging/array/gdb.expected
Normal file
@@ -0,0 +1,44 @@
|
||||
=============== array.normal begin ===============
|
||||
cuda::std::array<int, 3> = {
|
||||
[0] = -7,
|
||||
[1] = 0,
|
||||
[2] = 42
|
||||
}
|
||||
=============== array.normal end ===============
|
||||
=============== array.empty begin ===============
|
||||
cuda::std::array<int, 0>
|
||||
=============== array.empty end ===============
|
||||
=============== array.nested begin ===============
|
||||
cuda::std::array<cuda::std::array<int, 2>, 2> = {
|
||||
[0] = cuda::std::array<int, 2> = {
|
||||
[0] = 13,
|
||||
[1] = -5
|
||||
},
|
||||
[1] = cuda::std::array<int, 2> = {
|
||||
[0] = 0,
|
||||
[1] = 88
|
||||
}
|
||||
}
|
||||
=============== array.nested end ===============
|
||||
=============== array.alias begin ===============
|
||||
cuda::std::array<int, 4> = {
|
||||
[0] = -31,
|
||||
[1] = 17,
|
||||
[2] = 8,
|
||||
[3] = -64
|
||||
}
|
||||
=============== array.alias end ===============
|
||||
=============== array.update.before begin ===============
|
||||
cuda::std::array<int, 3> = {
|
||||
[0] = 6,
|
||||
[1] = -91,
|
||||
[2] = 52
|
||||
}
|
||||
=============== array.update.before end ===============
|
||||
=============== array.update.after begin ===============
|
||||
cuda::std::array<int, 3> = {
|
||||
[0] = 3,
|
||||
[1] = 85,
|
||||
[2] = -12
|
||||
}
|
||||
=============== array.update.after end ===============
|
||||
21
cccl_upstream/libcudacxx/test/debugging/array/lldb.expected
Normal file
21
cccl_upstream/libcudacxx/test/debugging/array/lldb.expected
Normal file
@@ -0,0 +1,21 @@
|
||||
=============== array.normal begin ===============
|
||||
(cuda::std::array<int, 3>) ([0] = -7, [1] = 0, [2] = 42)
|
||||
=============== array.normal end ===============
|
||||
=============== array.empty begin ===============
|
||||
(cuda::std::array<int, 0>)
|
||||
=============== array.empty end ===============
|
||||
=============== array.nested begin ===============
|
||||
(cuda::std::array<cuda::std::array<int, 2>, 2>) {
|
||||
[0] = ([0] = 13, [1] = -5)
|
||||
[1] = ([0] = 0, [1] = 88)
|
||||
}
|
||||
=============== array.nested end ===============
|
||||
=============== array.alias begin ===============
|
||||
(cuda::std::array<int, 4>) ([0] = -31, [1] = 17, [2] = 8, [3] = -64)
|
||||
=============== array.alias end ===============
|
||||
=============== array.update.before begin ===============
|
||||
(cuda::std::array<int, 3>) ([0] = 6, [1] = -91, [2] = 52)
|
||||
=============== array.update.before end ===============
|
||||
=============== array.update.after begin ===============
|
||||
(cuda::std::array<int, 3>) ([0] = 3, [1] = 85, [2] = -12)
|
||||
=============== array.update.after end ===============
|
||||
60
cccl_upstream/libcudacxx/test/debugging/array/source.cu
Normal file
60
cccl_upstream/libcudacxx/test/debugging/array/source.cu
Normal file
@@ -0,0 +1,60 @@
|
||||
// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <cuda/std/array>
|
||||
|
||||
template <class T>
|
||||
[[gnu::noinline]] void keep_for_debugger(const T& value)
|
||||
{
|
||||
asm volatile("" : : "g"(&value) : "memory");
|
||||
}
|
||||
|
||||
using array_alias = cuda::std::array<int, 4>;
|
||||
|
||||
[[gnu::noinline]] void inspect_normal(const cuda::std::array<int, 3>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_empty(const cuda::std::array<int, 0>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_nested(const cuda::std::array<cuda::std::array<int, 2>, 2>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_alias(const array_alias& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_before_update(const cuda::std::array<int, 3>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_after_update(const cuda::std::array<int, 3>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
int main()
|
||||
{
|
||||
const cuda::std::array<int, 3> normal = {-7, 0, 42};
|
||||
const cuda::std::array<int, 0> empty = {};
|
||||
const cuda::std::array<cuda::std::array<int, 2>, 2> nested = {{{13, -5}, {0, 88}}};
|
||||
const array_alias alias = {-31, 17, 8, -64};
|
||||
cuda::std::array<int, 3> updated_values = {6, -91, 52};
|
||||
|
||||
inspect_normal(normal);
|
||||
inspect_empty(empty);
|
||||
inspect_nested(nested);
|
||||
inspect_alias(alias);
|
||||
inspect_before_update(updated_values);
|
||||
updated_values = {3, 85, -12};
|
||||
inspect_after_update(updated_values);
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
libcudacxx_add_pretty_printer_test(
|
||||
NAME buffer
|
||||
SOURCES source.cu
|
||||
CASES
|
||||
# gersemi: off
|
||||
--case inspect_normal 1 buffer.normal normal_values
|
||||
--case inspect_alias 1 buffer.alias aliased_values
|
||||
--case inspect_vector 1 buffer.vector.0 "buffer_vector[0]"
|
||||
--case inspect_vector 1 buffer.vector.1 "buffer_vector[1]"
|
||||
--case inspect_host_device 1 buffer.host_device host_device_values
|
||||
--case inspect_empty 1 buffer.empty empty_values
|
||||
--case inspect_before_update 1 buffer.update.before updated_values
|
||||
--case inspect_after_update 1 buffer.update.after updated_values
|
||||
# gersemi: on
|
||||
)
|
||||
63
cccl_upstream/libcudacxx/test/debugging/buffer/gdb.expected
Normal file
63
cccl_upstream/libcudacxx/test/debugging/buffer/gdb.expected
Normal file
@@ -0,0 +1,63 @@
|
||||
=============== buffer.normal begin ===============
|
||||
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=10, align=4, data=<address> (device) = {
|
||||
[0] = -56,
|
||||
[1] = 22,
|
||||
[2] = 94,
|
||||
[3] = -13,
|
||||
[4] = 7,
|
||||
[5] = 41,
|
||||
[6] = -82,
|
||||
[7] = 0,
|
||||
[8] = 63,
|
||||
[9] = -5
|
||||
}
|
||||
=============== buffer.normal end ===============
|
||||
=============== buffer.alias begin ===============
|
||||
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) = {
|
||||
[0] = 17,
|
||||
[1] = -31,
|
||||
[2] = 8,
|
||||
[3] = 55
|
||||
}
|
||||
=============== buffer.alias end ===============
|
||||
=============== buffer.vector.0 begin ===============
|
||||
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=3, align=4, data=<address> (device) = {
|
||||
[0] = -2,
|
||||
[1] = 4,
|
||||
[2] = 6
|
||||
}
|
||||
=============== buffer.vector.0 end ===============
|
||||
=============== buffer.vector.1 begin ===============
|
||||
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=3, align=4, data=<address> (device) = {
|
||||
[0] = 11,
|
||||
[1] = -9,
|
||||
[2] = 27
|
||||
}
|
||||
=============== buffer.vector.1 end ===============
|
||||
=============== buffer.host_device begin ===============
|
||||
cuda::buffer<int, cuda::mr::device_accessible, cuda::mr::host_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (host/device) = {
|
||||
[0] = 3,
|
||||
[1] = 14,
|
||||
[2] = -15,
|
||||
[3] = 92
|
||||
}
|
||||
=============== buffer.host_device end ===============
|
||||
=============== buffer.empty begin ===============
|
||||
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=0, align=4, data=0x0 (device)
|
||||
=============== buffer.empty end ===============
|
||||
=============== buffer.update.before begin ===============
|
||||
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) = {
|
||||
[0] = 1,
|
||||
[1] = 2,
|
||||
[2] = 3,
|
||||
[3] = 4
|
||||
}
|
||||
=============== buffer.update.before end ===============
|
||||
=============== buffer.update.after begin ===============
|
||||
cuda::buffer<int, cuda::mr::device_accessible> mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) = {
|
||||
[0] = -8,
|
||||
[1] = 13,
|
||||
[2] = 21,
|
||||
[3] = -34
|
||||
}
|
||||
=============== buffer.update.after end ===============
|
||||
63
cccl_upstream/libcudacxx/test/debugging/buffer/lldb.expected
Normal file
63
cccl_upstream/libcudacxx/test/debugging/buffer/lldb.expected
Normal file
@@ -0,0 +1,63 @@
|
||||
=============== buffer.normal begin ===============
|
||||
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=10, align=4, data=<address> (device) {
|
||||
[0] = -56
|
||||
[1] = 22
|
||||
[2] = 94
|
||||
[3] = -13
|
||||
[4] = 7
|
||||
[5] = 41
|
||||
[6] = -82
|
||||
[7] = 0
|
||||
[8] = 63
|
||||
[9] = -5
|
||||
}
|
||||
=============== buffer.normal end ===============
|
||||
=============== buffer.alias begin ===============
|
||||
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) {
|
||||
[0] = 17
|
||||
[1] = -31
|
||||
[2] = 8
|
||||
[3] = 55
|
||||
}
|
||||
=============== buffer.alias end ===============
|
||||
=============== buffer.vector.0 begin ===============
|
||||
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=3, align=4, data=<address> (device) {
|
||||
[0] = -2
|
||||
[1] = 4
|
||||
[2] = 6
|
||||
}
|
||||
=============== buffer.vector.0 end ===============
|
||||
=============== buffer.vector.1 begin ===============
|
||||
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=3, align=4, data=<address> (device) {
|
||||
[0] = 11
|
||||
[1] = -9
|
||||
[2] = 27
|
||||
}
|
||||
=============== buffer.vector.1 end ===============
|
||||
=============== buffer.host_device begin ===============
|
||||
(cuda::buffer<int, cuda::mr::device_accessible, cuda::mr::host_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (host/device) {
|
||||
[0] = 3
|
||||
[1] = 14
|
||||
[2] = -15
|
||||
[3] = 92
|
||||
}
|
||||
=============== buffer.host_device end ===============
|
||||
=============== buffer.empty begin ===============
|
||||
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=0, align=4, data=0x0 (device)
|
||||
=============== buffer.empty end ===============
|
||||
=============== buffer.update.before begin ===============
|
||||
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) {
|
||||
[0] = 1
|
||||
[1] = 2
|
||||
[2] = 3
|
||||
[3] = 4
|
||||
}
|
||||
=============== buffer.update.before end ===============
|
||||
=============== buffer.update.after begin ===============
|
||||
(cuda::buffer<int, cuda::mr::device_accessible>) mr=cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>, stream=<address>, size=4, align=4, data=<address> (device) {
|
||||
[0] = -8
|
||||
[1] = 13
|
||||
[2] = 21
|
||||
[3] = -34
|
||||
}
|
||||
=============== buffer.update.after end ===============
|
||||
102
cccl_upstream/libcudacxx/test/debugging/buffer/source.cu
Normal file
102
cccl_upstream/libcudacxx/test/debugging/buffer/source.cu
Normal file
@@ -0,0 +1,102 @@
|
||||
// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <cuda/buffer>
|
||||
#include <cuda/memory_resource>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <vector>
|
||||
|
||||
#include <cuda_runtime_api.h>
|
||||
|
||||
template <class T>
|
||||
[[gnu::noinline]] void keep_for_debugger(const T& value)
|
||||
{
|
||||
asm volatile("" : : "g"(&value) : "memory");
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_normal(const cuda::device_buffer<int>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
using device_buffer_alias = cuda::buffer<int, cuda::mr::device_accessible>;
|
||||
|
||||
[[gnu::noinline]] void inspect_alias(const device_buffer_alias& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_vector(const std::vector<cuda::device_buffer<int>>& values)
|
||||
{
|
||||
keep_for_debugger(values[0]);
|
||||
keep_for_debugger(values[1]);
|
||||
}
|
||||
|
||||
template <class Buffer>
|
||||
[[gnu::noinline]] void inspect_host_device(const Buffer& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_empty(const cuda::device_buffer<int>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_before_update(const cuda::device_buffer<int>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_after_update(const cuda::device_buffer<int>& values)
|
||||
{
|
||||
keep_for_debugger(values);
|
||||
}
|
||||
|
||||
int main()
|
||||
{
|
||||
constexpr cuda::device_ref device{0};
|
||||
cuda::stream stream{device};
|
||||
|
||||
const cuda::std::array normal_host_values{-56, 22, 94, -13, 7, 41, -82, 0, 63, -5};
|
||||
const auto normal_values = cuda::make_device_buffer<int>(stream, device, normal_host_values);
|
||||
|
||||
const cuda::std::array alias_host_values{17, -31, 8, 55};
|
||||
const device_buffer_alias aliased_values = cuda::make_device_buffer<int>(stream, device, alias_host_values);
|
||||
|
||||
std::vector<cuda::device_buffer<int>> buffer_vector;
|
||||
buffer_vector.emplace_back(cuda::make_device_buffer<int>(stream, device, cuda::std::array{-2, 4, 6}));
|
||||
buffer_vector.emplace_back(cuda::make_device_buffer<int>(stream, device, cuda::std::array{11, -9, 27}));
|
||||
|
||||
cuda::mr::legacy_managed_memory_resource managed_resource;
|
||||
const cuda::std::array host_device_host_values{3, 14, -15, 92};
|
||||
const auto host_device_values = cuda::make_buffer<int>(stream, managed_resource, host_device_host_values);
|
||||
const auto empty_values = cuda::make_device_buffer<int>(stream, device);
|
||||
|
||||
const cuda::std::array initial_updated_host_values{1, 2, 3, 4};
|
||||
auto updated_values = cuda::make_device_buffer<int>(stream, device, initial_updated_host_values);
|
||||
|
||||
stream.sync();
|
||||
inspect_normal(normal_values);
|
||||
inspect_alias(aliased_values);
|
||||
inspect_vector(buffer_vector);
|
||||
inspect_host_device(host_device_values);
|
||||
inspect_empty(empty_values);
|
||||
inspect_before_update(updated_values);
|
||||
|
||||
const cuda::std::array replacement_host_values{-8, 13, 21, -34};
|
||||
if (cudaMemcpyAsync(updated_values.data(),
|
||||
replacement_host_values.data(),
|
||||
replacement_host_values.size() * sizeof(*updated_values.data()),
|
||||
cudaMemcpyDefault,
|
||||
stream.get())
|
||||
!= cudaSuccess)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
stream.sync();
|
||||
inspect_after_update(updated_values);
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
libcudacxx_add_pretty_printer_test(
|
||||
NAME memory_resource
|
||||
SOURCES source.cu
|
||||
CASES
|
||||
# gersemi: off
|
||||
--case inspect_device 1 memory_resource.device device_resource
|
||||
--case inspect_host_device 1 memory_resource.host_device host_device_resource
|
||||
--case inspect_alias 1 memory_resource.alias aliased_resource
|
||||
# gersemi: on
|
||||
)
|
||||
@@ -0,0 +1,9 @@
|
||||
=============== memory_resource.device begin ===============
|
||||
cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>
|
||||
=============== memory_resource.device end ===============
|
||||
=============== memory_resource.host_device begin ===============
|
||||
cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible> @ <address>
|
||||
=============== memory_resource.host_device end ===============
|
||||
=============== memory_resource.alias begin ===============
|
||||
cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>
|
||||
=============== memory_resource.alias end ===============
|
||||
@@ -0,0 +1,9 @@
|
||||
=============== memory_resource.device begin ===============
|
||||
(const device_resource_type) cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>
|
||||
=============== memory_resource.device end ===============
|
||||
=============== memory_resource.host_device begin ===============
|
||||
(const host_device_resource_type) cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible> @ <address>
|
||||
=============== memory_resource.host_device end ===============
|
||||
=============== memory_resource.alias begin ===============
|
||||
(const resource_alias) cuda::mr::any_resource<cuda::mr::device_accessible> @ <address>
|
||||
=============== memory_resource.alias end ===============
|
||||
@@ -0,0 +1,43 @@
|
||||
// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <cuda/memory_resource>
|
||||
|
||||
template <class T>
|
||||
[[gnu::noinline]] void keep_for_debugger(const T& value)
|
||||
{
|
||||
asm volatile("" : : "g"(&value) : "memory");
|
||||
}
|
||||
|
||||
using device_resource_type = cuda::mr::any_resource<cuda::mr::device_accessible>;
|
||||
using host_device_resource_type = cuda::mr::any_resource<cuda::mr::device_accessible, cuda::mr::host_accessible>;
|
||||
using resource_alias = device_resource_type;
|
||||
|
||||
[[gnu::noinline]] void inspect_device(const device_resource_type& resource)
|
||||
{
|
||||
keep_for_debugger(resource);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_host_device(const host_device_resource_type& resource)
|
||||
{
|
||||
keep_for_debugger(resource);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void inspect_alias(const resource_alias& resource)
|
||||
{
|
||||
keep_for_debugger(resource);
|
||||
}
|
||||
|
||||
int main()
|
||||
{
|
||||
using adapted_resource = cuda::mr::synchronous_resource_adapter<cuda::mr::legacy_managed_memory_resource>;
|
||||
const adapted_resource managed_resource{cuda::mr::legacy_managed_memory_resource{}};
|
||||
const device_resource_type device_resource{managed_resource};
|
||||
const host_device_resource_type host_device_resource{managed_resource};
|
||||
const resource_alias aliased_resource{managed_resource};
|
||||
|
||||
inspect_device(device_resource);
|
||||
inspect_host_device(host_device_resource);
|
||||
inspect_alias(aliased_resource);
|
||||
}
|
||||
@@ -0,0 +1,790 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""Run a libcudacxx pretty-printer scenario under LLDB or GDB."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import difflib
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from abc import ABC, abstractmethod
|
||||
from collections.abc import Sequence
|
||||
from dataclasses import dataclass
|
||||
from enum import StrEnum
|
||||
from pathlib import Path
|
||||
|
||||
_MARKER_EDGE = "=" * 15
|
||||
_MARKER_PATTERN = re.compile(
|
||||
rf"^{re.escape(_MARKER_EDGE)} (?P<section>.+) (?P<kind>begin|end) {re.escape(_MARKER_EDGE)}$"
|
||||
)
|
||||
_LLDB_ECHO_PATTERN = re.compile(r"^\s*\(lldb\)\s")
|
||||
_GDB_VALUE_PREFIX_PATTERN = re.compile(r"^\s*\$\d+ = ")
|
||||
_NONZERO_HEX_PATTERN = re.compile(r"\b0x(?!0+\b)[0-9a-fA-F]+\b")
|
||||
|
||||
|
||||
class HarnessError(RuntimeError):
|
||||
"""Report invalid test input or debugger output."""
|
||||
|
||||
|
||||
class DebuggerError(RuntimeError):
|
||||
"""Report a debugger launch, timeout, or exit failure."""
|
||||
|
||||
|
||||
class Debugger(StrEnum):
|
||||
LLDB = "lldb"
|
||||
GDB = "gdb"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Case:
|
||||
breakpoint: str
|
||||
frame: int
|
||||
section: str
|
||||
expression: str
|
||||
|
||||
|
||||
class CaseAction(argparse.Action):
|
||||
"""Parse and validate one four-part ``--case`` option."""
|
||||
|
||||
def __call__(
|
||||
self,
|
||||
parser: argparse.ArgumentParser,
|
||||
namespace: argparse.Namespace,
|
||||
values: Sequence[str],
|
||||
option_string: str | None = None,
|
||||
) -> None:
|
||||
"""Append one parsed case to the argument namespace.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
parser : argparse.ArgumentParser
|
||||
Parser handling the command line.
|
||||
namespace : argparse.Namespace
|
||||
Namespace receiving parsed cases.
|
||||
values : Sequence[str]
|
||||
Breakpoint, frame, section, and expression values.
|
||||
option_string : str or None
|
||||
Option spelling that supplied the values.
|
||||
|
||||
Raises
|
||||
------
|
||||
SystemExit
|
||||
If the frame, breakpoint, section, or expression is invalid, or if
|
||||
the section name is duplicated.
|
||||
"""
|
||||
breakpoint, raw_frame, section, expression = values
|
||||
try:
|
||||
frame = int(raw_frame)
|
||||
except ValueError:
|
||||
parser.error(f"invalid caller frame index {raw_frame!r}")
|
||||
if frame < 0:
|
||||
parser.error(f"caller frame index must be nonnegative: {frame}")
|
||||
for label, value in (
|
||||
("breakpoint", breakpoint),
|
||||
("section", section),
|
||||
("expression", expression),
|
||||
):
|
||||
if not value or "\n" in value or "\r" in value:
|
||||
parser.error(f"{label} must be a nonempty single line")
|
||||
|
||||
cases: list[Case] = getattr(namespace, self.dest) or []
|
||||
if any(case.section == section for case in cases):
|
||||
parser.error(f"duplicate section name: {section}")
|
||||
cases.append(Case(breakpoint, frame, section, expression))
|
||||
setattr(namespace, self.dest, cases)
|
||||
|
||||
|
||||
def marker(section: str, kind: str) -> str:
|
||||
"""Build an exact marker for a captured section.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
section : str
|
||||
Unique name of the output section.
|
||||
kind : str
|
||||
Marker kind, either ``begin`` or ``end``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Complete marker line expected in debugger output.
|
||||
"""
|
||||
return f"{_MARKER_EDGE} {section} {kind} {_MARKER_EDGE}"
|
||||
|
||||
|
||||
class DebuggerAdapter(ABC):
|
||||
"""Provide debugger-specific command and transcript hooks.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
executable : Path
|
||||
Debugger executable path.
|
||||
formatter_init : Path
|
||||
Pretty-printer entry-point path.
|
||||
program : Path
|
||||
Scenario executable path.
|
||||
"""
|
||||
|
||||
kind: Debugger
|
||||
|
||||
def __init__(self, executable: Path, formatter_init: Path, program: Path) -> None:
|
||||
"""Store paths shared by debugger-specific operations.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
executable : Path
|
||||
Debugger executable path.
|
||||
formatter_init : Path
|
||||
Pretty-printer entry-point path.
|
||||
program : Path
|
||||
Scenario executable path.
|
||||
"""
|
||||
self.executable = executable
|
||||
self.formatter_init = formatter_init
|
||||
self.program = program
|
||||
|
||||
def generate_commands(self, cases: Sequence[Case]) -> str:
|
||||
"""Generate commands for an ordered list of cases.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
cases : Sequence[Case]
|
||||
Cases to execute in order.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Complete debugger command-file contents.
|
||||
|
||||
Raises
|
||||
------
|
||||
HarnessError
|
||||
If a completed breakpoint group is reopened later.
|
||||
"""
|
||||
closed_stops: set[tuple[str, int]] = set()
|
||||
previous_stop: tuple[str, int] | None = None
|
||||
for case in cases:
|
||||
stop = (case.breakpoint, case.frame)
|
||||
if stop == previous_stop:
|
||||
continue
|
||||
if stop in closed_stops:
|
||||
raise HarnessError(
|
||||
f"breakpoint group {case.breakpoint!r} at frame {case.frame} was reopened"
|
||||
)
|
||||
if previous_stop is not None:
|
||||
closed_stops.add(previous_stop)
|
||||
previous_stop = stop
|
||||
return self._generate_commands(cases)
|
||||
|
||||
@abstractmethod
|
||||
def _generate_commands(self, cases: Sequence[Case]) -> str:
|
||||
"""Generate commands for validated cases.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
cases : Sequence[Case]
|
||||
Cases to execute in order.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Complete debugger command-file contents.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
@abstractmethod
|
||||
def command(self, command_file: Path) -> list[str]:
|
||||
"""Build the debugger subprocess argument list.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
command_file : Path
|
||||
Generated debugger command-file path.
|
||||
|
||||
Returns
|
||||
-------
|
||||
list[str]
|
||||
Subprocess arguments for this debugger.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
def include_transcript_line(self, line: str) -> bool:
|
||||
"""Return whether a line inside a marked section should be retained.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
line : str
|
||||
Transcript line inside a marked section.
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
``True`` when the line belongs in normalized output.
|
||||
"""
|
||||
return True
|
||||
|
||||
def normalize_line(self, line: str) -> str:
|
||||
"""Apply debugger-specific normalization to one output line.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
line : str
|
||||
Extracted debugger output line.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Line after debugger-specific normalization.
|
||||
"""
|
||||
return line
|
||||
|
||||
|
||||
class GDB(DebuggerAdapter):
|
||||
"""Provide GDB-specific pretty-printer test behavior."""
|
||||
|
||||
kind = Debugger.GDB
|
||||
|
||||
def _generate_commands(self, cases: Sequence[Case]) -> str:
|
||||
"""Generate a GDB command file for ordered cases.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
cases : Sequence[Case]
|
||||
Cases to execute in order.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Complete GDB command-file contents.
|
||||
"""
|
||||
lines = [
|
||||
"set pagination off",
|
||||
"set print pretty on",
|
||||
"set print array-indexes on",
|
||||
"set debuginfod enabled off",
|
||||
f"source {self.formatter_init}",
|
||||
]
|
||||
seen_breakpoints: set[str] = set()
|
||||
for case in cases:
|
||||
if case.breakpoint in seen_breakpoints:
|
||||
continue
|
||||
lines.append(f"break {case.breakpoint}")
|
||||
seen_breakpoints.add(case.breakpoint)
|
||||
lines.append("run")
|
||||
previous_stop: tuple[str, int] | None = None
|
||||
for case in cases:
|
||||
stop = (case.breakpoint, case.frame)
|
||||
if previous_stop is not None and stop != previous_stop:
|
||||
lines.append("continue")
|
||||
if stop != previous_stop:
|
||||
lines.append(f"frame {case.frame}")
|
||||
previous_stop = stop
|
||||
begin = marker(case.section, "begin")
|
||||
end = marker(case.section, "end")
|
||||
expression = repr(case.expression)
|
||||
lines.extend(
|
||||
[
|
||||
f"python print({begin!r})",
|
||||
"python",
|
||||
"try:",
|
||||
f" print(gdb.execute('print ' + {expression}, from_tty=True, to_string=True), end='')",
|
||||
"except Exception as error:",
|
||||
" print(error)",
|
||||
"end",
|
||||
f"python print({end!r})",
|
||||
]
|
||||
)
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
def command(self, command_file: Path) -> list[str]:
|
||||
"""Build the GDB subprocess argument list.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
command_file : Path
|
||||
Generated GDB command-file path.
|
||||
|
||||
Returns
|
||||
-------
|
||||
list[str]
|
||||
GDB subprocess arguments.
|
||||
"""
|
||||
return [
|
||||
str(self.executable),
|
||||
"--quiet",
|
||||
"--batch",
|
||||
"--nx",
|
||||
"--command",
|
||||
str(command_file),
|
||||
str(self.program),
|
||||
]
|
||||
|
||||
def normalize_line(self, line: str) -> str:
|
||||
"""Remove GDB value-history prefixes from an output line.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
line : str
|
||||
Extracted GDB output line.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Line without a leading ``$N =`` prefix.
|
||||
"""
|
||||
return _GDB_VALUE_PREFIX_PATTERN.sub("", line)
|
||||
|
||||
|
||||
class LLDB(DebuggerAdapter):
|
||||
"""Provide LLDB-specific pretty-printer test behavior."""
|
||||
|
||||
kind = Debugger.LLDB
|
||||
|
||||
def _generate_commands(self, cases: Sequence[Case]) -> str:
|
||||
"""Generate an LLDB command file for ordered cases.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
cases : Sequence[Case]
|
||||
Cases to execute in order.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Complete LLDB command-file contents.
|
||||
"""
|
||||
lines = [f'command script import "{self.formatter_init}"']
|
||||
seen_breakpoints: set[str] = set()
|
||||
for case in cases:
|
||||
if case.breakpoint in seen_breakpoints:
|
||||
continue
|
||||
lines.append(f"breakpoint set --name {case.breakpoint}")
|
||||
seen_breakpoints.add(case.breakpoint)
|
||||
lines.append("run")
|
||||
previous_stop: tuple[str, int] | None = None
|
||||
for case in cases:
|
||||
stop = (case.breakpoint, case.frame)
|
||||
if previous_stop is not None and stop != previous_stop:
|
||||
lines.append("continue")
|
||||
if stop != previous_stop:
|
||||
lines.append(f"frame select {case.frame}")
|
||||
previous_stop = stop
|
||||
begin = marker(case.section, "begin")
|
||||
end = marker(case.section, "end")
|
||||
debugger_command = f"dwim-print -- {case.expression}"
|
||||
lines.extend(
|
||||
[
|
||||
f"script print({begin!r})",
|
||||
"script result = lldb.SBCommandReturnObject(); "
|
||||
f"status = lldb.debugger.GetCommandInterpreter().HandleCommand({debugger_command!r}, result); "
|
||||
"print(result.GetOutput(), end=''); print(result.GetError(), end='')",
|
||||
f"script print({end!r})",
|
||||
]
|
||||
)
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
def command(self, command_file: Path) -> list[str]:
|
||||
"""Build the LLDB subprocess argument list.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
command_file : Path
|
||||
Generated LLDB command-file path.
|
||||
|
||||
Returns
|
||||
-------
|
||||
list[str]
|
||||
LLDB subprocess arguments.
|
||||
"""
|
||||
return [
|
||||
str(self.executable),
|
||||
"--batch",
|
||||
"--no-lldbinit",
|
||||
"--source",
|
||||
str(command_file),
|
||||
str(self.program),
|
||||
]
|
||||
|
||||
def include_transcript_line(self, line: str) -> bool:
|
||||
"""Exclude LLDB prompt and command-echo lines from marked output.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
line : str
|
||||
Transcript line inside a marked section.
|
||||
|
||||
Returns
|
||||
-------
|
||||
bool
|
||||
``False`` for LLDB prompt or command-echo lines.
|
||||
"""
|
||||
return _LLDB_ECHO_PATTERN.match(line) is None
|
||||
|
||||
|
||||
def extract_sections(
|
||||
transcript: str, section_order: Sequence[str], debugger: DebuggerAdapter
|
||||
) -> str:
|
||||
"""Extract and validate marked sections from a debugger transcript.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
transcript : str
|
||||
Complete combined debugger output.
|
||||
section_order : Sequence[str]
|
||||
Expected section names in manifest order.
|
||||
debugger : DebuggerAdapter
|
||||
Adapter for the debugger that produced the transcript.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Marked sections concatenated in manifest order.
|
||||
|
||||
Raises
|
||||
------
|
||||
HarnessError
|
||||
If markers are unexpected, missing, duplicated, nested, mismatched, or
|
||||
unterminated.
|
||||
"""
|
||||
expected_sections = set(section_order)
|
||||
captured: dict[str, list[str]] = {}
|
||||
active_section: str | None = None
|
||||
|
||||
for line in transcript.splitlines():
|
||||
match = _MARKER_PATTERN.fullmatch(line)
|
||||
if not match:
|
||||
if active_section is None:
|
||||
continue
|
||||
if not debugger.include_transcript_line(line):
|
||||
continue
|
||||
captured[active_section].append(line)
|
||||
continue
|
||||
|
||||
section = match.group("section")
|
||||
if section not in expected_sections:
|
||||
raise HarnessError(f"unexpected marked section: {section}")
|
||||
|
||||
kind = match.group("kind")
|
||||
|
||||
match kind:
|
||||
case "begin":
|
||||
if active_section is not None:
|
||||
raise HarnessError(
|
||||
f"nested section {section!r} inside {active_section!r}"
|
||||
)
|
||||
if section in captured:
|
||||
raise HarnessError(f"duplicate marked section: {section}")
|
||||
captured[section] = [line]
|
||||
active_section = section
|
||||
case "end":
|
||||
if active_section is None:
|
||||
raise HarnessError(f"end marker without begin marker: {section}")
|
||||
if active_section != section:
|
||||
raise HarnessError(
|
||||
f"mismatched end marker for {section!r}; expected {active_section!r}"
|
||||
)
|
||||
captured[section].append(line)
|
||||
active_section = None
|
||||
case _:
|
||||
raise HarnessError(f"invalid marker kind: {kind}")
|
||||
|
||||
if active_section is not None:
|
||||
raise HarnessError(f"unterminated marked section: {active_section}")
|
||||
missing = [section for section in section_order if section not in captured]
|
||||
if missing:
|
||||
raise HarnessError(f"missing marked sections: {', '.join(missing)}")
|
||||
|
||||
lines: list[str] = []
|
||||
for section in section_order:
|
||||
lines.extend(captured[section])
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def normalize_output(output: str, debugger: DebuggerAdapter) -> str:
|
||||
"""Normalize unstable values while preserving output structure.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
output : str
|
||||
Extracted marked output.
|
||||
debugger : DebuggerAdapter
|
||||
Adapter that applies debugger-specific line normalization.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Output with unstable addresses and debugger prefixes normalized.
|
||||
"""
|
||||
normalized_lines: list[str] = []
|
||||
for line in output.splitlines():
|
||||
line = debugger.normalize_line(line.rstrip())
|
||||
# Some debuggers may print C++98 style > > for multiple templates.
|
||||
line = re.sub(r">\s+>", ">>", line)
|
||||
line = _NONZERO_HEX_PATTERN.sub("<address>", line)
|
||||
normalized_lines.append(line)
|
||||
return "\n".join(normalized_lines) + "\n"
|
||||
|
||||
|
||||
def compare_expected(
|
||||
actual: str, expected: str, debugger: DebuggerAdapter, scenario: str
|
||||
) -> None:
|
||||
"""Compare normalized output with its checked-in golden text.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
actual : str
|
||||
Normalized debugger output.
|
||||
expected : str
|
||||
Checked-in golden output.
|
||||
debugger : DebuggerAdapter
|
||||
Adapter for the debugger that produced the output.
|
||||
scenario : str
|
||||
Scenario name used in diagnostics.
|
||||
|
||||
Raises
|
||||
------
|
||||
HarnessError
|
||||
If actual and expected output differ.
|
||||
"""
|
||||
if actual == expected:
|
||||
return
|
||||
difference = "".join(
|
||||
difflib.unified_diff(
|
||||
expected.splitlines(keepends=True),
|
||||
actual.splitlines(keepends=True),
|
||||
fromfile=f"{scenario}/{debugger.kind}.expected",
|
||||
tofile=f"{scenario}/{debugger.kind}.actual",
|
||||
)
|
||||
)
|
||||
raise HarnessError(
|
||||
f"{debugger.kind} pretty-printer output mismatch for {scenario}:\n{difference}"
|
||||
)
|
||||
|
||||
|
||||
def _parse_arguments(arguments: Sequence[str] | None) -> argparse.Namespace:
|
||||
"""Parse debugger configuration and ordered case definitions.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
arguments : Sequence[str] or None
|
||||
Command-line arguments, or ``None`` to use ``sys.argv``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
argparse.Namespace
|
||||
Parsed command-line namespace.
|
||||
"""
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--debugger", type=Debugger, choices=Debugger, required=True)
|
||||
parser.add_argument("--debugger-executable", type=Path, required=True)
|
||||
parser.add_argument("--program", type=Path, required=True)
|
||||
parser.add_argument("--formatter-init", type=Path, required=True)
|
||||
parser.add_argument("--expected", type=Path, required=True)
|
||||
parser.add_argument("--output-log", type=Path, required=True)
|
||||
parser.add_argument("--timeout", type=float, default=90.0)
|
||||
parser.add_argument("--update-expected", action="store_true")
|
||||
parser.add_argument(
|
||||
"--case",
|
||||
dest="cases",
|
||||
nargs=4,
|
||||
action=CaseAction,
|
||||
default=None,
|
||||
required=True,
|
||||
)
|
||||
return parser.parse_args(arguments)
|
||||
|
||||
|
||||
def _create_debugger(args: argparse.Namespace) -> DebuggerAdapter:
|
||||
"""Create the debugger adapter selected by command-line arguments.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
args : argparse.Namespace
|
||||
Parsed runner arguments.
|
||||
|
||||
Returns
|
||||
-------
|
||||
DebuggerAdapter
|
||||
Configured adapter for the selected debugger.
|
||||
|
||||
Raises
|
||||
------
|
||||
HarnessError
|
||||
If the selected debugger is unsupported.
|
||||
"""
|
||||
match args.debugger:
|
||||
case Debugger.LLDB:
|
||||
return LLDB(args.debugger_executable, args.formatter_init, args.program)
|
||||
case Debugger.GDB:
|
||||
return GDB(args.debugger_executable, args.formatter_init, args.program)
|
||||
case _:
|
||||
raise HarnessError(f"unsupported debugger: {args.debugger}")
|
||||
|
||||
|
||||
def _run_debugger(
|
||||
args: argparse.Namespace, debugger: DebuggerAdapter, command_file: Path
|
||||
) -> str:
|
||||
"""Run the debugger and persist its complete transcript.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
args : argparse.Namespace
|
||||
Parsed runner arguments.
|
||||
debugger : DebuggerAdapter
|
||||
Configured debugger adapter.
|
||||
command_file : Path
|
||||
Generated debugger command-file path.
|
||||
|
||||
Returns
|
||||
-------
|
||||
str
|
||||
Complete combined debugger output.
|
||||
|
||||
Raises
|
||||
------
|
||||
DebuggerError
|
||||
If the debugger cannot launch, times out, or exits with a nonzero status.
|
||||
OSError
|
||||
If the transcript cannot be written.
|
||||
"""
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
debugger.command(command_file),
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
text=True,
|
||||
timeout=args.timeout,
|
||||
)
|
||||
except subprocess.TimeoutExpired as error:
|
||||
if isinstance(error.stdout, str):
|
||||
args.output_log.write_text(error.stdout)
|
||||
raise DebuggerError(
|
||||
f"{debugger.kind} timed out after {args.timeout:g} seconds"
|
||||
) from error
|
||||
except OSError as error:
|
||||
raise DebuggerError(f"failed to launch {debugger.kind}: {error}") from error
|
||||
|
||||
args.output_log.write_text(completed.stdout)
|
||||
if completed.returncode != 0:
|
||||
raise DebuggerError(
|
||||
f"{debugger.kind} exited with status {completed.returncode}"
|
||||
)
|
||||
return completed.stdout
|
||||
|
||||
|
||||
def _match_output(
|
||||
args: argparse.Namespace,
|
||||
debugger: DebuggerAdapter,
|
||||
cases: Sequence[Case],
|
||||
transcript: str,
|
||||
) -> None:
|
||||
"""Extract, normalize, and compare or update debugger output.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
args : argparse.Namespace
|
||||
Parsed runner arguments.
|
||||
debugger : DebuggerAdapter
|
||||
Configured debugger adapter.
|
||||
cases : Sequence[Case]
|
||||
Validated cases in manifest order.
|
||||
transcript : str
|
||||
Complete combined debugger output.
|
||||
|
||||
Raises
|
||||
------
|
||||
HarnessError
|
||||
If marked output is invalid or differs from the golden.
|
||||
OSError
|
||||
If the golden file cannot be read or updated.
|
||||
"""
|
||||
extracted = extract_sections(transcript, [case.section for case in cases], debugger)
|
||||
actual = normalize_output(extracted, debugger)
|
||||
if args.update_expected:
|
||||
args.expected.write_text(actual)
|
||||
return
|
||||
compare_expected(
|
||||
actual,
|
||||
args.expected.read_text(),
|
||||
debugger,
|
||||
args.expected.parent.name,
|
||||
)
|
||||
|
||||
|
||||
def _report_error(
|
||||
args: argparse.Namespace,
|
||||
debugger: DebuggerAdapter,
|
||||
command_file: Path,
|
||||
error: Exception,
|
||||
) -> None:
|
||||
"""Report a debugger or output-matching failure with artifact paths.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
args : argparse.Namespace
|
||||
Parsed runner arguments.
|
||||
debugger : DebuggerAdapter
|
||||
Configured debugger adapter.
|
||||
command_file : Path
|
||||
Generated debugger command-file path.
|
||||
error : Exception
|
||||
Failure being reported.
|
||||
"""
|
||||
scenario = args.expected.parent.name
|
||||
print(
|
||||
f"error: {debugger.kind} pretty-printer test for {scenario}: {error}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
print(f"debugger commands: {command_file}", file=sys.stderr)
|
||||
if args.output_log.exists():
|
||||
print(f"complete transcript: {args.output_log}", file=sys.stderr)
|
||||
|
||||
|
||||
def main(arguments: Sequence[str] | None = None) -> int:
|
||||
"""Run one debugger pretty-printer test from command-line arguments.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
arguments : Sequence[str] or None
|
||||
Command-line arguments, or ``None`` to use ``sys.argv``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
int
|
||||
Zero on success and one for handled debugger or matching failures.
|
||||
|
||||
Raises
|
||||
------
|
||||
HarnessError
|
||||
If case setup is invalid.
|
||||
OSError
|
||||
If setup artifacts or golden files cannot be accessed.
|
||||
"""
|
||||
args = _parse_arguments(arguments)
|
||||
debugger = _create_debugger(args)
|
||||
commands = debugger.generate_commands(args.cases)
|
||||
command_file = args.output_log.with_suffix(".commands")
|
||||
args.output_log.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output_log.unlink(missing_ok=True)
|
||||
command_file.write_text(commands)
|
||||
|
||||
try:
|
||||
transcript = _run_debugger(args, debugger, command_file)
|
||||
except DebuggerError as error:
|
||||
_report_error(args, debugger, command_file, error)
|
||||
return 1
|
||||
|
||||
try:
|
||||
_match_output(args, debugger, args.cases, transcript)
|
||||
except HarnessError as error:
|
||||
_report_error(args, debugger, command_file, error)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
302
cccl_upstream/libcudacxx/test/libcudacxx/CMakeLists.txt
Normal file
302
cccl_upstream/libcudacxx/test/libcudacxx/CMakeLists.txt
Normal file
@@ -0,0 +1,302 @@
|
||||
option(
|
||||
LIBCUDACXX_TEST_WITH_NVRTC
|
||||
"Test libcu++ with runtime compilation instead of offline compilation. Only runs device side tests."
|
||||
OFF
|
||||
)
|
||||
|
||||
###############################################################################
|
||||
### C2H tests:
|
||||
cccl_get_c2h()
|
||||
|
||||
set(c2h_all_target "libcudacxx.test.c2h_all")
|
||||
add_custom_target(${c2h_all_target})
|
||||
|
||||
if (NOT LIBCUDACXX_TEST_WITH_NVRTC AND NOT CCCL_ENABLE_TILE)
|
||||
file(
|
||||
GLOB_RECURSE test_srcs
|
||||
RELATIVE "${CMAKE_CURRENT_LIST_DIR}"
|
||||
CONFIGURE_DEPENDS
|
||||
*.cu
|
||||
)
|
||||
|
||||
function(libcudacxx_add_test target_name_var source)
|
||||
string(REPLACE "/" "." target_name "${source}")
|
||||
string(PREPEND target_name "libcudacxx.test.")
|
||||
string(REGEX REPLACE "\\.[^.]+$" "" target_name "${target_name}")
|
||||
set(${target_name_var} ${target_name} PARENT_SCOPE)
|
||||
|
||||
cccl_add_executable(
|
||||
${target_name}
|
||||
ADD_CTEST
|
||||
NO_METATARGETS
|
||||
DIALECT ${CMAKE_CUDA_STANDARD}
|
||||
SOURCES "${source}"
|
||||
)
|
||||
target_include_directories(
|
||||
${target_name}
|
||||
PRIVATE
|
||||
"${libcudacxx_SOURCE_DIR}/test/libcudacxx/cuda/ccclrt/common"
|
||||
"${libcudacxx_SOURCE_DIR}/test/support"
|
||||
)
|
||||
target_link_libraries(
|
||||
${target_name}
|
||||
PRIVATE #
|
||||
libcudacxx.compiler_interface
|
||||
cccl.c2h.main
|
||||
)
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
target_compile_options(${target_name} PRIVATE "-Wno-attributes")
|
||||
endif()
|
||||
|
||||
add_dependencies(${c2h_all_target} ${target_name})
|
||||
endfunction()
|
||||
|
||||
foreach (test_src IN LISTS test_srcs)
|
||||
libcudacxx_add_test(test_target "${test_src}")
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
###############################################################################
|
||||
### Lit tests:
|
||||
macro(pythonize_bool var)
|
||||
if (${var})
|
||||
set(${var} True)
|
||||
else()
|
||||
set(${var} False)
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
cccl_get_cudatoolkit()
|
||||
cccl_get_dlpack()
|
||||
|
||||
get_target_property(CUDA_INCLUDE_DIR CUDA::cudart INTERFACE_INCLUDE_DIRECTORIES)
|
||||
|
||||
message(STATUS "Lit enabled CUDA architectures: ${CMAKE_CUDA_ARCHITECTURES}")
|
||||
|
||||
if (LIBCUDACXX_TEST_WITH_NVRTC)
|
||||
# TODO: Use project properties to get path to binary.
|
||||
# Should also set up dependency on the project when NVRTC is enabled
|
||||
foreach (include IN ITEMS ${CUDA_INCLUDE_DIR})
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -I'${include}'")
|
||||
endforeach()
|
||||
set(
|
||||
LIBCUDACXX_CUDA_COMPILER
|
||||
"${CMAKE_BINARY_DIR}/libcudacxx/test/utils/nvidia/nvrtc/nvrtcc"
|
||||
)
|
||||
set(LIBCUDACXX_CUDA_COMPILER_ARG1 "")
|
||||
set(LIBCUDACXX_CUDA_TEST_WITH_NVRTC "True")
|
||||
# Use the NVRTCC utility to run the built test outputs
|
||||
set(
|
||||
LIBCUDACXX_EXECUTOR
|
||||
"PrefixExecutor(['${LIBCUDACXX_CUDA_COMPILER}'], LocalExecutor())"
|
||||
)
|
||||
# Enable 128-bit types for NVRTC
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -device-int128")
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -device-float128")
|
||||
else() # NOT LIBCUDACXX_TEST_WITH_NVRTC
|
||||
set(
|
||||
LIBCUDACXX_FORCE_INCLUDE
|
||||
"-include ${libcudacxx_SOURCE_DIR}/test/libcudacxx/force_include.h"
|
||||
)
|
||||
set(LIBCUDACXX_CUDA_COMPILER "${CMAKE_CUDA_COMPILER}")
|
||||
set(LIBCUDACXX_CUDA_TEST_WITH_NVRTC "False")
|
||||
endif()
|
||||
|
||||
# enable exceptions and assertions in tests
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -DCCCL_ENABLE_ASSERTIONS")
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -I ${dlpack_SOURCE_DIR}/include")
|
||||
|
||||
# Disable dialect deprecation
|
||||
string(
|
||||
APPEND LIBCUDACXX_TEST_COMPILER_FLAGS
|
||||
" -DCCCL_IGNORE_DEPRECATED_CPP_DIALECT"
|
||||
)
|
||||
string(
|
||||
APPEND LIBCUDACXX_TEST_COMPILER_FLAGS
|
||||
" -DLIBCUDACXX_IGNORE_DEPRECATED_ABI"
|
||||
)
|
||||
|
||||
# enable tile support
|
||||
if (CCCL_ENABLE_TILE)
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " --enable-tile")
|
||||
set(LIBCUDACXX_ENABLE_TILE True)
|
||||
else()
|
||||
set(LIBCUDACXX_ENABLE_TILE False)
|
||||
endif()
|
||||
|
||||
if ("${CMAKE_CXX_COMPILER_ID}" STREQUAL "GNU")
|
||||
string(APPEND LIBCUDACXX_TEST_LINKER_FLAGS " -latomic")
|
||||
endif()
|
||||
|
||||
if (NOT MSVC AND NOT ${CMAKE_CUDA_COMPILER_ID} STREQUAL "Clang")
|
||||
set(
|
||||
LIBCUDACXX_WARNING_LEVEL
|
||||
"--compiler-options=-Wall --compiler-options=-Wextra"
|
||||
)
|
||||
endif()
|
||||
|
||||
if (MSVC)
|
||||
# We want to use cudaLaunchKernelEx which is guarded by __cplusplus
|
||||
if ("${CMAKE_CUDA_COMPILER_VERSION}" LESS "12.3.0")
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -Xcompiler=/Zc:__cplusplus")
|
||||
endif()
|
||||
|
||||
# Require the conforming preprocessor
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -Xcompiler=/Zc:preprocessor")
|
||||
if (MSVC_TOOLSET_VERSION LESS 143)
|
||||
# winbase.h(9572): warning C5105: macro expansion producing 'defined' has undefined behavior
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -Xcompiler=/wd5105")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (${CMAKE_CUDA_COMPILER_ID} STREQUAL "Clang")
|
||||
string(
|
||||
APPEND LIBCUDACXX_TEST_COMPILER_FLAGS
|
||||
" ${CMAKE_CUDA_FLAGS}"
|
||||
" -Xclang -fcuda-allow-variadic-functions"
|
||||
" -Xclang -Wno-unused-parameter"
|
||||
" -Wno-unknown-cuda-version"
|
||||
" ${LIBCUDACXX_FORCE_INCLUDE}"
|
||||
" -I${libcudacxx_SOURCE_DIR}/include"
|
||||
" ${LIBCUDACXX_WARNING_LEVEL}"
|
||||
)
|
||||
|
||||
string(
|
||||
APPEND LIBCUDACXX_TEST_LINKER_FLAGS
|
||||
" ${CMAKE_CUDA_FLAGS}"
|
||||
" -L${CUDAToolkit_LIBRARY_DIR}"
|
||||
" -lcuda"
|
||||
" -lcudart"
|
||||
)
|
||||
elseif (${CMAKE_CUDA_COMPILER_ID} STREQUAL "NVIDIA")
|
||||
string(
|
||||
APPEND LIBCUDACXX_TEST_COMPILER_FLAGS
|
||||
" ${LIBCUDACXX_FORCE_INCLUDE}"
|
||||
" ${LIBCUDACXX_WARNING_LEVEL}"
|
||||
" -Wno-deprecated-gpu-targets"
|
||||
)
|
||||
elseif (${CMAKE_CUDA_COMPILER_ID} STREQUAL "NVHPC")
|
||||
string(APPEND LIBCUDACXX_TEST_COMPILER_FLAGS " -stdpar")
|
||||
string(APPEND LIBCUDACXX_TEST_LINKER_FLAGS " -stdpar")
|
||||
endif()
|
||||
|
||||
include(AddLLVM)
|
||||
|
||||
set(LIBCUDACXX_BINARY_DIR "${CMAKE_CURRENT_BINARY_DIR}")
|
||||
|
||||
set(
|
||||
LIBCUDACXX_TARGET_INFO
|
||||
"libcudacxx.test.target_info.LocalTI"
|
||||
CACHE STRING
|
||||
"TargetInfo to use when setting up test environment."
|
||||
)
|
||||
set(
|
||||
LIBCUDACXX_EXECUTOR
|
||||
"None"
|
||||
CACHE STRING
|
||||
"Executor to use when running tests."
|
||||
)
|
||||
|
||||
set(
|
||||
LIBCUDACXX_TEST_TIMEOUT
|
||||
"200"
|
||||
CACHE STRING
|
||||
"Enable test timeouts (Default = 200, Off = 0)"
|
||||
)
|
||||
|
||||
set(
|
||||
AUTO_GEN_COMMENT
|
||||
"## Autogenerated by libcudacxx configuration.\n# Do not edit!"
|
||||
)
|
||||
|
||||
set(LIBCUDACXX_TEST_STANDARD_VER "c++${CMAKE_CUDA_STANDARD}")
|
||||
|
||||
# Pedantic = werror ON and system header pragma OFF (both at their cmake defaults)
|
||||
if (CCCL_ENABLE_WERROR AND NOT CCCL_ENABLE_PRAGMA_SYSTEM_HEADER)
|
||||
set(LIBCUDACXX_ENABLE_PEDANTIC_WARNINGS True)
|
||||
else()
|
||||
set(LIBCUDACXX_ENABLE_PEDANTIC_WARNINGS False)
|
||||
endif()
|
||||
|
||||
set(lit_site_cfg_path "${CMAKE_CURRENT_BINARY_DIR}/lit.site.cfg")
|
||||
configure_lit_site_cfg(
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/lit.site.cfg.in"
|
||||
"${lit_site_cfg_path}"
|
||||
)
|
||||
|
||||
add_lit_testsuite(check-cudacxx
|
||||
"Running libcu++ tests"
|
||||
"${CMAKE_CURRENT_BINARY_DIR}"
|
||||
)
|
||||
|
||||
find_program(libcudacxx_LIT lit REQUIRED)
|
||||
|
||||
set(
|
||||
libcudacxx_LIT_FLAGS
|
||||
""
|
||||
CACHE STRING
|
||||
"Semi-colon separated list of flags passed to the invocation of lit."
|
||||
)
|
||||
message(STATUS "libcudacxx_LIT_FLAGS: ${libcudacxx_LIT_FLAGS}")
|
||||
|
||||
if (NOT LIBCUDACXX_TEST_WITH_NVRTC)
|
||||
# Build but don't run the tests. Used by CI to pre-seed sccache for the test machines.
|
||||
# Only executed if explicitly requested.
|
||||
add_custom_target(
|
||||
libcudacxx.test.lit.precompile
|
||||
# HACK: There is no way to tell CMake/ninja to always build a target serially,
|
||||
# so we make this target depend on all other libcudacxx targets to avoid oversubscribing
|
||||
# the build machine.
|
||||
# FIXME: This has nasty side effects:
|
||||
# - It's fragile and must be updated every time we add new targets to libcudacxx
|
||||
# - It oversubs `-dev` presets that configure libcudacxx alongside other CCCL projects
|
||||
# - It makes it impossible to just build this target alone since it brings in the world
|
||||
# See related issue https://github.com/NVIDIA/cccl/issues/6163.
|
||||
DEPENDS
|
||||
libcudacxx.test.public_headers
|
||||
libcudacxx.test.internal_headers
|
||||
libcudacxx.test.public_headers_host_only
|
||||
libcudacxx.test.c2h_all
|
||||
# gersemi: off
|
||||
COMMAND
|
||||
"${CMAKE_COMMAND}" -E env "LIBCUDACXX_SITE_CONFIG=${lit_site_cfg_path}"
|
||||
"${libcudacxx_LIT}"
|
||||
-vv --no-progress-bar --time-tests
|
||||
${libcudacxx_LIT_FLAGS}
|
||||
"-Dexecutor=\"NoopExecutor()\""
|
||||
"${libcudacxx_SOURCE_DIR}/test/libcudacxx"
|
||||
# gersemi: on
|
||||
USES_TERMINAL
|
||||
)
|
||||
endif()
|
||||
|
||||
# Restricted to avoid oversubscribing the GPU:
|
||||
set(
|
||||
libcudacxx_LIT_PARALLEL_LEVEL
|
||||
8
|
||||
CACHE STRING
|
||||
"Parallelism used to run libcudacxx's lit test suite."
|
||||
)
|
||||
|
||||
add_test(
|
||||
NAME libcudacxx.test.lit
|
||||
# gersemi: off
|
||||
COMMAND
|
||||
"${CMAKE_COMMAND}" -E env "LIBCUDACXX_SITE_CONFIG=${lit_site_cfg_path}"
|
||||
"${libcudacxx_LIT}"
|
||||
-vv --no-progress-bar --time-tests
|
||||
${libcudacxx_LIT_FLAGS}
|
||||
-j "${libcudacxx_LIT_PARALLEL_LEVEL}"
|
||||
"${libcudacxx_SOURCE_DIR}/test/libcudacxx"
|
||||
# gersemi: on
|
||||
USES_TERMINAL
|
||||
)
|
||||
|
||||
set_tests_properties(
|
||||
libcudacxx.test.lit
|
||||
PROPERTIES
|
||||
# 6hr to match CI timeout
|
||||
TIMEOUT 21600
|
||||
RUN_SERIAL TRUE
|
||||
)
|
||||
@@ -0,0 +1,70 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "test_macros.h"
|
||||
#include "utils.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC __noinline__ void test_global_implicit_property(T ap, cudaAccessProperty cp)
|
||||
{
|
||||
// Test implicit conversions
|
||||
cudaAccessProperty v = ap;
|
||||
assert(cp == v);
|
||||
|
||||
// Test default, copy constructor, and copy-assignent
|
||||
cuda::access_property o(ap);
|
||||
cuda::access_property d;
|
||||
d = ap;
|
||||
|
||||
// Test explicit conversion to i64
|
||||
uint64_t x = (uint64_t) o;
|
||||
uint64_t y = (uint64_t) d;
|
||||
assert(x == y);
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_global()
|
||||
{
|
||||
cuda::access_property o(cuda::access_property::global{});
|
||||
uint64_t x = (uint64_t) o;
|
||||
unused(x);
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_shared()
|
||||
{
|
||||
(void) cuda::access_property::shared{};
|
||||
}
|
||||
|
||||
static_assert(sizeof(cuda::access_property::shared) == 1);
|
||||
static_assert(sizeof(cuda::access_property::global) == 1);
|
||||
static_assert(sizeof(cuda::access_property::persisting) == 1);
|
||||
static_assert(sizeof(cuda::access_property::normal) == 1);
|
||||
static_assert(sizeof(cuda::access_property::streaming) == 1);
|
||||
static_assert(sizeof(cuda::access_property) == 8);
|
||||
|
||||
static_assert(alignof(cuda::access_property::shared) == 1);
|
||||
static_assert(alignof(cuda::access_property::global) == 1);
|
||||
static_assert(alignof(cuda::access_property::persisting) == 1);
|
||||
static_assert(alignof(cuda::access_property::normal) == 1);
|
||||
static_assert(alignof(cuda::access_property::streaming) == 1);
|
||||
static_assert(alignof(cuda::access_property) == 8);
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_global_implicit_property(cuda::access_property::normal{}, cudaAccessProperty::cudaAccessPropertyNormal);
|
||||
test_global_implicit_property(cuda::access_property::streaming{}, cudaAccessProperty::cudaAccessPropertyStreaming);
|
||||
test_global_implicit_property(cuda::access_property::persisting{}, cudaAccessProperty::cudaAccessPropertyPersisting);
|
||||
|
||||
test_global();
|
||||
test_shared();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
// error: expression must have a constant value annotated_ptr.h: note #2701-D: attempt to access run-time storage
|
||||
// UNSUPPORTED: clang-14, gcc-11, gcc-10, gcc-9, gcc-8, gcc-7, msvc-19.29
|
||||
|
||||
#include <cuda/annotated_ptr>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_FUNC constexpr bool test_constexpr()
|
||||
{
|
||||
using namespace cuda;
|
||||
access_property a{}; // default constructor
|
||||
access_property b{a}; // copy constructor
|
||||
access_property c{cuda::std::move(a)}; // move constructor
|
||||
// user-declared ctor
|
||||
access_property d1{access_property::global{}};
|
||||
access_property d2{access_property::normal{}};
|
||||
access_property d3{access_property::streaming{}};
|
||||
access_property d4{access_property::persisting{}};
|
||||
auto p1 = static_cast<cudaAccessProperty>(access_property::normal{});
|
||||
auto p2 = static_cast<cudaAccessProperty>(access_property::streaming{});
|
||||
auto p3 = static_cast<cudaAccessProperty>(access_property::persisting{});
|
||||
// fraction ctor
|
||||
access_property e1{access_property::normal{}, 1.0f};
|
||||
access_property e2{access_property::streaming{}, 1.0f};
|
||||
access_property e3{access_property::persisting{}, 1.0f};
|
||||
access_property e4{access_property::normal{}, 1.0f, access_property::streaming{}};
|
||||
access_property e5{access_property::persisting{}, 1.0f, access_property::streaming{}};
|
||||
b = a; // copy assignment
|
||||
b = cuda::std::move(a); // move assignment
|
||||
auto value = static_cast<uint64_t>(a);
|
||||
unused(p1, p2, p3, b, c, d1, d2, d3, d4, e1, e2, e3, e4, e5, value);
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
static_assert(test_constexpr());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
TEST_FUNC __noinline__ void test_access_property_fail()
|
||||
{
|
||||
cuda::access_property o = cuda::access_property::normal{};
|
||||
// Test implicit conversion fails
|
||||
std::uint64_t x;
|
||||
x = o;
|
||||
unused(o);
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_access_property_fail();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,148 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// UNSUPPORTED: pre-sm-80
|
||||
|
||||
#include <cuda/annotated_ptr>
|
||||
#include <cuda/cmath>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename Prop>
|
||||
TEST_DEVICE_FUNC constexpr cuda::__l2_evict_t to_enum()
|
||||
{
|
||||
if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::normal>)
|
||||
{
|
||||
return cuda::__l2_evict_t::_L2_Evict_Normal_Demote;
|
||||
}
|
||||
else if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::streaming>)
|
||||
{
|
||||
return cuda::__l2_evict_t::_L2_Evict_First;
|
||||
}
|
||||
else if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::persisting>)
|
||||
{
|
||||
return cuda::__l2_evict_t::_L2_Evict_Last;
|
||||
}
|
||||
else // if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::global>)
|
||||
{
|
||||
return cuda::__l2_evict_t::_L2_Evict_Unchanged;
|
||||
}
|
||||
}
|
||||
|
||||
//----------------------------------------------------------------------------------------------------------------------
|
||||
// test range
|
||||
|
||||
template <typename Primary, typename Secondary = void, int I = 1>
|
||||
TEST_DEVICE_FUNC void test_fraction_constexpr()
|
||||
{
|
||||
if constexpr (I > 16)
|
||||
{
|
||||
return;
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr auto fraction = static_cast<float>(I) * (1.0f / 16.0f);
|
||||
auto policy = cuda::__createpolicy_fraction(to_enum<Primary>(), to_enum<Secondary>(), fraction);
|
||||
if constexpr (cuda::std::is_void_v<Secondary>)
|
||||
{
|
||||
constexpr cuda::access_property property{Primary{}, fraction};
|
||||
assert(static_cast<uint64_t>(property) == policy);
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr cuda::access_property property{Primary{}, fraction, Secondary{}};
|
||||
assert(static_cast<uint64_t>(property) == policy);
|
||||
}
|
||||
test_fraction_constexpr<Primary, Secondary, I + 1>();
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void test_fraction()
|
||||
{
|
||||
test_fraction_constexpr<cuda::access_property::normal>();
|
||||
test_fraction_constexpr<cuda::access_property::streaming>();
|
||||
test_fraction_constexpr<cuda::access_property::persisting>();
|
||||
test_fraction_constexpr<cuda::access_property::normal, cuda::access_property::streaming>();
|
||||
test_fraction_constexpr<cuda::access_property::persisting, cuda::access_property::streaming>();
|
||||
}
|
||||
|
||||
//----------------------------------------------------------------------------------------------------------------------
|
||||
// test range
|
||||
|
||||
template <typename Primary, typename Secondary>
|
||||
__global__ void test_range_kernel(void* ptr, uint64_t property, uint32_t primary_size, uint32_t total_size)
|
||||
{
|
||||
auto policy = __createpolicy_range(to_enum<Primary>(), to_enum<Secondary>(), ptr, primary_size, total_size);
|
||||
if (static_cast<uint64_t>(property) != policy)
|
||||
{
|
||||
printf(" primary_size = %u, total_size = %u\n", primary_size, total_size);
|
||||
printf(" primary = %u, secondary = %u\n", (int) to_enum<Primary>(), (int) to_enum<Secondary>());
|
||||
printf(" 0x%llX vs 0x%llX\n", static_cast<unsigned long long>(policy), static_cast<unsigned long long>(property));
|
||||
}
|
||||
assert(static_cast<uint64_t>(property) == policy);
|
||||
}
|
||||
|
||||
template <typename Primary, typename Secondary = void>
|
||||
void test_range_launch(void* ptr, uint32_t primary_size, uint32_t total_size)
|
||||
{
|
||||
cuda::access_property property;
|
||||
if constexpr (cuda::std::is_void_v<Secondary>)
|
||||
{
|
||||
property = cuda::access_property{ptr, primary_size, total_size, Primary{}};
|
||||
}
|
||||
else
|
||||
{
|
||||
property = cuda::access_property{ptr, primary_size, total_size, Primary{}, Secondary{}};
|
||||
}
|
||||
test_range_kernel<Primary, Secondary><<<1, 1>>>(ptr, static_cast<uint64_t>(property), primary_size, total_size);
|
||||
}
|
||||
|
||||
void test_range()
|
||||
{
|
||||
int* ptr = nullptr;
|
||||
ptr++;
|
||||
for (uint32_t total_size = 1, i = 0; i <= 31; i++, total_size <<= 1)
|
||||
{
|
||||
for (uint32_t primary_size = 1, j = 0; j <= i; j++, primary_size <<= 1)
|
||||
{
|
||||
test_range_launch<cuda::access_property::normal>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::persisting>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::global, cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::normal, cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::streaming, cuda::access_property::streaming>(
|
||||
ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::persisting, cuda::access_property::streaming>(
|
||||
ptr, primary_size, total_size);
|
||||
}
|
||||
}
|
||||
// PTX createpolicy_range and access_property behaviors don't match (for now)
|
||||
// uint32_t primary_size = 0xFFFFFFFF;
|
||||
// uint32_t total_size = 0xFFFFFFFF;
|
||||
// test_range_launch<cuda::access_property::normal>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::persisting>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::global, cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::normal, cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::persisting, cuda::access_property::streaming>(ptr, primary_size,
|
||||
// total_size);
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, (test_range();))
|
||||
NV_IF_TARGET(NV_IS_HOST, (test_fraction<<<1, 1>>>();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,164 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
static_assert(sizeof(cuda::annotated_ptr<int, cuda::access_property::global>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<char, cuda::access_property::global>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::global>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::persisting>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::normal>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::streaming>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property>) == 2 * sizeof(uintptr_t),
|
||||
"annotated_ptr<T,access_property> must be 2 * pointer size");
|
||||
|
||||
// NOTE: we could make these smaller in the future (e.g. 32-bit) but that would be an ABI breaking change:
|
||||
static_assert(sizeof(cuda::annotated_ptr<int, cuda::access_property::shared>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, shared> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<char, cuda::access_property::shared>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, shared> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::shared>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, shared> must be pointer size");
|
||||
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::global>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::persisting>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::normal>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::streaming>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
|
||||
// NOTE: we could lower the alignment in the future but that would be an ABI breaking change:
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::shared>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
|
||||
#define N 128
|
||||
|
||||
struct S
|
||||
{
|
||||
int x;
|
||||
TEST_FUNC S& operator=(int o)
|
||||
{
|
||||
this->x = o;
|
||||
return *this;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename In, typename T>
|
||||
TEST_FUNC __noinline__ void test_read_access(In i, T* r)
|
||||
{
|
||||
assert(i);
|
||||
assert(i - i == 0);
|
||||
assert((bool) i);
|
||||
const In o = i;
|
||||
|
||||
// assert(i->x == 0); // FAILS with shmem
|
||||
// assert(o->x == 0); // FAILS with shmem
|
||||
for (int n = 0; n < N; ++n)
|
||||
{
|
||||
assert(i[n].x == n);
|
||||
assert(&i[n] == &i[n]);
|
||||
assert(&i[n] == &r[n]);
|
||||
assert(o[n].x == n);
|
||||
assert(&o[n] == &o[n]);
|
||||
assert(&o[n] == &r[n]);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename In>
|
||||
TEST_FUNC __noinline__ void test_write_access(In i)
|
||||
{
|
||||
assert(i);
|
||||
assert((bool) i);
|
||||
const In o = i;
|
||||
|
||||
for (int n = 0; n < N; ++n)
|
||||
{
|
||||
i[n].x = 2 * n;
|
||||
assert(i[n].x == 2 * n);
|
||||
assert(i[n].x == 2 * n);
|
||||
i[n].x = n;
|
||||
|
||||
o[n].x = 2 * n;
|
||||
assert(o[n].x == 2 * n);
|
||||
assert(o[n].x == 2 * n);
|
||||
o[n].x = n;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void all_tests()
|
||||
{
|
||||
S* arr = global_alloc<S, N>();
|
||||
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::normal>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::streaming>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::persisting>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::global>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property>(arr), arr);
|
||||
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::normal>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::streaming>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::persisting>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::global>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property>(arr), arr);
|
||||
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::normal>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::streaming>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::persisting>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::global>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property>(arr), arr);
|
||||
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::normal>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::streaming>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::persisting>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::global>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property>(arr), arr);
|
||||
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::normal>(arr));
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::streaming>(arr));
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::persisting>(arr));
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::global>(arr));
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property>(arr));
|
||||
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::normal>(arr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::streaming>(arr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::persisting>(arr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::global>(arr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property>(arr));
|
||||
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(S* sarr = shared_alloc<S, N>(); // Allocating shared memory is only supported on device
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::shared>(sarr), sarr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::shared>(sarr), sarr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::shared>(sarr), sarr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::shared>(sarr), sarr);
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::shared>(sarr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::shared>(sarr));))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
all_tests();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,157 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "test_macros.h"
|
||||
#include "utils.h"
|
||||
|
||||
TEST_DEVICE_FUNC void annotated_ptr_timing_dev(int* in, int* out)
|
||||
{
|
||||
cuda::access_property ap(cuda::access_property::persisting{});
|
||||
// Retrieve global id
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
cuda::annotated_ptr<int, cuda::access_property> in_ann{in, ap};
|
||||
cuda::annotated_ptr<int, cuda::access_property> out_ann{out, ap};
|
||||
|
||||
DPRINTF("&out[i]:%p = &in[i]:%p for i = %d\n", &out[i], &in[i], i);
|
||||
DPRINTF("&out[i]:%p = &in_ann[i]:%p for i = %d\n", &out_ann[i], &in_ann[i], i);
|
||||
|
||||
out_ann[i] = in_ann[i];
|
||||
};
|
||||
|
||||
__global__ void annotated_ptr_timing(int* in, int* out)
|
||||
{
|
||||
annotated_ptr_timing_dev(in, out);
|
||||
}
|
||||
|
||||
TEST_DEVICE_FUNC void ptr_timing_dev(int* in, int* out)
|
||||
{
|
||||
// Retrieve global id
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
DPRINTF("&out[i]:%p = &in[i]:%p for i = %d\n", &out[i], &in[i], i);
|
||||
out[i] = in[i];
|
||||
};
|
||||
|
||||
__global__ void ptr_timing(int* in, int* out)
|
||||
{
|
||||
ptr_timing_dev(in, out);
|
||||
};
|
||||
|
||||
TEST_FUNC __noinline__ void bench()
|
||||
{
|
||||
#ifndef __CUDA_ARCH__
|
||||
static const size_t ARR_SZ = 1 << 22;
|
||||
static const size_t THREAD_CNT = 128;
|
||||
static const size_t BLOCK_CNT = ARR_SZ / THREAD_CNT;
|
||||
const dim3 threads(THREAD_CNT, 1, 1), blocks(BLOCK_CNT, 1, 1);
|
||||
cudaEvent_t start, stop;
|
||||
#else
|
||||
static const size_t ARR_SZ = 1 << 10;
|
||||
#endif
|
||||
int* arr0 = nullptr;
|
||||
int* arr1 = nullptr;
|
||||
float annotated_time = 0.f, pointer_time = 0.f;
|
||||
|
||||
#ifdef __CUDA_ARCH__
|
||||
arr0 = (int*) malloc(ARR_SZ * sizeof(int));
|
||||
arr1 = (int*) malloc(ARR_SZ * sizeof(int));
|
||||
#else
|
||||
assert_rt(cudaMallocManaged((void**) &arr0, ARR_SZ * sizeof(int)));
|
||||
assert_rt(cudaMallocManaged((void**) &arr1, ARR_SZ * sizeof(int)));
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
#endif
|
||||
|
||||
#ifdef __CUDA_ARCH__
|
||||
ptr_timing_dev(arr0, arr1);
|
||||
#else
|
||||
ptr_timing<<<blocks, threads>>>(arr0, arr1);
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
#endif
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
arr0[i] = static_cast<int>(i);
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
#ifdef __CUDA_ARCH__
|
||||
ptr_timing_dev(arr0, arr1);
|
||||
#else
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
assert_rt(cudaEventCreate(&start));
|
||||
assert_rt(cudaEventCreate(&stop));
|
||||
assert_rt(cudaEventRecord(start));
|
||||
ptr_timing<<<blocks, threads>>>(arr0, arr1);
|
||||
assert_rt(cudaEventRecord(stop));
|
||||
assert_rt(cudaEventSynchronize(stop));
|
||||
assert_rt(cudaEventElapsedTime(&pointer_time, start, stop));
|
||||
assert_rt(cudaEventDestroy(start));
|
||||
assert_rt(cudaEventDestroy(stop));
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF("arr1[%d] == %d, should be:%d\n", i, arr1[i], i);
|
||||
assert(arr1[i] == static_cast<int>(i));
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE,
|
||||
(annotated_ptr_timing_dev(arr0, arr1);),
|
||||
(assert_rt(cudaDeviceSynchronize()); annotated_ptr_timing<<<blocks, threads>>>(arr0, arr1);
|
||||
assert_rt(cudaDeviceSynchronize());))
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
arr0[i] = static_cast<int>(i);
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(annotated_ptr_timing_dev(arr0, arr1);),
|
||||
(assert_rt(cudaDeviceSynchronize()); assert_rt(cudaEventCreate(&start)); assert_rt(cudaEventCreate(&stop));
|
||||
assert_rt(cudaEventRecord(start));
|
||||
annotated_ptr_timing<<<blocks, threads>>>(arr0, arr1);
|
||||
assert_rt(cudaEventRecord(stop));
|
||||
assert_rt(cudaEventSynchronize(stop));
|
||||
assert_rt(cudaEventElapsedTime(&annotated_time, start, stop));
|
||||
assert_rt(cudaEventDestroy(start));
|
||||
assert_rt(cudaEventDestroy(stop));
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i) {
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF("arr1[%d] == %d, should be:%d\n", i, arr1[i], i);
|
||||
assert(arr1[i] == static_cast<int>(i));
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}))
|
||||
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr0); free(arr1);), (assert_rt(cudaFree(arr0)); assert_rt(cudaFree(arr1));))
|
||||
|
||||
printf("array(ms):%f, arrotated_ptr(ms):%f\n", pointer_time, annotated_time);
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (bench();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
// error: expression must have a constant value annotated_ptr.h: note #2701-D: attempt to access run-time storage
|
||||
// UNSUPPORTED: clang-14, gcc-12, gcc-11, gcc-10, gcc-9, gcc-8, gcc-7, msvc-19.29
|
||||
// UNSUPPORTED: msvc && nvcc-12.0
|
||||
|
||||
#include <cuda/annotated_ptr>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_FUNC constexpr bool test_public_methods()
|
||||
{
|
||||
using namespace cuda;
|
||||
using annotated_ptr = cuda::annotated_ptr<const int, access_property::persisting>;
|
||||
using annotated_smem_ptr [[maybe_unused]] = cuda::annotated_ptr<const int, access_property::shared>;
|
||||
annotated_ptr a{}; // default constructor
|
||||
annotated_ptr b{a}; // copy constructor
|
||||
annotated_ptr c{cuda::std::move(a)}; // move constructor
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (annotated_smem_ptr d{nullptr};)) // pointer constructor
|
||||
b = a; // copy assignment
|
||||
b = cuda::std::move(a); // move assignment
|
||||
auto diff = a - b;
|
||||
auto pred = static_cast<bool>(a);
|
||||
auto prop = a.__property();
|
||||
unused(c);
|
||||
unused(diff);
|
||||
unused(pred);
|
||||
unused(prop);
|
||||
return true;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test_interleave_values()
|
||||
{
|
||||
using namespace cuda;
|
||||
constexpr auto normal = __l2_interleave(__l2_evict_t::_L2_Evict_Unchanged, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
|
||||
constexpr auto streaming = __l2_interleave(__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
|
||||
constexpr auto persisting = __l2_interleave(__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
|
||||
constexpr auto normal_demote =
|
||||
__l2_interleave(__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
|
||||
static_assert(normal == __l2_interleave_normal);
|
||||
static_assert(streaming == __l2_interleave_streaming);
|
||||
static_assert(persisting == __l2_interleave_persisting);
|
||||
static_assert(normal_demote == __l2_interleave_normal_demote);
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
static_assert(test_interleave_values());
|
||||
static_assert(test_public_methods());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_ctor(T* ptr)
|
||||
{
|
||||
// default ctor, cpy and cpy assignment
|
||||
cuda::annotated_ptr<T, P> def;
|
||||
{
|
||||
cuda::annotated_ptr<T, P> temp;
|
||||
temp = def;
|
||||
unused(temp);
|
||||
}
|
||||
cuda::annotated_ptr<T, P> other(def);
|
||||
unused(other);
|
||||
// from ptr
|
||||
cuda::annotated_ptr<T, P> a(ptr);
|
||||
assert(a);
|
||||
|
||||
// cpy ctor & assign to cv
|
||||
cuda::annotated_ptr<const T, P> c(def);
|
||||
cuda::annotated_ptr<volatile T, P> d(def);
|
||||
cuda::annotated_ptr<const volatile T, P> e(def);
|
||||
c = def;
|
||||
d = def;
|
||||
e = def;
|
||||
|
||||
// from c|v to c|v|cv
|
||||
cuda::annotated_ptr<const T, P> f(c);
|
||||
cuda::annotated_ptr<volatile T, P> g(d);
|
||||
cuda::annotated_ptr<const volatile T, P> h(e);
|
||||
f = c;
|
||||
g = d;
|
||||
h = e;
|
||||
unused(f, g, h);
|
||||
|
||||
// to cv
|
||||
cuda::annotated_ptr<const volatile T, P> i(c);
|
||||
cuda::annotated_ptr<const volatile T, P> j(d);
|
||||
i = c;
|
||||
j = d;
|
||||
}
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_global_ctor()
|
||||
{
|
||||
T* rp = nullptr;
|
||||
rp++;
|
||||
test_ctor<T, P>(rp);
|
||||
// from ptr + prop
|
||||
P p;
|
||||
cuda::annotated_ptr<T, cuda::access_property> a(rp, p);
|
||||
cuda::annotated_ptr<const T, cuda::access_property> b(rp, p);
|
||||
cuda::annotated_ptr<volatile T, cuda::access_property> c(rp, p);
|
||||
cuda::annotated_ptr<const volatile T, cuda::access_property> d(rp, p);
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_global_ctors()
|
||||
{
|
||||
test_global_ctor<int, cuda::access_property::normal>();
|
||||
test_global_ctor<int, cuda::access_property::streaming>();
|
||||
test_global_ctor<int, cuda::access_property::persisting>();
|
||||
test_global_ctor<int, cuda::access_property::global>();
|
||||
test_global_ctor<int, cuda::access_property>();
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (__shared__ int smem_value; test_ctor<int, cuda::access_property::shared>(&smem_value);))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_global_ctors();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_ctor()
|
||||
{
|
||||
// default ctor, cpy and cpy assignment
|
||||
cuda::annotated_ptr<T, P> def;
|
||||
def = def;
|
||||
cuda::annotated_ptr<T, P> other(def);
|
||||
|
||||
// from ptr
|
||||
T* rp = nullptr;
|
||||
cuda::annotated_ptr<T, P> a(rp);
|
||||
assert(!a);
|
||||
|
||||
// cpy ctor & assign to cv
|
||||
cuda::annotated_ptr<const T, P> c(def);
|
||||
cuda::annotated_ptr<volatile T, P> d(def);
|
||||
cuda::annotated_ptr<const volatile T, P> e(def);
|
||||
c = e; // FAIL
|
||||
d = d; // FAIL
|
||||
}
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_global_ctor()
|
||||
{
|
||||
test_ctor<T, P>();
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_global_ctors()
|
||||
{
|
||||
test_global_ctor<int, cuda::access_property::normal>();
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_global_ctors();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
cuda::access_property ap(cuda::access_property::persisting{});
|
||||
int* array0 = new int[9];
|
||||
cuda::annotated_ptr<int, cuda::access_property> array_anno_ptr{array0, ap};
|
||||
cuda::annotated_ptr<int, cuda::access_property::shared> shared_ptr;
|
||||
|
||||
array_anno_ptr = shared_ptr; // fail to compile, as expected
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
cuda::access_property ap(cuda::access_property::persisting{});
|
||||
int* array0 = new int[9];
|
||||
cuda::annotated_ptr<int, cuda::access_property> array_anno_ptr{array0, ap};
|
||||
cuda::annotated_ptr<int, cuda::access_property::shared> shared_ptr;
|
||||
|
||||
array_anno_ptr = shared_ptr; // fail to compile, as expected
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
// NVRTC does not do host side testing
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
TEST_FUNC static void fails_from_host()
|
||||
{
|
||||
int a;
|
||||
__nv_associate_access_property(&a, uint64_t{0});
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
// calling from host needs to fail and kill the app
|
||||
fails_from_host();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "test_macros.h"
|
||||
#include "utils.h"
|
||||
|
||||
template <typename T, typename U>
|
||||
TEST_DEVICE_FUNC __noinline__ void shared_mem_test_dev()
|
||||
{
|
||||
T* smem = shared_alloc<T, 128>();
|
||||
smem[10] = 42;
|
||||
|
||||
cuda::annotated_ptr<U, cuda::access_property::shared> p{smem + 10};
|
||||
|
||||
assert(*p == 42);
|
||||
}
|
||||
|
||||
TEST_DEVICE_FUNC __noinline__ void test_all()
|
||||
{
|
||||
shared_mem_test_dev<int, int>();
|
||||
shared_mem_test_dev<int, const int>();
|
||||
shared_mem_test_dev<int, volatile int>();
|
||||
shared_mem_test_dev<int, const volatile int>();
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (test_all();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
constexpr size_t array_size = 128;
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test(P ap)
|
||||
{
|
||||
T* arr = global_alloc<T, array_size>();
|
||||
|
||||
cuda::apply_access_property(arr, array_size * sizeof(T), ap);
|
||||
|
||||
for (size_t i = 0; i < array_size; ++i)
|
||||
{
|
||||
assert(static_cast<size_t>(arr[i]) == i);
|
||||
}
|
||||
|
||||
dealloc<T>(arr);
|
||||
}
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_aligned(P ap)
|
||||
{
|
||||
T* arr = global_alloc<T, array_size>();
|
||||
|
||||
cuda::apply_access_property(arr, cuda::aligned_size_t<sizeof(T)>(array_size * sizeof(T)), ap);
|
||||
|
||||
for (size_t i = 0; i < array_size; ++i)
|
||||
{
|
||||
assert(static_cast<size_t>(arr[i]) == i);
|
||||
}
|
||||
|
||||
dealloc<T>(arr);
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_all()
|
||||
{
|
||||
test<int>(cuda::access_property::normal{});
|
||||
test<int>(cuda::access_property::persisting{});
|
||||
test_aligned<int>(cuda::access_property::normal{});
|
||||
test_aligned<int>(cuda::access_property::persisting{});
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_all();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "utils.h"
|
||||
#define ARR_SZ 128
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test(P ap)
|
||||
{
|
||||
T* arr = global_alloc<T, ARR_SZ>();
|
||||
|
||||
arr = cuda::associate_access_property(arr, ap);
|
||||
|
||||
for (int i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
assert(arr[i] == i);
|
||||
}
|
||||
|
||||
dealloc<T>(arr);
|
||||
}
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_shared(P ap)
|
||||
{
|
||||
T* arr = shared_alloc<T, ARR_SZ>();
|
||||
|
||||
arr = cuda::associate_access_property(arr, ap);
|
||||
|
||||
for (int i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
assert(arr[i] == i);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_all()
|
||||
{
|
||||
test<int>(cuda::access_property::normal{});
|
||||
test<int>(cuda::access_property::persisting{});
|
||||
test<int>(cuda::access_property::streaming{});
|
||||
test<int>(cuda::access_property::global{});
|
||||
test<int>(cuda::access_property{});
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (test_shared<int>(cuda::access_property::shared{});))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_all();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,121 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: pre-sm-70
|
||||
|
||||
#include <cooperative_groups.h>
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
// TODO: global-shared
|
||||
// TODO: read const
|
||||
TEST_FUNC __noinline__ void test_memcpy_async()
|
||||
{
|
||||
size_t ARR_SZ = 1 << 10;
|
||||
int* arr0 = nullptr;
|
||||
int* arr1 = nullptr;
|
||||
cuda::access_property ap(cuda::access_property::persisting{});
|
||||
cuda::barrier<cuda::thread_scope_system> bar0, bar1, bar2, bar3;
|
||||
init(&bar0, 1);
|
||||
init(&bar1, 1);
|
||||
init(&bar2, 1);
|
||||
init(&bar3, 1);
|
||||
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(arr0 = (int*) malloc(ARR_SZ * sizeof(int)); arr1 = (int*) malloc(ARR_SZ * sizeof(int));),
|
||||
(assert_rt(cudaMallocManaged((void**) &arr0, ARR_SZ * sizeof(int)));
|
||||
assert_rt(cudaMallocManaged((void**) &arr1, ARR_SZ * sizeof(int)));
|
||||
assert_rt(cudaDeviceSynchronize());))
|
||||
|
||||
cuda::annotated_ptr<int, cuda::access_property> ann0{arr0, ap};
|
||||
cuda::annotated_ptr<int, cuda::access_property> ann1{arr1, ap};
|
||||
// cuda::annotated_ptr<const int, cuda::access_property> cann0{arr0, ap};
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
arr0[i] = static_cast<int>(i);
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
cuda::memcpy_async(ann1, ann0, ARR_SZ * sizeof(int), bar0);
|
||||
// cuda::memcpy_async(ann1, cann0, ARR_SZ * sizeof(int), bar0);
|
||||
bar0.arrive_and_wait();
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
|
||||
assert(arr1[i] == static_cast<int>(i));
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
cuda::memcpy_async(arr1, ann0, ARR_SZ * sizeof(int), bar1);
|
||||
// cuda::memcpy_async(arr1, cann0, ARR_SZ * sizeof(int), bar1);
|
||||
bar1.arrive_and_wait();
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
|
||||
assert(arr1[i] == static_cast<int>(i));
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(
|
||||
auto group = cooperative_groups::this_thread_block();
|
||||
|
||||
cuda::memcpy_async(group, ann1, ann0, ARR_SZ * sizeof(int), bar2);
|
||||
// cuda::memcpy_async(group, ann1, cann0, ARR_SZ * sizeof(int), bar2);
|
||||
bar2.arrive_and_wait();
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i) {
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
|
||||
assert(arr1[i] == (int) i);
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
cuda::memcpy_async(group, arr1, ann0, ARR_SZ * sizeof(int), bar3);
|
||||
// cuda::memcpy_async(group, arr1, cann0, ARR_SZ * sizeof(int), bar3);
|
||||
bar3.arrive_and_wait();
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i) {
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
|
||||
assert(arr1[i] == (int) i);
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}))
|
||||
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr0); free(arr1);), (assert_rt(cudaFree(arr0)); assert_rt(cudaFree(arr1));))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_memcpy_async();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_DIAG_SUPPRESS_MSVC(4505)
|
||||
|
||||
#include <cuda/annotated_ptr>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include <nv/target>
|
||||
|
||||
#if defined(DEBUG)
|
||||
# define DPRINTF(...) \
|
||||
{ \
|
||||
printf(__VA_ARGS__); \
|
||||
}
|
||||
#else
|
||||
# define DPRINTF(...) \
|
||||
do \
|
||||
{ \
|
||||
} while (false)
|
||||
#endif
|
||||
|
||||
TEST_FUNC void assert_rt_wrap(cudaError_t code, const char* file, int line)
|
||||
{
|
||||
if (code != cudaSuccess)
|
||||
{
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
NV_IF_ELSE_TARGET(NV_IS_HOST,
|
||||
(printf("assert: %s %s %d\n", cudaGetErrorString(code), file, line);),
|
||||
(printf("assert: error=%d %s %d\n", code, file, line);))
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
assert(code == cudaSuccess);
|
||||
}
|
||||
}
|
||||
#define assert_rt(ret) \
|
||||
{ \
|
||||
assert_rt_wrap((ret), __FILE__, __LINE__); \
|
||||
}
|
||||
|
||||
template <typename T, int N>
|
||||
TEST_FUNC __noinline__ T* global_alloc()
|
||||
{
|
||||
T* arr = nullptr;
|
||||
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_DEVICE, (arr = (T*) malloc(N * sizeof(T));), (assert_rt(cudaMallocManaged((void**) &arr, N * sizeof(T)));))
|
||||
|
||||
for (int i = 0; i < N; ++i)
|
||||
{
|
||||
arr[i] = i;
|
||||
}
|
||||
return arr;
|
||||
}
|
||||
|
||||
template <typename T, int N>
|
||||
TEST_DEVICE_FUNC __noinline__ T* shared_alloc()
|
||||
{
|
||||
__shared__ T data[N];
|
||||
|
||||
for (int i = 0; i < N; ++i)
|
||||
{
|
||||
data[i] = i;
|
||||
}
|
||||
return data;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC __noinline__ void dealloc(T* arr)
|
||||
{
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr);), assert_rt(cudaFree(arr));)
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
struct minimal_comparable_value
|
||||
{
|
||||
int value;
|
||||
};
|
||||
|
||||
TEST_FUNC constexpr bool operator<(minimal_comparable_value lhs, minimal_comparable_value rhs)
|
||||
{
|
||||
return lhs.value < rhs.value;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool operator==(minimal_comparable_value lhs, minimal_comparable_value rhs)
|
||||
{
|
||||
return lhs.value == rhs.value;
|
||||
}
|
||||
|
||||
namespace cuda::std
|
||||
{
|
||||
template <>
|
||||
class numeric_limits<minimal_comparable_value>
|
||||
{
|
||||
public:
|
||||
static constexpr bool is_specialized = true;
|
||||
|
||||
TEST_FUNC static constexpr minimal_comparable_value lowest() noexcept
|
||||
{
|
||||
return {0};
|
||||
}
|
||||
|
||||
TEST_FUNC static constexpr minimal_comparable_value max() noexcept
|
||||
{
|
||||
return {100};
|
||||
}
|
||||
};
|
||||
} // namespace cuda::std
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// --- static_bounds ---
|
||||
|
||||
// Basic static bounds
|
||||
{
|
||||
constexpr auto b = cuda::args::static_bounds<1, 4096>{};
|
||||
static_assert(b.lower() == 1);
|
||||
static_assert(b.upper() == 4096);
|
||||
}
|
||||
|
||||
// Exact static bounds
|
||||
{
|
||||
constexpr auto b = cuda::args::static_bounds<42, 42>{};
|
||||
static_assert(b.lower() == 42);
|
||||
static_assert(b.upper() == 42);
|
||||
}
|
||||
|
||||
// Long type deduced from NTTPs
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<decltype(cuda::args::static_bounds<0L, 1000L>::lower()), long>);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Static bounds preserve their original NTTP types
|
||||
{
|
||||
constexpr auto b = cuda::args::bounds<1.0f, 8.0f>();
|
||||
static_assert(b.lower() == 1.0f);
|
||||
static_assert(b.upper() == 8);
|
||||
static_assert(cuda::std::is_same_v<decltype(b.lower()), float>);
|
||||
static_assert(cuda::std::is_same_v<decltype(b.upper()), float>);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// --- runtime_bounds ---
|
||||
|
||||
// Basic runtime bounds
|
||||
{
|
||||
auto b = cuda::args::runtime_bounds{10, 100};
|
||||
assert(b.lower() == 10);
|
||||
assert(b.upper() == 100);
|
||||
static_assert(cuda::std::is_same_v<decltype(b.lower()), int>);
|
||||
}
|
||||
|
||||
// Default runtime bounds span the element type's numeric_limits range
|
||||
{
|
||||
constexpr cuda::args::runtime_bounds<int> b{};
|
||||
static_assert(b.lower() == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(b.upper() == (cuda::std::numeric_limits<int>::max)());
|
||||
}
|
||||
|
||||
// --- argument_bounds factory functions ---
|
||||
|
||||
// Static via factory
|
||||
{
|
||||
constexpr auto b = cuda::args::bounds<1, 8>();
|
||||
static_assert(b.lower() == 1);
|
||||
static_assert(b.upper() == 8);
|
||||
static_assert(cuda::args::__is_static_bounds_cv_v<decltype(b)>);
|
||||
static_assert(!cuda::args::__is_runtime_bounds_cv_v<decltype(b)>);
|
||||
static_assert(cuda::args::__is_bounds_v<decltype(b)>);
|
||||
}
|
||||
|
||||
// Runtime via factory
|
||||
{
|
||||
auto b = cuda::args::bounds(10, 100);
|
||||
assert(b.lower() == 10);
|
||||
assert(b.upper() == 100);
|
||||
static_assert(!cuda::args::__is_static_bounds_cv_v<decltype(b)>);
|
||||
static_assert(cuda::args::__is_runtime_bounds_cv_v<decltype(b)>);
|
||||
static_assert(cuda::args::__is_bounds_v<decltype(b)>);
|
||||
}
|
||||
|
||||
// Runtime bounds only require operator< and operator==.
|
||||
{
|
||||
constexpr auto b = cuda::args::bounds(minimal_comparable_value{10}, minimal_comparable_value{20});
|
||||
static_assert(b.lower() == minimal_comparable_value{10});
|
||||
static_assert(b.upper() == minimal_comparable_value{20});
|
||||
}
|
||||
|
||||
// Static and runtime bounds intersection
|
||||
{
|
||||
static_assert(cuda::args::__has_bounds_intersection<int, cuda::args::static_bounds<1, 100>>(
|
||||
cuda::args::runtime_bounds<int>{50, 200}));
|
||||
static_assert(!cuda::args::__has_bounds_intersection<int, cuda::args::static_bounds<100, 200>>(
|
||||
cuda::args::runtime_bounds<int>{0, 50}));
|
||||
}
|
||||
|
||||
// Runtime bounds validation with no static bounds only requires operator< and operator==.
|
||||
{
|
||||
minimal_comparable_value values[] = {{10}, {20}};
|
||||
[[maybe_unused]] auto arg = cuda::args::deferred_sequence{
|
||||
cuda::std::span<minimal_comparable_value>{values, 2},
|
||||
cuda::args::bounds(minimal_comparable_value{5}, minimal_comparable_value{50})};
|
||||
}
|
||||
|
||||
// Unsigned no-bounds arguments must not instantiate a pointless `value < 0` comparison.
|
||||
{
|
||||
unsigned int value = 0;
|
||||
[[maybe_unused]] auto arg = cuda::args::deferred{&value};
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Static/runtime bounds intersection only requires operator< and operator==.
|
||||
{
|
||||
using static_bounds_t = cuda::args::static_bounds<minimal_comparable_value{10}, minimal_comparable_value{50}>;
|
||||
|
||||
constexpr auto runtime_bounds = cuda::args::bounds(minimal_comparable_value{20}, minimal_comparable_value{40});
|
||||
static_assert(cuda::args::__has_bounds_intersection<minimal_comparable_value, static_bounds_t>(runtime_bounds));
|
||||
static_assert(!cuda::args::__has_bounds_intersection<minimal_comparable_value, static_bounds_t>(
|
||||
cuda::args::bounds(minimal_comparable_value{60}, minimal_comparable_value{70})));
|
||||
|
||||
cuda::args::__validate_static_element_bounds<minimal_comparable_value, static_bounds_t>(
|
||||
minimal_comparable_value{30});
|
||||
cuda::args::__validate_runtime_element_bounds(minimal_comparable_value{30}, runtime_bounds);
|
||||
|
||||
minimal_comparable_value values[] = {{20}, {30}};
|
||||
[[maybe_unused]] auto arg = cuda::args::__immediate_sequence{
|
||||
cuda::std::span<minimal_comparable_value>{values, 2}, static_bounds_t{}, runtime_bounds};
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// Non-bounds type
|
||||
{
|
||||
static_assert(!cuda::args::__is_bounds_v<int>);
|
||||
}
|
||||
|
||||
// Bounds types accepted by argument wrapper template parameters
|
||||
{
|
||||
static_assert(cuda::args::__valid_static_bounds_v<int, cuda::args::no_bounds>);
|
||||
static_assert(cuda::args::__valid_static_bounds_v<int, cuda::args::static_bounds<1, 8>>);
|
||||
static_assert(!cuda::args::__valid_static_bounds_v<int, cuda::args::runtime_bounds<int>>);
|
||||
static_assert(!cuda::args::__valid_static_bounds_v<int, int>);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/complex>
|
||||
#include <cuda/std/expected>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/mdspan>
|
||||
#include <cuda/std/optional>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/tuple>
|
||||
#include <cuda/std/type_traits>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include "test_iterators.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
enum class color
|
||||
{
|
||||
red,
|
||||
green,
|
||||
blue
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
struct element_type_like
|
||||
{
|
||||
using element_type = _Tp;
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
struct range_like
|
||||
{
|
||||
using iterator = _Tp*;
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
struct value_type_like
|
||||
{
|
||||
using value_type = _Tp;
|
||||
};
|
||||
|
||||
struct non_sequence_value
|
||||
{};
|
||||
|
||||
TEST_FUNC void test()
|
||||
{
|
||||
// --- __is_sequence_v ---
|
||||
|
||||
// builtin and class type are not sequences
|
||||
static_assert(!cuda::args::__is_sequence_v<int>);
|
||||
static_assert(!cuda::args::__is_sequence_v<color>);
|
||||
static_assert(!cuda::args::__is_sequence_v<non_sequence_value>);
|
||||
static_assert(!cuda::args::__is_sequence_v<range_like<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<element_type_like<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<value_type_like<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::complex<float>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::pair<float, int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::tuple<float, int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::optional<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::expected<int, int>>);
|
||||
|
||||
// iterators and pointers can be sequences if they are at least random access
|
||||
static_assert(cuda::args::__is_sequence_v<int*>);
|
||||
static_assert(cuda::args::__is_sequence_v<const int*>);
|
||||
static_assert(cuda::args::__is_sequence_v<cuda::counting_iterator<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<bidirectional_iterator<int*>>);
|
||||
|
||||
// ranges and arrays are sequences
|
||||
static_assert(cuda::args::__is_sequence_v<int[]>);
|
||||
static_assert(cuda::args::__is_sequence_v<const int[]>);
|
||||
static_assert(cuda::args::__is_sequence_v<int[42]>);
|
||||
static_assert(cuda::args::__is_sequence_v<const int[42]>);
|
||||
static_assert(cuda::args::__is_sequence_v<cuda::std::span<int, 1>>);
|
||||
static_assert(cuda::args::__is_sequence_v<const cuda::std::span<int, 1>&>);
|
||||
static_assert(cuda::args::__is_sequence_v<cuda::std::span<int>>);
|
||||
static_assert(cuda::args::__is_sequence_v<cuda::std::array<int, 3>>);
|
||||
|
||||
// --- __element_type_of_t ---
|
||||
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<const cuda::std::span<int, 1>&>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<int*>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::counting_iterator<int>>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::std::array<int, 3>>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<range_like<int>>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<element_type_like<int>>, int>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::std::mdspan<const int, cuda::std::extents<int, 1>>>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<value_type_like<int>>, int>);
|
||||
|
||||
// --- argument_traits: is_deferred ---
|
||||
|
||||
static_assert(!cuda::args::__traits<int>::is_deferred);
|
||||
static_assert(!cuda::args::__traits<cuda::args::immediate<int>>::is_deferred);
|
||||
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_deferred);
|
||||
static_assert(!cuda::args::__traits<cuda::args::constant<42>>::is_deferred);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_deferred);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
static_assert(cuda::args::__traits<cuda::args::deferred<cuda::std::span<int, 1>>>::is_deferred);
|
||||
static_assert(cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>::is_deferred);
|
||||
|
||||
// --- argument_traits: is_single_value ---
|
||||
|
||||
static_assert(cuda::args::__traits<int>::is_single_value);
|
||||
static_assert(cuda::args::__traits<int*>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<int>>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<int*>>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<cuda::counting_iterator<int>>>::is_single_value);
|
||||
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::constant<42>>::is_single_value);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(
|
||||
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
static_assert(cuda::args::__traits<cuda::args::deferred<int*>>::is_single_value);
|
||||
static_assert(!cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>::is_single_value);
|
||||
|
||||
// --- argument_traits: value_type ---
|
||||
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__traits<int>::value_type, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::immediate<int>>::value_type, int>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::value_type,
|
||||
cuda::std::span<int>>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::constant<42>>::value_type, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::constant<10, float>>::value_type, float>);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(cuda::std::is_same_v<
|
||||
cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::value_type,
|
||||
cuda::std::array<int, 3>>);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// --- argument_traits: lowest / highest ---
|
||||
|
||||
static_assert(cuda::args::__traits<int>::lowest == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__traits<int>::highest == (cuda::std::numeric_limits<int>::max)());
|
||||
static_assert(cuda::args::__traits<const int>::lowest == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__traits<int&>::highest == (cuda::std::numeric_limits<int>::max)());
|
||||
static_assert(cuda::args::__traits<float>::lowest == cuda::std::numeric_limits<float>::lowest());
|
||||
static_assert(cuda::args::__traits<float>::highest == (cuda::std::numeric_limits<float>::max)());
|
||||
static_assert(cuda::args::__traits<const cuda::args::immediate<int, cuda::args::static_bounds<1, 8>>>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<int, cuda::args::static_bounds<1, 8>>&>::highest == 8);
|
||||
static_assert(
|
||||
cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>, cuda::args::static_bounds<1, 8>>>::highest
|
||||
== 8);
|
||||
static_assert(cuda::args::__traits<cuda::args::constant<10, float>>::lowest == 10.0f);
|
||||
static_assert(cuda::args::__traits<cuda::args::constant<10, float>>::highest == 10.0f);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{3, 1, 2}>>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{3, 1, 2}>>::highest == 3);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// --- Free function bounds on plain values ---
|
||||
|
||||
static_assert(cuda::args::__lowest_(42) == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__highest_(42) == (cuda::std::numeric_limits<int>::max)());
|
||||
static_assert(cuda::args::__lowest_(1.0f) == cuda::std::numeric_limits<float>::lowest());
|
||||
static_assert(cuda::args::__highest_(1.0f) == (cuda::std::numeric_limits<float>::max)());
|
||||
|
||||
// --- Scalar and sequence wrappers expose distinct single-value traits ---
|
||||
|
||||
static_assert(cuda::args::__traits<cuda::args::constant<42>>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<int>>::is_single_value);
|
||||
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(
|
||||
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,174 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// Deferred single value via span<T, 1>
|
||||
{
|
||||
int val = 42;
|
||||
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}};
|
||||
assert(cuda::args::__unwrap(def)[0] == 42);
|
||||
assert(cuda::args::__access::__arg(def)[0] == 42);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == (cuda::std::numeric_limits<int>::max)());
|
||||
}
|
||||
|
||||
// Deferred single value with static bounds
|
||||
{
|
||||
int val = 42;
|
||||
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds<1, 1000>()};
|
||||
assert(cuda::args::__unwrap(def)[0] == 42);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 1000);
|
||||
}
|
||||
|
||||
// Deferred single value via pointer
|
||||
{
|
||||
int val = 42;
|
||||
using def_t = cuda::args::deferred<int*, cuda::args::static_bounds<0, 100>>;
|
||||
static_assert(cuda::args::__traits<def_t>::lowest == 0);
|
||||
static_assert(cuda::args::__traits<def_t>::highest == 100);
|
||||
// Also verify construction works
|
||||
auto def = cuda::args::deferred{&val, cuda::args::bounds<0, 100>()};
|
||||
assert(cuda::args::__unwrap(def) == &val);
|
||||
}
|
||||
|
||||
// Deferred single value via fancy iterator
|
||||
{
|
||||
auto it = cuda::counting_iterator<int>{42};
|
||||
auto def = cuda::args::deferred{it, cuda::args::bounds<0, 100>()};
|
||||
assert(cuda::args::__unwrap(def)[0] == 42);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 0);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 100);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::is_single_value);
|
||||
}
|
||||
|
||||
// Deferred single value with both bounds, runtime bounds first
|
||||
{
|
||||
int val = 42;
|
||||
auto def =
|
||||
cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds(5, 100), cuda::args::bounds<1, 256>()};
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 256);
|
||||
assert(cuda::args::__access::__runtime_bounds(def).lower() == 5);
|
||||
assert(cuda::args::__access::__runtime_bounds(def).upper() == 100);
|
||||
assert(cuda::args::__lowest_(def) == 5);
|
||||
assert(cuda::args::__highest_(def) == 100);
|
||||
cuda::args::__access::__runtime_bounds(def) = cuda::args::bounds(5, 90);
|
||||
assert(cuda::args::__highest_(def) == 90);
|
||||
}
|
||||
|
||||
// Deferred sequence via fancy iterator
|
||||
{
|
||||
auto it = cuda::counting_iterator<int>{10};
|
||||
auto def = cuda::args::deferred_sequence{it, cuda::args::bounds<0, 100>()};
|
||||
assert(cuda::args::__unwrap(def)[0] == 10);
|
||||
assert(cuda::args::__unwrap(def)[2] == 12);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 0);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 100);
|
||||
static_assert(!cuda::args::__traits<decltype(def)>::is_single_value);
|
||||
}
|
||||
|
||||
// Deferred sequence with both bounds
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto def = cuda::args::deferred_sequence{
|
||||
cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 4096>(), cuda::args::bounds(5, 100)};
|
||||
assert(cuda::args::__access::__arg(def).size() == 4);
|
||||
assert(cuda::args::__access::__runtime_bounds(def).lower() == 5);
|
||||
assert(cuda::args::__access::__runtime_bounds(def).upper() == 100);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
|
||||
assert(cuda::args::__lowest_(def) == 5);
|
||||
assert(cuda::args::__highest_(def) == 100);
|
||||
}
|
||||
|
||||
// Deferred sequence with both bounds, runtime bounds first
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto def = cuda::args::deferred_sequence{
|
||||
cuda::std::span<int>{arr, 4}, cuda::args::bounds(5, 100), cuda::args::bounds<1, 4096>()};
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 4096);
|
||||
assert(cuda::args::__lowest_(def) == 5);
|
||||
assert(cuda::args::__highest_(def) == 100);
|
||||
}
|
||||
|
||||
// Traits: deferred is single value
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::deferred<cuda::std::span<int, 1>>>;
|
||||
static_assert(traits::is_deferred);
|
||||
static_assert(traits::is_single_value);
|
||||
}
|
||||
|
||||
// Traits: deferred with pointer is also single value
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::deferred<int*>>;
|
||||
static_assert(traits::is_deferred);
|
||||
static_assert(traits::is_single_value);
|
||||
}
|
||||
|
||||
// Traits: deferred_sequence is not single value
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>;
|
||||
static_assert(traits::is_deferred);
|
||||
static_assert(!traits::is_single_value);
|
||||
}
|
||||
|
||||
// Unwrap: deferred
|
||||
{
|
||||
int val = 99;
|
||||
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}};
|
||||
auto& v = cuda::args::__unwrap(def);
|
||||
assert(v[0] == 99);
|
||||
}
|
||||
|
||||
// Unwrap: deferred_sequence
|
||||
{
|
||||
int arr[3] = {10, 20, 30};
|
||||
auto def = cuda::args::deferred_sequence{cuda::std::span<int>{arr, 3}};
|
||||
const auto& v = cuda::args::__unwrap(def);
|
||||
assert(v.size() == 3);
|
||||
assert(v[1] == 20);
|
||||
}
|
||||
|
||||
// Unwrap: rvalue deferred returns by value
|
||||
{
|
||||
int val = 99;
|
||||
auto v = cuda::args::__unwrap(cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}});
|
||||
assert(v[0] == 99);
|
||||
}
|
||||
|
||||
// Unwrap: rvalue deferred_sequence returns by value
|
||||
{
|
||||
int arr[3] = {10, 20, 30};
|
||||
auto v = cuda::args::__unwrap(cuda::args::deferred_sequence{cuda::std::span<int>{arr, 3}});
|
||||
assert(v.size() == 3);
|
||||
assert(v[2] == 30);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
[[maybe_unused]] cuda::args::deferred_sequence<int> invalid_arg{0};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
using traits = cuda::args::__traits<cuda::args::deferred_sequence<int>>;
|
||||
|
||||
[[maybe_unused]] constexpr bool invalid_traits = traits::is_deferred;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
struct non_sequence_value
|
||||
{
|
||||
int payload;
|
||||
};
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// Uniform scalar via CTAD
|
||||
{
|
||||
auto da = cuda::args::immediate{5};
|
||||
assert(cuda::args::__unwrap(da) == 5);
|
||||
assert(cuda::args::__access::__arg(da) == 5);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == (cuda::std::numeric_limits<int>::max)());
|
||||
assert(cuda::args::__lowest_(da) == 5);
|
||||
assert(cuda::args::__highest_(da) == 5);
|
||||
cuda::args::__access::__arg(da) = 6;
|
||||
assert(cuda::args::__unwrap(da) == 6);
|
||||
}
|
||||
|
||||
// Uniform scalar with static bounds
|
||||
{
|
||||
auto da = cuda::args::immediate{5, cuda::args::bounds<1, 8>()};
|
||||
assert(cuda::args::__unwrap(da) == 5);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == 8);
|
||||
assert(cuda::args::__lowest_(da) == 5);
|
||||
assert(cuda::args::__highest_(da) == 5);
|
||||
}
|
||||
|
||||
// Non-sequence values are accepted without scalar-only restrictions
|
||||
{
|
||||
auto da = cuda::args::immediate{non_sequence_value{7}};
|
||||
assert(cuda::args::__unwrap(da).payload == 7);
|
||||
}
|
||||
|
||||
// Pointer-like types can still represent a single value when explicitly wrapped that way
|
||||
{
|
||||
int value = 11;
|
||||
auto da = cuda::args::immediate{&value};
|
||||
static_assert(cuda::args::__traits<decltype(da)>::is_single_value);
|
||||
assert(*cuda::args::__unwrap(da) == 11);
|
||||
}
|
||||
|
||||
// Per-segment span with runtime bounds
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}, cuda::args::bounds(1L, 100L)};
|
||||
assert(cuda::args::__unwrap(da).size() == 4);
|
||||
assert(cuda::args::__access::__arg(da).size() == 4);
|
||||
assert(cuda::args::__access::__runtime_bounds(da).lower() == 1);
|
||||
assert(cuda::args::__access::__runtime_bounds(da).upper() == 100);
|
||||
assert(cuda::args::__lowest_(da) == 1);
|
||||
assert(cuda::args::__highest_(da) == 100);
|
||||
cuda::args::__access::__runtime_bounds(da) = cuda::args::bounds(1, 90);
|
||||
assert(cuda::args::__highest_(da) == 90);
|
||||
}
|
||||
|
||||
// Per-segment span with both bounds
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto da = cuda::args::__immediate_sequence{
|
||||
cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 256>(), cuda::args::bounds(10, 200)};
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == 256);
|
||||
assert(cuda::args::__lowest_(da) == 10);
|
||||
assert(cuda::args::__highest_(da) == 200);
|
||||
}
|
||||
|
||||
// Per-segment span with both bounds, runtime bounds first
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto da = cuda::args::__immediate_sequence{
|
||||
cuda::std::span<int>{arr, 4}, cuda::args::bounds(10, 200), cuda::args::bounds<1, 256>()};
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == 256);
|
||||
assert(cuda::args::__lowest_(da) == 10);
|
||||
assert(cuda::args::__highest_(da) == 200);
|
||||
}
|
||||
|
||||
// Per-segment via span
|
||||
{
|
||||
int arr[4] = {1, 2, 3, 4};
|
||||
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}};
|
||||
assert(cuda::args::__unwrap(da).size() == 4);
|
||||
assert(cuda::args::__unwrap(da)[0] == 1);
|
||||
assert(cuda::args::__unwrap(da)[3] == 4);
|
||||
}
|
||||
|
||||
// Per-segment with static bounds
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 100>()};
|
||||
assert(cuda::args::__unwrap(da).size() == 4);
|
||||
assert(cuda::args::__unwrap(da)[2] == 30);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == 100);
|
||||
}
|
||||
|
||||
// Traits
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::immediate<int>>;
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, int>);
|
||||
}
|
||||
|
||||
// Sequence traits
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>;
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(!traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, cuda::std::span<int>>);
|
||||
}
|
||||
|
||||
// __is_sequence_v on unwrapped types
|
||||
{
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::args::__traits<cuda::args::immediate<int>>::value_type>);
|
||||
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
|
||||
}
|
||||
|
||||
// Unwrap: scalar
|
||||
{
|
||||
auto da = cuda::args::immediate{7};
|
||||
auto& v = cuda::args::__unwrap(da);
|
||||
assert(v == 7);
|
||||
v = 8;
|
||||
assert(cuda::args::__unwrap(da) == 8);
|
||||
}
|
||||
|
||||
// Unwrap: span
|
||||
{
|
||||
int arr[3] = {10, 20, 30};
|
||||
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 3}};
|
||||
const auto& v = cuda::args::__unwrap(da);
|
||||
assert(v.size() == 3);
|
||||
assert(v[1] == 20);
|
||||
}
|
||||
|
||||
// Unwrap: rvalue scalar returns by value
|
||||
{
|
||||
const auto& v = cuda::args::__unwrap(cuda::args::immediate{7});
|
||||
assert(v == 7);
|
||||
}
|
||||
|
||||
// Unwrap: rvalue span returns by value
|
||||
{
|
||||
int arr[3] = {10, 20, 30};
|
||||
auto v = cuda::args::__unwrap(cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 3}});
|
||||
assert(v.size() == 3);
|
||||
assert(v[2] == 30);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
// A type without a cuda::std::numeric_limits specialization has no meaningful implicit bounds. Default-constructing
|
||||
// runtime_bounds for such a type must be rejected at compile time instead of silently producing a degenerate range.
|
||||
struct unspecialized_type
|
||||
{};
|
||||
|
||||
[[maybe_unused]] cuda::args::runtime_bounds<unspecialized_type> invalid_bounds{};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,197 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
struct non_sequence_value
|
||||
{
|
||||
int payload;
|
||||
};
|
||||
|
||||
enum class dependent_direction
|
||||
{
|
||||
min,
|
||||
max
|
||||
};
|
||||
|
||||
template <dependent_direction Value>
|
||||
struct dependent_direction_tag
|
||||
{
|
||||
static constexpr auto value = Value;
|
||||
};
|
||||
|
||||
template <class Tag>
|
||||
TEST_FUNC void test_dependent_constant_type()
|
||||
{
|
||||
constexpr auto direction = Tag::value;
|
||||
using constant_t = cuda::args::constant<direction>;
|
||||
|
||||
// Regression: NVCC bug generated a host stub using a cv/ref-qualified constant type while device registration used
|
||||
// the unqualified type, causing cudaErrorInvalidDeviceFunction when launching the kernel.
|
||||
static_assert(cuda::std::is_same_v<typename constant_t::value_type, dependent_direction>);
|
||||
static_assert(cuda::std::is_same_v<constant_t, cuda::args::constant<Tag::value, dependent_direction>>);
|
||||
}
|
||||
|
||||
TEST_FUNC void test()
|
||||
{
|
||||
// Basic value
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<42>{};
|
||||
static_assert(cuda::args::__unwrap(sa) == 42);
|
||||
static_assert(cuda::std::is_same_v<decltype(sa)::value_type, int>);
|
||||
}
|
||||
|
||||
// Different types
|
||||
{
|
||||
constexpr auto sa_long = cuda::args::constant<100L>{};
|
||||
static_assert(cuda::args::__unwrap(sa_long) == 100L);
|
||||
static_assert(cuda::std::is_same_v<decltype(sa_long)::value_type, long>);
|
||||
|
||||
constexpr auto sa_float = cuda::args::constant<10, float>{};
|
||||
static_assert(cuda::args::__unwrap(sa_float) == 10.0f);
|
||||
static_assert(cuda::std::is_same_v<decltype(sa_float)::value_type, float>);
|
||||
static_assert(cuda::std::is_same_v<decltype(cuda::args::__unwrap(sa_float)), float>);
|
||||
}
|
||||
|
||||
// Negative value
|
||||
{
|
||||
constexpr auto sa_neg = cuda::args::constant<-1>{};
|
||||
static_assert(cuda::args::__unwrap(sa_neg) == -1);
|
||||
}
|
||||
|
||||
// Dependent value
|
||||
{
|
||||
test_dependent_constant_type<dependent_direction_tag<dependent_direction::max>>();
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Non-sequence values are accepted without scalar-only restrictions
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<non_sequence_value{7}>{};
|
||||
static_assert(cuda::args::__unwrap(sa).payload == 7);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Array sequence
|
||||
{
|
||||
constexpr auto sa_arr = cuda::args::__constant_sequence<cuda::std::array<int, 3>{128, 256, 512}>{};
|
||||
static_assert(cuda::args::__unwrap(sa_arr)[0] == 128);
|
||||
static_assert(cuda::args::__unwrap(sa_arr)[1] == 256);
|
||||
static_assert(cuda::args::__unwrap(sa_arr)[2] == 512);
|
||||
static_assert(cuda::std::is_same_v<decltype(sa_arr)::value_type, cuda::std::array<int, 3>>);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// Bounds: scalar
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<42>{};
|
||||
static_assert(cuda::args::__lowest_(sa) == 42);
|
||||
static_assert(cuda::args::__highest_(sa) == 42);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Bounds: array sequence computes lowest/highest of elements
|
||||
{
|
||||
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 3>{128, 256, 512}>{};
|
||||
static_assert(cuda::args::__lowest_(sa) == 128);
|
||||
static_assert(cuda::args::__highest_(sa) == 512);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Bounds: empty array sequence has unconstrained element bounds
|
||||
{
|
||||
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 0>{}>{};
|
||||
static_assert(cuda::args::__lowest_(sa) == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__highest_(sa) == (cuda::std::numeric_limits<int>::max)());
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// Traits
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::constant<42>>;
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(traits::is_constant);
|
||||
static_assert(traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, int>);
|
||||
static_assert(traits::lowest == 42);
|
||||
static_assert(traits::highest == 42);
|
||||
}
|
||||
|
||||
// Traits: explicit constant value type
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::constant<10, float>>;
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(traits::is_constant);
|
||||
static_assert(traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, float>);
|
||||
static_assert(cuda::std::is_same_v<traits::element_type, float>);
|
||||
static_assert(traits::lowest == 10.0f);
|
||||
static_assert(traits::highest == 10.0f);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Sequence traits
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>;
|
||||
static_assert(traits::is_constant);
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(!traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, cuda::std::array<int, 3>>);
|
||||
static_assert(cuda::std::is_same_v<traits::element_type, int>);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// Single value: scalar is single, sequence is not
|
||||
{
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::args::__traits<cuda::args::constant<42>>::value_type>);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(
|
||||
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
}
|
||||
|
||||
// Unwrap: scalar
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<42>{};
|
||||
constexpr auto val = cuda::args::__unwrap(sa);
|
||||
static_assert(val == 42);
|
||||
}
|
||||
|
||||
// Unwrap: scalar with explicit value type
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<10, float>{};
|
||||
constexpr auto val = cuda::args::__unwrap(sa);
|
||||
static_assert(val == 10.0f);
|
||||
static_assert(cuda::std::is_same_v<decltype(val), const float>);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Unwrap: sequence
|
||||
{
|
||||
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 3>{10, 20, 30}>{};
|
||||
constexpr auto val = cuda::args::__unwrap(sa);
|
||||
static_assert(val[0] == 10);
|
||||
static_assert(val[2] == 30);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
using arg_t = cuda::args::immediate<int, cuda::args::runtime_bounds<int>>;
|
||||
|
||||
[[maybe_unused]] arg_t invalid_arg{0};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
using arg_t = cuda::args::immediate<unsigned char, cuda::args::static_bounds<0, 1000>>;
|
||||
|
||||
[[maybe_unused]] constexpr auto invalid_highest = cuda::args::__traits<arg_t>::highest;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
[[maybe_unused]] constexpr auto invalid_bounds = cuda::args::static_bounds<0, 1L>{};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
// Reading the implicit bounds of __traits for an element type without a cuda::std::numeric_limits specialization must
|
||||
// fail to compile rather than silently yielding a value-initialized (and therefore meaningless) bound. This exercises
|
||||
// the __traits_impl primary-template path, which is the bound surface read by generic consumers.
|
||||
struct unspecialized_type
|
||||
{};
|
||||
|
||||
[[maybe_unused]] constexpr auto invalid_lowest = cuda::args::__traits<unspecialized_type>::lowest;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,214 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// Integration test: demonstrates how an algorithm consumes argument wrappers
|
||||
// to make compile-time and runtime resource decisions.
|
||||
// All argument types (plain values, constants, immediate values, deferred values) work uniformly
|
||||
// through the free functions.
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/std/algorithm>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/span>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
constexpr int shared_memory_capacity = 256;
|
||||
constexpr int default_max_segment_size = 1024;
|
||||
|
||||
enum class algorithm_variant
|
||||
{
|
||||
shared_memory,
|
||||
global_memory
|
||||
};
|
||||
|
||||
// Static scaling: choose algorithm variant at compile time.
|
||||
template <class _SegSizeArg>
|
||||
TEST_FUNC constexpr algorithm_variant select_variant(_SegSizeArg)
|
||||
{
|
||||
if constexpr (cuda::args::__traits<_SegSizeArg>::highest <= shared_memory_capacity)
|
||||
{
|
||||
return algorithm_variant::shared_memory;
|
||||
}
|
||||
else
|
||||
{
|
||||
return algorithm_variant::global_memory;
|
||||
}
|
||||
}
|
||||
|
||||
// Dynamic scaling: compute buffer size at runtime, clamped to default.
|
||||
template <class _SegSizeArg>
|
||||
TEST_FUNC constexpr int compute_buffer_size(_SegSizeArg __seg_size, int __num_segments)
|
||||
{
|
||||
auto __highest = cuda::std::min(default_max_segment_size, static_cast<int>(cuda::args::__highest_(__seg_size)));
|
||||
return __highest * __num_segments;
|
||||
}
|
||||
|
||||
// Process: use the actual unwrapped value.
|
||||
template <class _SegSizeArg>
|
||||
TEST_FUNC constexpr int process_segments(_SegSizeArg __seg_size)
|
||||
{
|
||||
const auto& __val = cuda::args::__unwrap(__seg_size);
|
||||
|
||||
if constexpr (cuda::args::__traits<_SegSizeArg>::is_single_value)
|
||||
{
|
||||
return static_cast<int>(__val);
|
||||
}
|
||||
else
|
||||
{
|
||||
int __total = 0;
|
||||
for (size_t __i = 0; __i < __val.size(); ++__i)
|
||||
{
|
||||
__total += static_cast<int>(__val[__i]);
|
||||
}
|
||||
return __total;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// Plain scalar: no bounds, global memory, buffer clamped to default
|
||||
{
|
||||
static_assert(select_variant(100) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(100, 4) == default_max_segment_size * 4);
|
||||
assert(process_segments(100) == 100);
|
||||
}
|
||||
|
||||
#if 0 // FIXME(miscco): This should not work
|
||||
// Plain span: per-segment, no bounds, global memory
|
||||
{
|
||||
int sizes[3] = {64, 128, 96};
|
||||
auto seg = cuda::std::span<int>{sizes, 3};
|
||||
assert(select_variant(seg) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg, 3) == default_max_segment_size * 3);
|
||||
assert(process_segments(seg) == 64 + 128 + 96);
|
||||
}
|
||||
#endif
|
||||
|
||||
// constant: scalar, fits in shared memory, buffer = value
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::constant<128>{};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 128 * 4);
|
||||
assert(process_segments(seg_size) == 128);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// __constant_sequence: array sequence, highest fits in shared memory
|
||||
{
|
||||
constexpr auto seg_sizes = cuda::args::__constant_sequence<cuda::std::array{64, 128, 256}>{};
|
||||
static_assert(select_variant(seg_sizes) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_sizes, 3) == 256 * 3);
|
||||
assert(process_segments(seg_sizes) == 64 + 128 + 256);
|
||||
}
|
||||
|
||||
// __constant_sequence: array sequence, highest exceeds shared memory, buffer clamped
|
||||
{
|
||||
constexpr auto seg_sizes = cuda::args::__constant_sequence<cuda::std::array{64, 128, 512}>{};
|
||||
static_assert(select_variant(seg_sizes) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg_sizes, 3) == 512 * 3);
|
||||
assert(process_segments(seg_sizes) == 64 + 128 + 512);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// immediate: tight static bounds, shared memory, buffer = value
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::immediate{100, cuda::args::bounds<1, 256>()};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
|
||||
assert(process_segments(seg_size) == 100);
|
||||
}
|
||||
|
||||
// immediate: wide static bounds, global memory, buffer = value
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::immediate{100, cuda::args::bounds<1, 4096>()};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
|
||||
assert(process_segments(seg_size) == 100);
|
||||
}
|
||||
|
||||
// immediate: no bounds, global memory, buffer = value
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::immediate{100};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
|
||||
assert(process_segments(seg_size) == 100);
|
||||
}
|
||||
|
||||
// __immediate_sequence: per-segment span with runtime bounds only
|
||||
{
|
||||
int sizes[3] = {64, 128, 96};
|
||||
auto seg_sizes = cuda::args::__immediate_sequence{cuda::std::span<int>{sizes, 3}, cuda::args::bounds(1, 200)};
|
||||
assert(select_variant(seg_sizes) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg_sizes, 3) == 200 * 3);
|
||||
assert(process_segments(seg_sizes) == 64 + 128 + 96);
|
||||
}
|
||||
|
||||
// __immediate_sequence: per-segment span with both bounds
|
||||
{
|
||||
int sizes[3] = {64, 128, 96};
|
||||
auto seg_sizes = cuda::args::__immediate_sequence{
|
||||
cuda::std::span<int>{sizes, 3}, cuda::args::bounds<1, 256>(), cuda::args::bounds(1, 200)};
|
||||
static_assert(cuda::args::__traits<decltype(seg_sizes)>::highest <= shared_memory_capacity);
|
||||
assert(select_variant(seg_sizes) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_sizes, 3) == 200 * 3);
|
||||
assert(process_segments(seg_sizes) == 64 + 128 + 96);
|
||||
}
|
||||
|
||||
// deferred: uniform, bounds for decisions only
|
||||
{
|
||||
int val = 100;
|
||||
auto seg_size =
|
||||
cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds<1, 256>(), cuda::args::bounds(1, 200)};
|
||||
static_assert(cuda::args::__traits<decltype(seg_size)>::highest <= shared_memory_capacity);
|
||||
assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 200 * 4);
|
||||
}
|
||||
|
||||
// --- Floating point cases ---
|
||||
|
||||
// Plain float: no bounds
|
||||
{
|
||||
static_assert(select_variant(1.0f) == algorithm_variant::global_memory);
|
||||
assert(process_segments(1.0f) == 1);
|
||||
}
|
||||
|
||||
// constant float using an integer NTTP and explicit value type
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::constant<128, float>{};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(process_segments(seg_size) == 128);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// constant float (float NTTPs require C++20)
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::constant<128.0f>{};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(process_segments(seg_size) == 128);
|
||||
}
|
||||
|
||||
// immediate float with static bounds
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::immediate{100.0f, cuda::args::bounds<1.0f, 256.0f>()};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(process_segments(seg_size) == 100);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,115 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
// UNSUPPORTED: nvcc-11, nvcc-12
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
// TODO: Add support for new half
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "atomic_helpers.h"
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFn
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Fetch min
|
||||
{
|
||||
using A = cuda::atomic<T, ThreadScope>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(T(-5)) == T(-1));
|
||||
printf("%i == %i\n", (int) t.load(), (int) T(-5));
|
||||
NV_IF_TARGET(NV_IS_HOST, (fflush(stdout);))
|
||||
assert(t.load() == T(-5));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T, ThreadScope>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(T(-5)) == T(-1));
|
||||
assert(t.load() == T(-5));
|
||||
}
|
||||
// Test not lesser
|
||||
{
|
||||
using A = cuda::atomic<T, ThreadScope>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(4) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T, ThreadScope>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(4) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
// Fetch max
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(1);
|
||||
assert(t.fetch_max(2) == T(1));
|
||||
assert(t.load() == T(2));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(1);
|
||||
assert(t.fetch_max(2) == T(1));
|
||||
assert(t.load() == T(2));
|
||||
}
|
||||
// Test not greater
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_max(2) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_max(2) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(NV_IS_HOST,
|
||||
(TestFn<__half, local_memory_selector, cuda::thread_scope::thread_scope_thread>()();),
|
||||
NV_PROVIDES_SM_70,
|
||||
(TestFn<__half, local_memory_selector, cuda::thread_scope::thread_scope_thread>()();))
|
||||
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(TestFn<__half, shared_memory_selector, cuda::thread_scope::thread_scope_thread>()();
|
||||
TestFn<__half, global_memory_selector, cuda::thread_scope::thread_scope_thread>()();))
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,140 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "atomic_helpers.h"
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <class T,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope ThreadScope,
|
||||
bool Signed = cuda::std::is_signed<T>::value>
|
||||
struct TestFn
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Test greater
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(1);
|
||||
assert(t.fetch_max(2) == T(1));
|
||||
assert(t.load() == T(2));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(1);
|
||||
assert(t.fetch_max(2) == T(1));
|
||||
assert(t.load() == T(2));
|
||||
}
|
||||
// Test not greater
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_max(2) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_max(2) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFn<T, Selector, ThreadScope, true>
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Call unsigned tests
|
||||
TestFn<T, Selector, ThreadScope, false>()();
|
||||
// Test greater, but with signed math
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-5);
|
||||
assert(t.fetch_max(-1) == T(-5));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-5);
|
||||
assert(t.fetch_max(-1) == T(-5));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
// Test not greater
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_max(-5) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_max(-5) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFnDispatch
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestFn<T, Selector, ThreadScope>()();
|
||||
}
|
||||
};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();),
|
||||
NV_PROVIDES_SM_70,
|
||||
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();))
|
||||
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(TestEachIntegralType<TestFnDispatch, shared_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, shared_memory_selector>()();
|
||||
TestEachIntegralType<TestFnDispatch, global_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, global_memory_selector>()();))
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,140 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "atomic_helpers.h"
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <class T,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope ThreadScope,
|
||||
bool Signed = cuda::std::is_signed<T>::value>
|
||||
struct TestFn
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Test lesser
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(5);
|
||||
assert(t.fetch_min(4) == T(5));
|
||||
assert(t.load() == T(4));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(5);
|
||||
assert(t.fetch_min(4) == T(5));
|
||||
assert(t.load() == T(4));
|
||||
}
|
||||
// Test not lesser
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_min(4) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_min(4) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFn<T, Selector, ThreadScope, true>
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Call unsigned tests
|
||||
TestFn<T, Selector, ThreadScope, false>()();
|
||||
// Test lesser, but with signed math
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(-5) == T(-1));
|
||||
assert(t.load() == T(-5));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(-5) == T(-1));
|
||||
assert(t.load() == T(-5));
|
||||
}
|
||||
// Test not lesser
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(4) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(4) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFnDispatch
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestFn<T, Selector, ThreadScope>()();
|
||||
}
|
||||
};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();),
|
||||
NV_PROVIDES_SM_70,
|
||||
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();))
|
||||
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(TestEachIntegralType<TestFnDispatch, shared_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, shared_memory_selector>()();
|
||||
TestEachIntegralType<TestFnDispatch, global_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, global_memory_selector>()();))
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,102 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef ATOMIC_HELPERS_H
|
||||
#define ATOMIC_HELPERS_H
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
struct UserAtomicType
|
||||
{
|
||||
int i;
|
||||
|
||||
TEST_FUNC explicit UserAtomicType(int d = 0) noexcept
|
||||
: i(d)
|
||||
{}
|
||||
|
||||
TEST_FUNC friend bool operator==(const UserAtomicType& x, const UserAtomicType& y)
|
||||
{
|
||||
return x.i == y.i;
|
||||
}
|
||||
};
|
||||
|
||||
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope Scope
|
||||
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
= cuda::thread_scope_system
|
||||
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
>
|
||||
struct TestEachIntegralType
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestFunctor<char, Selector, Scope>()();
|
||||
TestFunctor<signed char, Selector, Scope>()();
|
||||
TestFunctor<unsigned char, Selector, Scope>()();
|
||||
TestFunctor<short, Selector, Scope>()();
|
||||
TestFunctor<unsigned short, Selector, Scope>()();
|
||||
TestFunctor<int, Selector, Scope>()();
|
||||
TestFunctor<unsigned int, Selector, Scope>()();
|
||||
TestFunctor<long, Selector, Scope>()();
|
||||
TestFunctor<unsigned long, Selector, Scope>()();
|
||||
TestFunctor<long long, Selector, Scope>()();
|
||||
TestFunctor<unsigned long long, Selector, Scope>()();
|
||||
TestFunctor<wchar_t, Selector, Scope>();
|
||||
TestFunctor<char16_t, Selector, Scope>()();
|
||||
TestFunctor<char32_t, Selector, Scope>()();
|
||||
TestFunctor<int8_t, Selector, Scope>()();
|
||||
TestFunctor<uint8_t, Selector, Scope>()();
|
||||
TestFunctor<int16_t, Selector, Scope>()();
|
||||
TestFunctor<uint16_t, Selector, Scope>()();
|
||||
TestFunctor<int32_t, Selector, Scope>()();
|
||||
TestFunctor<uint32_t, Selector, Scope>()();
|
||||
TestFunctor<int64_t, Selector, Scope>()();
|
||||
TestFunctor<uint64_t, Selector, Scope>()();
|
||||
}
|
||||
};
|
||||
|
||||
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope Scope
|
||||
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
= cuda::thread_scope_system
|
||||
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
>
|
||||
struct TestEachFloatingPointType
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestFunctor<float, Selector, Scope>()();
|
||||
TestFunctor<double, Selector, Scope>()();
|
||||
}
|
||||
};
|
||||
|
||||
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope Scope
|
||||
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
= cuda::thread_scope_system
|
||||
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
>
|
||||
struct TestEachAtomicType
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestEachIntegralType<TestFunctor, Selector, Scope>()();
|
||||
TestEachFloatingPointType<TestFunctor, Selector, Scope>()();
|
||||
TestFunctor<UserAtomicType, Selector, Scope>()();
|
||||
TestFunctor<int*, Selector, Scope>()();
|
||||
TestFunctor<const int*, Selector, Scope>()();
|
||||
}
|
||||
};
|
||||
|
||||
#endif // ATOMIC_HELPER_H
|
||||
@@ -0,0 +1,13 @@
|
||||
// -*- C++ -*-
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T store(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.store(in + 1, cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T compare_exchange_weak(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
T old = T(7);
|
||||
x.compare_exchange_weak(old, T(42), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T compare_exchange_strong(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
T old = T(7);
|
||||
x.compare_exchange_strong(old, T(42), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T exchange(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
T out = x.exchange(T(1), cuda::memory_order_relaxed);
|
||||
return out + x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_add(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_add(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_sub(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_sub(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_and(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_and(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_or(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_or(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_xor(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_xor(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_min(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_min(T(7), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_max(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_max(T(7), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC inline void tests()
|
||||
{
|
||||
const T tid = threadIdx.x;
|
||||
assert(tid + T(1) == store(tid));
|
||||
assert(T(1) + tid == exchange(tid));
|
||||
assert(tid == T(7) ? T(42) : tid == compare_exchange_weak(tid));
|
||||
assert(tid == T(7) ? T(42) : tid == compare_exchange_strong(tid));
|
||||
assert((tid + T(1)) == fetch_add(tid));
|
||||
assert((tid & T(1)) == fetch_and(tid));
|
||||
assert((tid | T(1)) == fetch_or(tid));
|
||||
assert((tid ^ T(1)) == fetch_xor(tid));
|
||||
assert(min(tid, T(7)) == fetch_min(tid));
|
||||
assert(max(tid, T(7)) == fetch_max(tid));
|
||||
assert(T(tid - T(1)) == fetch_sub(tid));
|
||||
}
|
||||
|
||||
int main(int arg, char** argv)
|
||||
{
|
||||
#if !defined(_CCCL_ATOMIC_UNSAFE_AUTOMATIC_STORAGE)
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_HOST,
|
||||
(cuda_thread_count = 64;),
|
||||
(tests<uint8_t>(); tests<uint16_t>(); tests<uint32_t>(); tests<uint64_t>(); tests<int8_t>(); tests<int16_t>();
|
||||
tests<int32_t>();
|
||||
tests<int64_t>();))
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,119 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#define _LIBCUDACXX_FORCE_PTX_AUTOMATIC_STORAGE_PATH 1 // Force using the PTX is_local atomics path
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
/*
|
||||
Test goals:
|
||||
Pre-load registers with values that will be used to trigger the wrong codepath in local device atomics.
|
||||
|
||||
This test is architecture and driver dependent. It is not possible to reproduce this when compiled to SASS on 12.0, but
|
||||
will repro on 12.8.
|
||||
|
||||
Compiled to SASS is an important point, compiling to PTX will show the failure to initialize the local test flag for
|
||||
isspacep.local to 0, but that might be compiled out by the JIT compiler in the driver
|
||||
*/
|
||||
__global__ void __launch_bounds__(1024) device_test(char* gmem)
|
||||
{
|
||||
constexpr int threads = 1024;
|
||||
|
||||
__shared__ int hidx;
|
||||
__shared__ int histogram[threads];
|
||||
|
||||
cuda::atomic<int, cuda::thread_scope_thread> xatom(0);
|
||||
|
||||
constexpr int passes = 16;
|
||||
constexpr int ops = 32;
|
||||
constexpr int expected = passes * ops;
|
||||
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
hidx = 0;
|
||||
memset(histogram, sizeof(histogram), 0);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
for (xatom = 0; xatom.load() < passes; xatom++)
|
||||
{
|
||||
using A = cuda::atomic_ref<int, cuda::std::thread_scope_block>;
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 0]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 1]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 2]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 3]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 4]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 5]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 6]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 7]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 8]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 9]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 10]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 11]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 12]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 13]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 14]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 15]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 16]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 17]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 18]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 19]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 20]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 21]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 22]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 23]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 24]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 25]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 26]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 27]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 28]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 29]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 30]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 31]);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
if (histogram[threadIdx.x] != expected)
|
||||
{
|
||||
printf("[%i] = %i\r\n", threadIdx.x, histogram[threadIdx.x]);
|
||||
}
|
||||
assert(histogram[threadIdx.x] == expected);
|
||||
}
|
||||
|
||||
void launch_kernel()
|
||||
{
|
||||
cudaError_t err;
|
||||
char* inptr = nullptr;
|
||||
CUDA_CALL(err, cudaGetLastError());
|
||||
CUDA_CALL(err, cudaMalloc(&inptr, 1024));
|
||||
CUDA_CALL(err, cudaMemset(inptr, 1, 1024));
|
||||
device_test<<<1, 1024>>>(inptr);
|
||||
CUDA_CALL(err, cudaGetLastError());
|
||||
CUDA_CALL(err, cudaDeviceSynchronize());
|
||||
}
|
||||
|
||||
int main(int arg, char** argv)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, (launch_kernel();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: windows
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Check that atomics on host may be constructed
|
||||
template <class T>
|
||||
TEST_FUNC void do_test()
|
||||
{
|
||||
T v(0);
|
||||
cuda::atomic_ref<T> a(v);
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
do_test<__int128_t>();
|
||||
do_test<__uint128_t>();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Check that host atomics fail to build
|
||||
template <class T>
|
||||
TEST_FUNC void do_test()
|
||||
{
|
||||
T v(0);
|
||||
cuda::atomic_ref<T> a(v);
|
||||
a.store(1);
|
||||
assert(a++ == 1);
|
||||
assert(a.load() == 2);
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, (do_test<__int128_t>(); do_test<__uint128_t>();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: pre-sm-70
|
||||
// UNSUPPORTED: windows
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC constexpr T combine_literal(uint64_t lower, uint64_t upper)
|
||||
{
|
||||
return T(lower) | (T(upper) << 64);
|
||||
}
|
||||
|
||||
template <template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
TEST_FUNC void test()
|
||||
{
|
||||
{
|
||||
using T = __int128_t;
|
||||
using A = cuda::atomic_ref<T, ThreadScope>;
|
||||
Selector<T, constructor_initializer> sel;
|
||||
T& t = *sel.construct();
|
||||
t = T(0);
|
||||
A atom(t);
|
||||
auto test_v = combine_literal<T>(0x01234567DEADBEEF, 0x1337B33701234567);
|
||||
atom.store(test_v, cuda::std::memory_order_release);
|
||||
assert(atom.load() == test_v);
|
||||
}
|
||||
{
|
||||
using T = __uint128_t;
|
||||
using A = cuda::atomic_ref<T, ThreadScope>;
|
||||
Selector<T, constructor_initializer> sel;
|
||||
T& t = *sel.construct();
|
||||
t = T(0);
|
||||
A atom(t);
|
||||
auto test_v = combine_literal<T>(0x01234567DEADBEEF, 0x1337B33701234567);
|
||||
atom.store(test_v);
|
||||
assert(atom.load() == test_v);
|
||||
}
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
#if __cccl_ptx_isa >= 840
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70,
|
||||
(test<local_memory_selector, cuda::thread_scope_thread>(); test<shared_memory_selector, cuda::thread_scope_block>();
|
||||
test<global_memory_selector, cuda::thread_scope_block>();
|
||||
test<global_memory_selector, cuda::thread_scope_device>();))
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
/*
|
||||
Test goals:
|
||||
Interleaved 8b/16b access to a 32b window while there is thread contention.
|
||||
|
||||
for 8b:
|
||||
Launch 1024 threads, fetch_add(1) each window, value at end of kernel should be 0xFF..FF. This checks for corruption
|
||||
caused by interleaved access to different parts of the window.
|
||||
|
||||
for 16b:
|
||||
Launch 1024 threads, fetch_add(1), checking for 0x01FF01FF.
|
||||
*/
|
||||
|
||||
template <class T, int Inc>
|
||||
TEST_FUNC void fetch_add_into_window(T* window, uint16_t* atomHistory)
|
||||
{
|
||||
using Atom = cuda::atomic_ref<T, cuda::thread_scope_block>;
|
||||
|
||||
Atom a(*window);
|
||||
*atomHistory = a.fetch_add(Inc);
|
||||
}
|
||||
|
||||
template <class T>
|
||||
TEST_DEVICE_FUNC void device_do_test(uint32_t expected)
|
||||
{
|
||||
constexpr uint32_t threadCount = 1024;
|
||||
constexpr uint32_t histogramResultCount = 256 * sizeof(T);
|
||||
constexpr uint32_t histogramEntriesPerThread = 4 / sizeof(T);
|
||||
|
||||
__shared__ uint16_t atomHistory[threadCount];
|
||||
__shared__ uint8_t atomHistogram[histogramResultCount];
|
||||
__shared__ uint32_t atomicStorage;
|
||||
|
||||
cuda::atomic_ref<uint32_t, cuda::thread_scope_block> bucket(atomicStorage);
|
||||
|
||||
constexpr uint32_t offsetMask = ((4 / sizeof(T)) - 1);
|
||||
// Access offset is interleaved meaning threads 4, 5, 6, 7 access window 0, 1, 2, 3 and so on.
|
||||
const uint32_t threadOffset = threadIdx.x & offsetMask;
|
||||
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
memset(atomHistogram, 0, histogramResultCount);
|
||||
bucket.store(0);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
T* window = reinterpret_cast<T*>(&atomicStorage) + threadOffset;
|
||||
fetch_add_into_window<T, 1>(window, atomHistory + threadIdx.x);
|
||||
|
||||
__syncthreads();
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
// For each thread, add its atomic result into the corresponding bucket
|
||||
for (uint32_t i = 0; i < threadCount; i++)
|
||||
{
|
||||
atomHistogram[atomHistory[i]]++;
|
||||
}
|
||||
// Check that each bucket has exactly (4 / sizeof(T)) entries
|
||||
// This checks that atomic fetch operations were sequential. i.e. 4xfetch_add(1) returns [0, 1, 2, 3]
|
||||
for (uint32_t i = 0; i < histogramResultCount; i++)
|
||||
{
|
||||
assert(atomHistogram[i] == histogramEntriesPerThread);
|
||||
}
|
||||
printf("expected: 0x%X\r\n", expected);
|
||||
printf("result: 0x%X\r\n", bucket.load());
|
||||
assert(bucket.load() == expected);
|
||||
}
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(NV_IS_HOST,
|
||||
(cuda_thread_count = 1024;),
|
||||
NV_IS_DEVICE,
|
||||
(device_do_test<uint8_t>(0); device_do_test<uint16_t>(0x02000200);));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
// cuda::atomic<key>
|
||||
|
||||
// Original test issue:
|
||||
// https://github.com/NVIDIA/libcudacxx/issues/160
|
||||
|
||||
#include <cuda/atomic>
|
||||
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <template <typename, typename> class Selector>
|
||||
struct TestFn
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
{
|
||||
struct key
|
||||
{
|
||||
int32_t a;
|
||||
int32_t b;
|
||||
};
|
||||
using A = cuda::std::atomic<key>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
cuda::std::atomic_init(&t, key{1, 2});
|
||||
auto r = t.load();
|
||||
auto d = key{5, 5};
|
||||
t.store(r);
|
||||
(void) t.exchange(r);
|
||||
(void) t.compare_exchange_weak(r, d, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
|
||||
(void) t.compare_exchange_strong(d, r, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
|
||||
}
|
||||
{
|
||||
struct alignas(8) key
|
||||
{
|
||||
int32_t a;
|
||||
int32_t b;
|
||||
};
|
||||
using A = cuda::std::atomic<key>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
cuda::std::atomic_init(&t, key{1, 2});
|
||||
auto r = t.load();
|
||||
auto d = key{5, 5};
|
||||
t.store(r);
|
||||
(void) t.exchange(r);
|
||||
(void) t.compare_exchange_weak(r, d, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
|
||||
(void) t.compare_exchange_strong(d, r, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(NV_IS_HOST, TestFn<local_memory_selector>()();
|
||||
, NV_PROVIDES_SM_70, TestFn<local_memory_selector>()();)
|
||||
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (TestFn<shared_memory_selector>()(); TestFn<global_memory_selector>()();))
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
#ifndef TEST_ARRIVE_TX_H_
|
||||
#define TEST_ARRIVE_TX_H_
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/memory>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include "concurrent_agents.h"
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
template <typename Barrier>
|
||||
inline TEST_DEVICE_FUNC void mbarrier_complete_tx(Barrier& b, int transaction_count)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_90,
|
||||
(
|
||||
if (cuda::device::is_address_from(cuda::device::barrier_native_handle(b), cuda::device::address_space::shared)) {
|
||||
asm volatile(
|
||||
"mbarrier.complete_tx.relaxed.cta.shared::cta.b64 [%0], %1;"
|
||||
:
|
||||
: "r"((unsigned int) __cvta_generic_to_shared(cuda::device::barrier_native_handle(b))), "r"(transaction_count)
|
||||
: "memory");
|
||||
} else { __trap(); }),
|
||||
NV_ANY_TARGET,
|
||||
(
|
||||
// On architectures pre-SM90 (and on host), we drop the transaction count
|
||||
// update. The barriers do not keep track of transaction counts.
|
||||
__trap();));
|
||||
}
|
||||
|
||||
template <bool split_arrive_and_expect>
|
||||
TEST_DEVICE_FUNC void thread(cuda::barrier<cuda::thread_scope_block>& b, int arrives_per_thread)
|
||||
{
|
||||
constexpr int tx_count = 1;
|
||||
typename cuda::barrier<cuda::thread_scope_block>::arrival_token tok;
|
||||
|
||||
if _CCCL_CONSTEXPR_CXX20 (split_arrive_and_expect)
|
||||
{
|
||||
cuda::device::barrier_expect_tx(b, tx_count);
|
||||
tok = b.arrive(arrives_per_thread);
|
||||
}
|
||||
else
|
||||
{
|
||||
tok = cuda::device::barrier_arrive_tx(b, arrives_per_thread, tx_count);
|
||||
}
|
||||
|
||||
// Manually increase the transaction count of the barrier.
|
||||
mbarrier_complete_tx(b, tx_count);
|
||||
|
||||
b.wait(cuda::std::move(tok));
|
||||
}
|
||||
|
||||
template <bool split_arrive_and_expect>
|
||||
TEST_DEVICE_FUNC void test()
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(
|
||||
// Run all threads, each arriving with arrival count 1
|
||||
using barrier_t = cuda::barrier<cuda::thread_scope_block>;
|
||||
|
||||
shared_memory_selector<barrier_t, constructor_initializer> sel_1;
|
||||
barrier_t* bar_1 = sel_1.construct(blockDim.x);
|
||||
__syncthreads();
|
||||
thread<split_arrive_and_expect>(*bar_1, 1);
|
||||
|
||||
// Run all threads, each arriving with arrival count 2
|
||||
shared_memory_selector<barrier_t, constructor_initializer> sel_2;
|
||||
barrier_t* bar_2 = sel_2.construct(2 * blockDim.x);
|
||||
__syncthreads();
|
||||
thread<split_arrive_and_expect>(*bar_2, 2);));
|
||||
}
|
||||
|
||||
#endif // TEST_ARRIVE_TX_H_
|
||||
@@ -0,0 +1,56 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// UNSUPPORTED: clang && !nvcc
|
||||
|
||||
// UNSUPPORTED: no_execute
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include <cooperative_groups.h>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// When PR #416 is merged, uncomment this line:
|
||||
// cuda_cluster_size = 2;
|
||||
),
|
||||
NV_IS_DEVICE,
|
||||
(__shared__ cuda::barrier<cuda::thread_scope_block> bar;
|
||||
|
||||
if (threadIdx.x == 0) { init(&bar, blockDim.x); } namespace cg = cooperative_groups;
|
||||
auto cluster = cg::this_cluster();
|
||||
|
||||
cluster.sync();
|
||||
|
||||
// This test currently fails at this point because support for
|
||||
// clusters has not yet been added.
|
||||
cuda::barrier<cuda::thread_scope_block> * remote_bar;
|
||||
remote_bar = cluster.map_shared_rank(&bar, cluster.block_rank() ^ 1);
|
||||
|
||||
// When PR #416 is merged, this should fail here because the barrier
|
||||
// is in device memory.
|
||||
auto token = cuda::device::barrier_arrive_tx(*remote_bar, 1, 0);));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 256;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: no_execute
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
TEST_DEVICE_FUNC uint64_t bar_storage;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(cuda::barrier<cuda::thread_scope_block> * bar_ptr;
|
||||
bar_ptr = reinterpret_cast<cuda::barrier<cuda::thread_scope_block>*>(bar_storage);
|
||||
|
||||
if (threadIdx.x == 0) { init(bar_ptr, blockDim.x); } __syncthreads();
|
||||
|
||||
// Should fail because the barrier is in device memory.
|
||||
[[maybe_unused]] auto token = cuda::device::barrier_arrive_tx(*bar_ptr, 1, 0);));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#ifndef __cccl_lib_local_barrier_arrive_tx
|
||||
static_assert(false, "should define __cccl_lib_local_barrier_arrive_tx");
|
||||
#endif // __cccl_lib_local_barrier_arrive_tx
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
|
||||
// UNSUPPORTED: pre-sm-70
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(__shared__ cuda::barrier<cuda::thread_scope_block> bar;
|
||||
if (threadIdx.x == 0) { init(&bar, blockDim.x); } __syncthreads();
|
||||
|
||||
// barrier_arrive_tx should fail on SM70 and SM80, because it is hidden.
|
||||
auto token = cuda::device::barrier_arrive_tx(bar, 1, 0);
|
||||
|
||||
#ifdef __cccl_lib_local_barrier_arrive_tx
|
||||
static_assert(false, "Fail manually for SM90 and up.");
|
||||
#endif // __cccl_lib_local_barrier_arrive_tx
|
||||
));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 2;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 32;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,107 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/utility> // cuda::std::move
|
||||
|
||||
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
using barrier = cuda::barrier<cuda::thread_scope_block>;
|
||||
namespace cde = cuda::device::experimental;
|
||||
|
||||
static constexpr int buf_len = 1024;
|
||||
alignas(128) TEST_GLOBAL_VARIABLE int gmem_buffer[buf_len];
|
||||
|
||||
TEST_DEVICE_FUNC void test()
|
||||
{
|
||||
// SETUP: fill global memory buffer
|
||||
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
|
||||
{
|
||||
gmem_buffer[i] = i;
|
||||
}
|
||||
// Ensure that writes to global memory are visible to others, including
|
||||
// those in the async proxy.
|
||||
__threadfence();
|
||||
__syncthreads();
|
||||
|
||||
// TEST: Add i to buffer[i]
|
||||
alignas(16) __shared__ int smem_buffer[buf_len];
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ barrier bar;
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Load data:
|
||||
uint64_t token;
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
cde::cp_async_bulk_global_to_shared(smem_buffer, gmem_buffer, sizeof(smem_buffer), bar);
|
||||
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
|
||||
}
|
||||
else
|
||||
{
|
||||
token = bar.arrive();
|
||||
}
|
||||
bar.wait(cuda::std::move(token));
|
||||
|
||||
// Update in shared memory
|
||||
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
|
||||
{
|
||||
smem_buffer[i] += i;
|
||||
}
|
||||
cde::fence_proxy_async_shared_cta();
|
||||
__syncthreads();
|
||||
|
||||
// Write back to global memory:
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
cde::cp_async_bulk_shared_to_global(gmem_buffer, smem_buffer, sizeof(smem_buffer));
|
||||
cde::cp_async_bulk_commit_group();
|
||||
cde::cp_async_bulk_wait_group_read<0>();
|
||||
}
|
||||
__threadfence();
|
||||
__syncthreads();
|
||||
|
||||
// TEAR-DOWN: check that global memory is correct
|
||||
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
|
||||
{
|
||||
assert(gmem_buffer[i] == 2 * i);
|
||||
}
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512;));
|
||||
|
||||
NV_DISPATCH_TARGET(NV_IS_DEVICE, (test();));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#ifndef __cccl_lib_experimental_ctk12_cp_async_exposure
|
||||
static_assert(false, "should define __cccl_lib_experimental_ctk12_cp_async_exposure");
|
||||
#endif
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
using barrier = cuda::barrier<cuda::thread_scope_block>;
|
||||
namespace cde = cuda::device::experimental;
|
||||
|
||||
// Kernels below are intended to be compiled, but not run. This is to check if
|
||||
// all generated PTX is valid.
|
||||
__global__ void test_bulk_tensor(CUtensorMap* map)
|
||||
{
|
||||
__shared__ int smem;
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ barrier bar;
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
cde::cp_async_bulk_tensor_1d_global_to_shared(&smem, map, 0, bar);
|
||||
cde::cp_async_bulk_tensor_2d_global_to_shared(&smem, map, 0, 0, bar);
|
||||
cde::cp_async_bulk_tensor_3d_global_to_shared(&smem, map, 0, 0, 0, bar);
|
||||
cde::cp_async_bulk_tensor_4d_global_to_shared(&smem, map, 0, 0, 0, 0, bar);
|
||||
cde::cp_async_bulk_tensor_5d_global_to_shared(&smem, map, 0, 0, 0, 0, 0, bar);
|
||||
|
||||
cde::cp_async_bulk_tensor_1d_shared_to_global(map, 0, &smem);
|
||||
cde::cp_async_bulk_tensor_2d_shared_to_global(map, 0, 0, &smem);
|
||||
cde::cp_async_bulk_tensor_3d_shared_to_global(map, 0, 0, 0, &smem);
|
||||
cde::cp_async_bulk_tensor_4d_shared_to_global(map, 0, 0, 0, 0, &smem);
|
||||
cde::cp_async_bulk_tensor_5d_shared_to_global(map, 0, 0, 0, 0, 0, &smem);
|
||||
}
|
||||
|
||||
__global__ void test_bulk(void* gmem)
|
||||
{
|
||||
__shared__ int smem;
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
cde::cp_async_bulk_global_to_shared(&smem, gmem, 1024, bar);
|
||||
cde::cp_async_bulk_shared_to_global(gmem, &smem, 1024);
|
||||
}
|
||||
|
||||
__global__ void test_fences_async_group(void*)
|
||||
{
|
||||
cde::fence_proxy_async_shared_cta();
|
||||
|
||||
cde::cp_async_bulk_commit_group();
|
||||
// Wait for up to 8 groups
|
||||
cde::cp_async_bulk_wait_group_read<0>();
|
||||
cde::cp_async_bulk_wait_group_read<1>();
|
||||
cde::cp_async_bulk_wait_group_read<2>();
|
||||
cde::cp_async_bulk_wait_group_read<3>();
|
||||
cde::cp_async_bulk_wait_group_read<4>();
|
||||
cde::cp_async_bulk_wait_group_read<5>();
|
||||
cde::cp_async_bulk_wait_group_read<6>();
|
||||
cde::cp_async_bulk_wait_group_read<7>();
|
||||
cde::cp_async_bulk_wait_group_read<8>();
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,221 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
// UNSUPPORTED: clang && !nvcc
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/utility> // cuda::std::move
|
||||
|
||||
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
|
||||
|
||||
// NVRTC does not support cuda.h (due to import of stdlib.h)
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# include <cudaTypedefs.h> // PFN_cuTensorMapEncodeTiled, CUtensorMap
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
using barrier = cuda::barrier<cuda::thread_scope_block>;
|
||||
namespace cde = cuda::device::experimental;
|
||||
|
||||
constexpr size_t GMEM_WIDTH = 1024; // Width of tensor (in # elements)
|
||||
constexpr size_t GMEM_HEIGHT = 1024; // Height of tensor (in # elements)
|
||||
constexpr size_t gmem_len = GMEM_WIDTH * GMEM_HEIGHT;
|
||||
|
||||
constexpr int SMEM_WIDTH = 32; // Width of shared memory buffer (in # elements)
|
||||
constexpr int SMEM_HEIGHT = 8; // Height of shared memory buffer (in # elements)
|
||||
|
||||
static constexpr int buf_len = SMEM_HEIGHT * SMEM_WIDTH;
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
// We need a type with a size. On NVRTC, cuda.h cannot be imported, so we don't
|
||||
// have access to the definition of CUTensorMap (only to the declaration of CUtensorMap inside
|
||||
// cuda/barrier). So we use this type instead and reinterpret_cast in the
|
||||
// kernel.
|
||||
struct fake_cutensormap
|
||||
{
|
||||
alignas(64) uint64_t opaque[16];
|
||||
};
|
||||
__constant__ fake_cutensormap global_fake_tensor_map;
|
||||
|
||||
TEST_DEVICE_FUNC void test(int base_i, int base_j)
|
||||
{
|
||||
CUtensorMap* global_tensor_map = reinterpret_cast<CUtensorMap*>(&global_fake_tensor_map);
|
||||
|
||||
// SETUP: fill global memory buffer
|
||||
for (int i = threadIdx.x; i < static_cast<int>(gmem_len); i += blockDim.x)
|
||||
{
|
||||
gmem_tensor[i] = i;
|
||||
}
|
||||
// Ensure that writes to global memory are visible to others, including
|
||||
// those in the async proxy.
|
||||
__threadfence();
|
||||
__syncthreads();
|
||||
|
||||
// TEST: Add i to buffer[i]
|
||||
alignas(128) __shared__ int smem_buffer[buf_len];
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ barrier bar;
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Load data:
|
||||
uint64_t token;
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
// Fastest moving coordinate first.
|
||||
cde::cp_async_bulk_tensor_2d_global_to_shared(smem_buffer, global_tensor_map, base_j, base_i, bar);
|
||||
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
|
||||
}
|
||||
else
|
||||
{
|
||||
token = bar.arrive();
|
||||
}
|
||||
bar.wait(cuda::std::move(token));
|
||||
|
||||
// Check smem
|
||||
for (int i = 0; i < SMEM_HEIGHT; ++i)
|
||||
{
|
||||
for (int j = 0; j < SMEM_HEIGHT; ++j)
|
||||
{
|
||||
const int gmem_lin_idx = (base_i + i) * GMEM_WIDTH + base_j + j;
|
||||
const int smem_lin_idx = i * SMEM_WIDTH + j;
|
||||
|
||||
assert(smem_buffer[smem_lin_idx] == gmem_lin_idx);
|
||||
}
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Update smem
|
||||
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
|
||||
{
|
||||
smem_buffer[i] = 2 * smem_buffer[i] + 1;
|
||||
}
|
||||
cde::fence_proxy_async_shared_cta();
|
||||
__syncthreads();
|
||||
|
||||
// Write back to global memory:
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
cde::cp_async_bulk_tensor_2d_shared_to_global(global_tensor_map, base_j, base_i, smem_buffer);
|
||||
cde::cp_async_bulk_commit_group();
|
||||
cde::cp_async_bulk_wait_group_read<0>();
|
||||
}
|
||||
__threadfence();
|
||||
__syncthreads();
|
||||
|
||||
// TEAR-DOWN: check that global memory is correct
|
||||
for (int i = 0; i < SMEM_HEIGHT; ++i)
|
||||
{
|
||||
for (int j = 0; j < SMEM_HEIGHT; ++j)
|
||||
{
|
||||
int gmem_lin_idx = (base_i + i) * GMEM_WIDTH + base_j + j;
|
||||
|
||||
assert(gmem_tensor[gmem_lin_idx] == 2 * gmem_lin_idx + 1);
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# if _CCCL_CTK_BELOW(12, 5)
|
||||
PFN_cuTensorMapEncodeTiled get_cuTensorMapEncodeTiled()
|
||||
{
|
||||
void* driver_ptr = nullptr;
|
||||
cudaDriverEntryPointQueryResult driver_status;
|
||||
auto code = cudaGetDriverEntryPoint("cuTensorMapEncodeTiled", &driver_ptr, cudaEnableDefault, &driver_status);
|
||||
assert(code == cudaSuccess && "Could not get driver API");
|
||||
return reinterpret_cast<PFN_cuTensorMapEncodeTiled>(driver_ptr);
|
||||
}
|
||||
# else // ^^^ _CCCL_CTK_BELOW(12, 5) ^^^ / vvv _CCCL_CTK_AT_LEAST(12, 5) vvv
|
||||
PFN_cuTensorMapEncodeTiled_v12000 get_cuTensorMapEncodeTiled()
|
||||
{
|
||||
void* driver_ptr = nullptr;
|
||||
cudaDriverEntryPointQueryResult driver_status;
|
||||
auto code =
|
||||
cudaGetDriverEntryPointByVersion("cuTensorMapEncodeTiled", &driver_ptr, 12000, cudaEnableDefault, &driver_status);
|
||||
assert(code == cudaSuccess && "Could not get driver API");
|
||||
return reinterpret_cast<PFN_cuTensorMapEncodeTiled_v12000>(driver_ptr);
|
||||
}
|
||||
# endif // _CCCL_CTK_AT_LEAST(12, 5)
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512;
|
||||
|
||||
int* tensor_ptr = nullptr;
|
||||
auto code = cudaGetSymbolAddress((void**) &tensor_ptr, gmem_tensor);
|
||||
assert(code == cudaSuccess && "getsymboladdress failed.");
|
||||
|
||||
// https://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TENSOR__MEMORY.html
|
||||
CUtensorMap local_tensor_map{};
|
||||
// rank is the number of dimensions of the array.
|
||||
constexpr uint32_t rank = 2;
|
||||
uint64_t size[rank] = {GMEM_WIDTH, GMEM_HEIGHT};
|
||||
// The stride is the number of bytes to traverse from the first element of one row to the next.
|
||||
// It must be a multiple of 16.
|
||||
uint64_t stride[rank - 1] = {GMEM_WIDTH * sizeof(int)};
|
||||
// The box_size is the size of the shared memory buffer that is used as the
|
||||
// destination of a TMA transfer.
|
||||
uint32_t box_size[rank] = {SMEM_WIDTH, SMEM_HEIGHT};
|
||||
// The distance between elements in units of sizeof(element). A stride of 2
|
||||
// can be used to load only the real component of a complex-valued tensor, for instance.
|
||||
uint32_t elem_stride[rank] = {1, 1};
|
||||
|
||||
// Get a function pointer to the cuTensorMapEncodeTiled driver API.
|
||||
auto cuTensorMapEncodeTiled = get_cuTensorMapEncodeTiled();
|
||||
|
||||
// Create the tensor descriptor.
|
||||
CUresult res = cuTensorMapEncodeTiled(
|
||||
&local_tensor_map, // CUtensorMap *tensorMap,
|
||||
CUtensorMapDataType::CU_TENSOR_MAP_DATA_TYPE_INT32,
|
||||
rank, // cuuint32_t tensorRank,
|
||||
tensor_ptr, // void *globalAddress,
|
||||
size, // const cuuint64_t *globalDim,
|
||||
stride, // const cuuint64_t *globalStrides,
|
||||
box_size, // const cuuint32_t *boxDim,
|
||||
elem_stride, // const cuuint32_t *elementStrides,
|
||||
CUtensorMapInterleave::CU_TENSOR_MAP_INTERLEAVE_NONE,
|
||||
CUtensorMapSwizzle::CU_TENSOR_MAP_SWIZZLE_NONE,
|
||||
CUtensorMapL2promotion::CU_TENSOR_MAP_L2_PROMOTION_NONE,
|
||||
CUtensorMapFloatOOBfill::CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE);
|
||||
|
||||
assert(res == CUDA_SUCCESS && "tensormap creation failed.");
|
||||
code = cudaMemcpyToSymbol(global_fake_tensor_map, &local_tensor_map, sizeof(CUtensorMap));
|
||||
assert(code == cudaSuccess && "memcpytosymbol failed.");));
|
||||
|
||||
NV_DISPATCH_TARGET(NV_IS_DEVICE, (test(0, 0); test(4, 0); test(4, 4);));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 1> GMEM_DIMS{256};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 1> GMEM_DIMS_DEV{256};
|
||||
constexpr cuda::std::array<uint32_t, 1> SMEM_DIMS{32};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 1> SMEM_DIMS_DEV{32};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 1> TEST_SMEM_COORDS[] = {{0}, {4}, {8}};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 2> GMEM_DIMS{8, 11};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 2> GMEM_DIMS_DEV{8, 11};
|
||||
constexpr cuda::std::array<uint32_t, 2> SMEM_DIMS{4, 2};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 2> SMEM_DIMS_DEV{4, 2};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 2> TEST_SMEM_COORDS[] = {
|
||||
{0, 0},
|
||||
{4, 1},
|
||||
{4, 5},
|
||||
{0, 5},
|
||||
};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 3> GMEM_DIMS{8, 11, 13};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 3> GMEM_DIMS_DEV{8, 11, 13};
|
||||
constexpr cuda::std::array<uint32_t, 3> SMEM_DIMS{4, 2, 4};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 3> SMEM_DIMS_DEV{4, 2, 4};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 3> TEST_SMEM_COORDS[] = {{0, 0, 0}, {4, 1, 3}, {4, 5, 1}};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 4> GMEM_DIMS{8, 11, 13, 3};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 4> GMEM_DIMS_DEV{8, 11, 13, 3};
|
||||
constexpr cuda::std::array<uint32_t, 4> SMEM_DIMS{4, 2, 4, 1};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 4> SMEM_DIMS_DEV{4, 2, 4, 1};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 4> TEST_SMEM_COORDS[] = {
|
||||
{0, 0, 0, 0}, {4, 1, 3, 0}, {4, 8, 7, 2}, {4, 5, 1, 1}};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 5> GMEM_DIMS{8, 11, 13, 3, 3};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 5> GMEM_DIMS_DEV{8, 11, 13, 3, 3};
|
||||
constexpr cuda::std::array<uint32_t, 5> SMEM_DIMS{4, 2, 4, 1, 1};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 5> SMEM_DIMS_DEV{4, 2, 4, 1, 1};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 5> TEST_SMEM_COORDS[] = {
|
||||
{0, 0, 0, 0, 0}, {4, 1, 3, 0, 1}, {4, 5, 1, 1, 2}};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,349 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#ifndef TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
|
||||
#define TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/ptx>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/utility> // cuda::std::move
|
||||
|
||||
namespace ptx = cuda::ptx;
|
||||
|
||||
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
|
||||
|
||||
// NVRTC does not support cuda.h (due to import of stdlib.h)
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# include <cstdio>
|
||||
|
||||
# include <cudaTypedefs.h> // PFN_cuTensorMapEncodeTiled, CUtensorMap
|
||||
#endif // ! TEST_COMPILER(NVRTC)
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
using barrier = cuda::barrier<cuda::thread_scope_block>;
|
||||
namespace cde = cuda::device::experimental;
|
||||
|
||||
/*
|
||||
* This header supports the 1d, 2d, ..., 5d test of the TMA PTX wrappers.
|
||||
*
|
||||
* The functions below help convert Nd coordinates into something useful.
|
||||
*
|
||||
*/
|
||||
|
||||
// Compute the total number of elements in a tensor
|
||||
template <class T, size_t num_dims>
|
||||
constexpr TEST_FUNC int tensor_len(cuda::std::array<T, num_dims> dims)
|
||||
{
|
||||
T len = 1;
|
||||
for (T d : dims)
|
||||
{
|
||||
len *= d;
|
||||
}
|
||||
return static_cast<int>(len);
|
||||
}
|
||||
|
||||
// Function to convert:
|
||||
// a linear index into a shared memory tensor
|
||||
// into
|
||||
// a linear index into a global memory tensor.
|
||||
template <size_t num_dims>
|
||||
inline TEST_DEVICE_FUNC int smem_lin_idx_to_gmem_lin_idx(
|
||||
int smem_lin_idx,
|
||||
cuda::std::array<uint32_t, num_dims> smem_coord,
|
||||
cuda::std::array<uint32_t, num_dims> smem_dims,
|
||||
cuda::std::array<uint64_t, num_dims> gmem_dims)
|
||||
{
|
||||
assert(smem_coord.size() == smem_dims.size());
|
||||
assert(smem_coord.size() == gmem_dims.size());
|
||||
|
||||
int gmem_lin_idx = 0;
|
||||
int gmem_stride = 1;
|
||||
for (int i = 0; i < (int) smem_coord.size(); ++i)
|
||||
{
|
||||
int smem_i_idx = smem_lin_idx % smem_dims.begin()[i];
|
||||
gmem_lin_idx += (smem_coord.begin()[i] + smem_i_idx) * gmem_stride;
|
||||
|
||||
smem_lin_idx /= smem_dims.begin()[i];
|
||||
gmem_stride *= gmem_dims.begin()[i];
|
||||
}
|
||||
return gmem_lin_idx;
|
||||
}
|
||||
|
||||
template <size_t num_dims>
|
||||
TEST_DEVICE_FUNC inline void cp_tensor_global_to_shared(
|
||||
CUtensorMap* tensor_map, cuda::std::array<uint32_t, num_dims> indices, void* smem, barrier& bar)
|
||||
{
|
||||
switch (indices.size())
|
||||
{
|
||||
case 1:
|
||||
cde::cp_async_bulk_tensor_1d_global_to_shared(smem, tensor_map, indices[0], bar);
|
||||
break;
|
||||
case 2:
|
||||
cde::cp_async_bulk_tensor_2d_global_to_shared(smem, tensor_map, indices[0], indices[1], bar);
|
||||
break;
|
||||
case 3:
|
||||
cde::cp_async_bulk_tensor_3d_global_to_shared(smem, tensor_map, indices[0], indices[1], indices[2], bar);
|
||||
break;
|
||||
case 4:
|
||||
cde::cp_async_bulk_tensor_4d_global_to_shared(
|
||||
smem, tensor_map, indices[0], indices[1], indices[2], indices[3], bar);
|
||||
break;
|
||||
case 5:
|
||||
cde::cp_async_bulk_tensor_5d_global_to_shared(
|
||||
smem, tensor_map, indices[0], indices[1], indices[2], indices[3], indices[4], bar);
|
||||
break;
|
||||
default:
|
||||
assert(false && "Wrong number of dimensions.");
|
||||
}
|
||||
}
|
||||
|
||||
template <size_t num_dims>
|
||||
TEST_DEVICE_FUNC inline void
|
||||
cp_tensor_shared_to_global(CUtensorMap* tensor_map, cuda::std::array<uint32_t, num_dims> indices, void* smem)
|
||||
{
|
||||
switch (indices.size())
|
||||
{
|
||||
case 1:
|
||||
cde::cp_async_bulk_tensor_1d_shared_to_global(tensor_map, indices[0], smem);
|
||||
break;
|
||||
case 2:
|
||||
cde::cp_async_bulk_tensor_2d_shared_to_global(tensor_map, indices[0], indices[1], smem);
|
||||
break;
|
||||
case 3:
|
||||
cde::cp_async_bulk_tensor_3d_shared_to_global(tensor_map, indices[0], indices[1], indices[2], smem);
|
||||
break;
|
||||
case 4:
|
||||
cde::cp_async_bulk_tensor_4d_shared_to_global(tensor_map, indices[0], indices[1], indices[2], indices[3], smem);
|
||||
break;
|
||||
case 5:
|
||||
cde::cp_async_bulk_tensor_5d_shared_to_global(
|
||||
tensor_map, indices[0], indices[1], indices[2], indices[3], indices[4], smem);
|
||||
break;
|
||||
default:
|
||||
assert(false && "Wrong number of dimensions.");
|
||||
}
|
||||
}
|
||||
|
||||
// To define a tensor map in constant memory, we need a type with a size. On
|
||||
// NVRTC, cuda.h cannot be imported, so we don't have access to the definition
|
||||
// of CUTensorMap (only to the declaration of CUtensorMap inside cuda/barrier).
|
||||
// So we use this type instead and reinterpret_cast in the kernel.
|
||||
struct fake_cutensormap
|
||||
{
|
||||
alignas(64) uint64_t opaque[16];
|
||||
};
|
||||
__constant__ fake_cutensormap global_fake_tensor_map;
|
||||
|
||||
/*
|
||||
* This test has as primary purpose to make sure that the indices in the mapping
|
||||
* from C++ to PTX didn't get mixed up.
|
||||
*
|
||||
* How does it test this?
|
||||
*
|
||||
* 1. It fills a global memory tensor with linear coordinates 0, 1, ...
|
||||
* 2. It loads a tile into shared memory at some coordinate (x, y, ... )
|
||||
* 3. It checks that the coordinates that were received in shared memory match the expected.
|
||||
* 4. It modifies the coordinates (c = 2 * c + 1)
|
||||
* 5. It writes the tile back to global memory
|
||||
* 6. It checks that all the values in global are properly modified.
|
||||
*/
|
||||
template <size_t smem_len, size_t num_dims>
|
||||
TEST_DEVICE_FUNC void
|
||||
test(cuda::std::array<uint32_t, num_dims> smem_coord,
|
||||
cuda::std::array<uint32_t, num_dims> smem_dims,
|
||||
cuda::std::array<uint64_t, num_dims> gmem_dims,
|
||||
int* gmem_tensor,
|
||||
int gmem_len)
|
||||
{
|
||||
CUtensorMap* global_tensor_map = reinterpret_cast<CUtensorMap*>(&global_fake_tensor_map);
|
||||
|
||||
// SETUP: fill global memory buffer
|
||||
for (int i = threadIdx.x; i < gmem_len; i += blockDim.x)
|
||||
{
|
||||
gmem_tensor[i] = i;
|
||||
}
|
||||
// Ensure that writes to global memory are visible to others, including
|
||||
// those in the async proxy.
|
||||
// ahendriksen: Issuing threadfence and fence.proxy.async.global. The
|
||||
// fence.proxy.async.global should suffice, but I am keeping the threadfence
|
||||
// out of an abundance of caution.
|
||||
__threadfence();
|
||||
ptx::fence_proxy_async(ptx::space_global);
|
||||
__syncthreads();
|
||||
|
||||
// TEST: Add i to buffer[i]
|
||||
alignas(128) __shared__ int smem_buffer[smem_len];
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ barrier bar;
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Load data:
|
||||
uint64_t token;
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
// Fastest moving coordinate first.
|
||||
cp_tensor_global_to_shared(global_tensor_map, smem_coord, smem_buffer, bar);
|
||||
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
|
||||
}
|
||||
else
|
||||
{
|
||||
token = bar.arrive();
|
||||
}
|
||||
bar.wait(cuda::std::move(token));
|
||||
|
||||
// Check smem
|
||||
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
|
||||
{
|
||||
int gmem_lin_idx = smem_lin_idx_to_gmem_lin_idx(i, smem_coord, smem_dims, gmem_dims);
|
||||
assert(smem_buffer[i] == gmem_lin_idx);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Update smem
|
||||
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
|
||||
{
|
||||
smem_buffer[i] = 2 * smem_buffer[i] + 1;
|
||||
}
|
||||
cde::fence_proxy_async_shared_cta();
|
||||
__syncthreads();
|
||||
|
||||
// Write back to global memory:
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
cp_tensor_shared_to_global(global_tensor_map, smem_coord, smem_buffer);
|
||||
cde::cp_async_bulk_commit_group();
|
||||
cde::cp_async_bulk_wait_group_read<0>();
|
||||
}
|
||||
// ahendriksen: Issuing threadfence and fence.proxy.async.global. The
|
||||
// fence.proxy.async.global should suffice, but I am keeping the threadfence
|
||||
// out of an abundance of caution.
|
||||
__threadfence();
|
||||
ptx::fence_proxy_async(ptx::space_global);
|
||||
__syncthreads();
|
||||
|
||||
// // TEAR-DOWN: check that global memory is correct
|
||||
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
|
||||
{
|
||||
int gmem_lin_idx = smem_lin_idx_to_gmem_lin_idx(i, smem_coord, smem_dims, gmem_dims);
|
||||
|
||||
assert(gmem_tensor[gmem_lin_idx] == 2 * gmem_lin_idx + 1);
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# if _CCCL_CTK_BELOW(12, 5)
|
||||
PFN_cuTensorMapEncodeTiled get_cuTensorMapEncodeTiled()
|
||||
{
|
||||
void* driver_ptr = nullptr;
|
||||
cudaDriverEntryPointQueryResult driver_status;
|
||||
auto code = cudaGetDriverEntryPoint("cuTensorMapEncodeTiled", &driver_ptr, cudaEnableDefault, &driver_status);
|
||||
assert(code == cudaSuccess && "Could not get driver API");
|
||||
return reinterpret_cast<PFN_cuTensorMapEncodeTiled>(driver_ptr);
|
||||
}
|
||||
# else // ^^^ _CCCL_CTK_BELOW(12, 5) ^^^ / vvv _CCCL_CTK_AT_LEAST(12, 5) vvv
|
||||
PFN_cuTensorMapEncodeTiled_v12000 get_cuTensorMapEncodeTiled()
|
||||
{
|
||||
void* driver_ptr = nullptr;
|
||||
cudaDriverEntryPointQueryResult driver_status;
|
||||
auto code =
|
||||
cudaGetDriverEntryPointByVersion("cuTensorMapEncodeTiled", &driver_ptr, 12000, cudaEnableDefault, &driver_status);
|
||||
assert(code == cudaSuccess && "Could not get driver API");
|
||||
return reinterpret_cast<PFN_cuTensorMapEncodeTiled_v12000>(driver_ptr);
|
||||
}
|
||||
# endif // _CCCL_CTK_AT_LEAST(12, 5)
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
template <typename T, size_t num_dims>
|
||||
CUtensorMap map_encode(T* tensor_ptr,
|
||||
const cuda::std::array<uint64_t, num_dims>& gmem_dims,
|
||||
const cuda::std::array<uint32_t, num_dims>& smem_dims)
|
||||
{
|
||||
// https://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TENSOR__MEMORY.html
|
||||
CUtensorMap tensor_map{};
|
||||
|
||||
// The stride is the number of bytes to traverse from the first element of one row to the next.
|
||||
// It must be a multiple of 16.
|
||||
// cuTensorMapEncodeTiled requires that the stride array is a valid pointer, so we add one superfluous element
|
||||
// This is necessary for num_dims == 1
|
||||
cuda::std::array<uint64_t, num_dims> stride;
|
||||
uint64_t base_stride = sizeof(T);
|
||||
for (size_t i = 0; i < stride.size() - 1; ++i)
|
||||
{
|
||||
base_stride *= gmem_dims[i];
|
||||
stride[i] = base_stride;
|
||||
}
|
||||
|
||||
// The distance between elements in units of sizeof(element). A stride of 2
|
||||
// can be used to load only the real component of a complex-valued tensor, for instance.
|
||||
cuda::std::array<uint32_t, num_dims> elem_stride; // = {1, .., 1};
|
||||
for (size_t i = 0; i < elem_stride.size(); ++i)
|
||||
{
|
||||
elem_stride[i] = 1;
|
||||
}
|
||||
|
||||
// Get a function pointer to the cuTensorMapEncodeTiled driver API.
|
||||
auto cuTensorMapEncodeTiled = get_cuTensorMapEncodeTiled();
|
||||
|
||||
// Create the tensor descriptor.
|
||||
CUresult res = cuTensorMapEncodeTiled(
|
||||
&tensor_map, // CUtensorMap *tensorMap,
|
||||
CUtensorMapDataType::CU_TENSOR_MAP_DATA_TYPE_INT32,
|
||||
num_dims, // cuuint32_t tensorRank,
|
||||
tensor_ptr, // void *globalAddress,
|
||||
gmem_dims.data(), // const cuuint64_t *globalDim,
|
||||
stride.data(), // const cuuint64_t *globalStrides,
|
||||
smem_dims.data(), // const cuuint32_t *boxDim,
|
||||
elem_stride.data(), // const cuuint32_t *elementStrides,
|
||||
CUtensorMapInterleave::CU_TENSOR_MAP_INTERLEAVE_NONE,
|
||||
CUtensorMapSwizzle::CU_TENSOR_MAP_SWIZZLE_NONE,
|
||||
CUtensorMapL2promotion::CU_TENSOR_MAP_L2_PROMOTION_NONE,
|
||||
CUtensorMapFloatOOBfill::CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE);
|
||||
|
||||
assert(res == CUDA_SUCCESS && "tensormap creation failed.");
|
||||
|
||||
return tensor_map;
|
||||
}
|
||||
|
||||
template <typename T, size_t num_dims>
|
||||
void init_tensor_map(const T& gmem_tensor_symbol,
|
||||
const cuda::std::array<uint64_t, num_dims>& gmem_dims,
|
||||
const cuda::std::array<uint32_t, num_dims>& smem_dims)
|
||||
{
|
||||
// Get pointer to gmem_tensor to create tensor map.
|
||||
int* tensor_ptr = nullptr;
|
||||
auto code = cudaGetSymbolAddress((void**) &tensor_ptr, gmem_tensor_symbol);
|
||||
assert(code == cudaSuccess && "Could not get symbol address.");
|
||||
|
||||
// Create tensor map
|
||||
CUtensorMap local_tensor_map = map_encode(tensor_ptr, gmem_dims, smem_dims);
|
||||
|
||||
// Copy it to device
|
||||
code = cudaMemcpyToSymbol(global_fake_tensor_map, &local_tensor_map, sizeof(CUtensorMap));
|
||||
assert(code == cudaSuccess && "Could not copy symbol to device.");
|
||||
}
|
||||
#endif // ! TEST_COMPILER(NVRTC)
|
||||
|
||||
#endif // TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 256;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: no_execute
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
[[maybe_unused]] TEST_GLOBAL_VARIABLE uint64_t bar_storage;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(cuda::barrier<cuda::thread_scope_block> * bar_ptr;
|
||||
bar_ptr = reinterpret_cast<cuda::barrier<cuda::thread_scope_block>*>(bar_storage);
|
||||
|
||||
if (threadIdx.x == 0) { init(bar_ptr, blockDim.x); } __syncthreads();
|
||||
|
||||
// Should fail because the barrier is in device memory.
|
||||
cuda::device::barrier_expect_tx(*bar_ptr, 1);));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 2;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 32;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: pre-sm-70
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "cuda_space_selector.h"
|
||||
|
||||
template <cuda::thread_scope Sco, template <typename, typename> class BarrierSelector>
|
||||
TEST_FUNC void test()
|
||||
{
|
||||
cuda::barrier<Sco> b(3);
|
||||
|
||||
init(&b, 2);
|
||||
|
||||
auto token = b.arrive();
|
||||
b.arrive_and_wait();
|
||||
b.wait(std::move(token));
|
||||
}
|
||||
|
||||
template <cuda::thread_scope Sco>
|
||||
TEST_FUNC void test_select_barrier()
|
||||
{
|
||||
test<Sco, local_memory_selector>();
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (test<Sco, shared_memory_selector>(); test<Sco, global_memory_selector>();))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_select_barrier<cuda::thread_scope_system>();
|
||||
test_select_barrier<cuda::thread_scope_device>();
|
||||
test_select_barrier<cuda::thread_scope_block>();
|
||||
test_select_barrier<cuda::thread_scope_thread>();
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: pre-sm-80
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
TEST_NV_DIAG_SUPPRESS(set_but_not_used)
|
||||
|
||||
TEST_DEVICE_FUNC void test()
|
||||
{
|
||||
__shared__ cuda::barrier<cuda::thread_scope_block>* b;
|
||||
shared_memory_selector<cuda::barrier<cuda::thread_scope_block>, constructor_initializer> sel;
|
||||
b = sel.construct(2);
|
||||
|
||||
[[maybe_unused]] uint64_t token;
|
||||
asm volatile("mbarrier.arrive.b64 %0, [%1];" : "=l"(token) : "l"(cuda::device::barrier_native_handle(*b)) : "memory");
|
||||
|
||||
b->arrive_and_wait();
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_80, test();)
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/bit>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
using nl = cuda::std::numeric_limits<T>;
|
||||
constexpr T all_ones = static_cast<T>(~T{0});
|
||||
constexpr T half_low = all_ones >> (nl::digits / 2u);
|
||||
constexpr T half_high = static_cast<T>(all_ones << (nl::digits / 2u));
|
||||
static_assert(cuda::bit_reverse(all_ones) == all_ones);
|
||||
static_assert(cuda::bit_reverse(T{0}) == T{0});
|
||||
static_assert(cuda::bit_reverse(half_low) == half_high);
|
||||
static_assert(cuda::bit_reverse(T{0b11001001}) == (T{0b10010011} << (nl::digits - 8u)));
|
||||
static_assert(cuda::bit_reverse(T{T{0b10010011} << (nl::digits - 8u)}) == T{0b11001001});
|
||||
unused(all_ones);
|
||||
unused(half_low);
|
||||
unused(half_high);
|
||||
return true;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
test<unsigned char>();
|
||||
test<unsigned short>();
|
||||
test<unsigned>();
|
||||
test<unsigned long>();
|
||||
test<unsigned long long>();
|
||||
|
||||
test<uint8_t>();
|
||||
test<uint16_t>();
|
||||
test<uint32_t>();
|
||||
test<uint64_t>();
|
||||
test<size_t>();
|
||||
test<uintmax_t>();
|
||||
test<uintptr_t>();
|
||||
|
||||
#if _CCCL_HAS_INT128()
|
||||
test<__uint128_t>();
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
assert(test());
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/bit>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
using T = uint32_t;
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, -1, 1));
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, 0, -1));
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, 0, 33));
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, 32, 1));
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, 20, 20));
|
||||
|
||||
static_assert(cuda::bitfield_extract(T{0}, -1, 1));
|
||||
static_assert(cuda::bitfield_extract(T{0}, 0, -1));
|
||||
static_assert(cuda::bitfield_extract(T{0}, 0, 33));
|
||||
static_assert(cuda::bitfield_extract(T{0}, 32, 1));
|
||||
static_assert(cuda::bitfield_extract(T{0}, 20, 20));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/bit>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
using nl = cuda::std::numeric_limits<T>;
|
||||
constexpr T all_ones = static_cast<T>(~T{0});
|
||||
unused(all_ones);
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, 0, 1) == 1);
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, 1, 1) == 0b10);
|
||||
assert(cuda::bitfield_insert(T{0b10}, all_ones, 0, 1) == 0b11);
|
||||
assert(cuda::bitfield_insert(all_ones, all_ones, 0, 0) == all_ones);
|
||||
assert(cuda::bitfield_insert(all_ones, all_ones, 0, 1) == all_ones);
|
||||
assert(cuda::bitfield_insert(all_ones, all_ones, 2, 1) == all_ones);
|
||||
assert(cuda::bitfield_insert(all_ones, T{0b1000}, 1, 2) == (all_ones & static_cast<T>(~T{0b110})));
|
||||
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, 0, 2) == 0b11);
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, 3, 2) == 0b11000);
|
||||
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, 3, 2) == 0b10111000);
|
||||
assert(cuda::bitfield_insert(T{0b10100000}, T{0b11}, 3, 2) == 0b10111000);
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, nl::digits - 1, 1) == (T{1} << (nl::digits - 1u)));
|
||||
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, 0, nl::digits) == all_ones);
|
||||
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, nl::digits, 0) == T{0b10100000});
|
||||
|
||||
assert(cuda::bitfield_extract(T{0}, 3, 4) == 0);
|
||||
assert(cuda::bitfield_extract(T{0b1011}, 0, 1) == 1);
|
||||
assert(cuda::bitfield_extract(T{0b1011}, 1, 1) == 1);
|
||||
assert(cuda::bitfield_extract(T{0b1011}, 2, 2) == 0b10);
|
||||
assert(cuda::bitfield_extract(all_ones, 0, 0) == 0);
|
||||
assert(cuda::bitfield_extract(all_ones, 0, 4) == 0b1111);
|
||||
assert(cuda::bitfield_extract(all_ones, 2, 4) == 0b1111);
|
||||
|
||||
assert(cuda::bitfield_extract(T{0b1010010}, 0, 2) == 0b10);
|
||||
assert(cuda::bitfield_extract(T{0b10101100}, 3, 2) == 1);
|
||||
assert(cuda::bitfield_extract(T{0b10100000}, 3, 3) == 0b100);
|
||||
|
||||
assert(cuda::bitfield_extract(T{all_ones}, nl::digits - 1, 1) == 1);
|
||||
assert(cuda::bitfield_extract(T{0b10100000}, 0, nl::digits) == T{0b10100000});
|
||||
assert(cuda::bitfield_extract(T{0b10100000}, nl::digits, 0) == 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
test<unsigned char>();
|
||||
test<unsigned short>();
|
||||
test<unsigned>();
|
||||
test<unsigned long>();
|
||||
test<unsigned long long>();
|
||||
|
||||
test<uint8_t>();
|
||||
test<uint16_t>();
|
||||
test<uint32_t>();
|
||||
test<uint64_t>();
|
||||
test<size_t>();
|
||||
test<uintmax_t>();
|
||||
test<uintptr_t>();
|
||||
|
||||
#if _CCCL_HAS_INT128()
|
||||
test<__uint128_t>();
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
assert(test());
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user