[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,34 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""GDB entry point for CUDA C++ Core Libraries pretty printers.
Requires Python 3.12 or newer.
"""
from __future__ import annotations
import sys
from pathlib import Path
import gdb
_SCRIPT_DIRECTORY = str(Path(__file__).resolve().parent)
if _SCRIPT_DIRECTORY not in sys.path:
sys.path.insert(0, _SCRIPT_DIRECTORY)
import buffer # noqa: E402
import memory_resource # noqa: E402
import std_array # noqa: E402
_PRINTERS = (memory_resource, buffer, std_array)
def register() -> None:
"""Register every CCCL GDB pretty printer."""
for printer in _PRINTERS:
printer.register(gdb)
register()

View File

@@ -0,0 +1,127 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""GDB pretty printer for cuda::buffer."""
from __future__ import annotations
from collections.abc import Iterator
from types import ModuleType
import memory_resource
import gdb
import gdb.printing
# GDB sees cudaMemcpyKind through cudaMemcpy's declaration, but its expression
# parser does not necessarily import the enum constants. This is the value of
# cudaMemcpyDefault from the CUDA Runtime API.
_CUDA_MEMCPY_DEFAULT = 4
def _template_name(value_type: gdb.Type) -> str:
return str(value_type).split("<", 1)[0]
def _is_cuda_buffer(value_type: gdb.Type) -> bool:
# strip_typedefs resolves aliases that can hide accessibility properties.
value_type = value_type.strip_typedefs().unqualified()
template_name = _template_name(value_type)
return (
template_name.startswith("cuda::")
and template_name.rsplit("::", 1)[-1] == "buffer"
)
class BufferPrinter:
"""Expose cuda::buffer metadata and elements to GDB."""
def __init__(self, value: gdb.Value) -> None:
self.value = value
self.type = value.type.strip_typedefs().unqualified()
self.type_name = memory_resource.public_type_name(self.type)
self.value_type = self.type.template_argument(0)
storage = value["__buf_"]
self.memory_resource = storage["__mr_"]
self.stream = int(storage["__stream_"]["__stream"])
self.size = int(storage["__count_"])
self.alignment = int(storage["__alignment_"])
raw_address = int(storage["__buf_"])
self.data_address = (raw_address + self.alignment - 1) & ~(self.alignment - 1)
host_accessible = "host_accessible" in self.type_name
device_accessible = "device_accessible" in self.type_name
if host_accessible and device_accessible:
self.accessibility = "host/device"
elif device_accessible:
self.accessibility = "device"
elif host_accessible:
self.accessibility = "host"
else:
self.accessibility = "unknown"
self.host_copy: gdb.Value | None = None
self._copy_to_host()
def __del__(self) -> None:
self.clear()
def clear(self) -> None:
"""Release state staged in the inferior for synthetic children."""
if self.host_copy is None:
return
try:
gdb.parse_and_eval(f"(void)free((void*){int(self.host_copy):#x})")
except gdb.error:
pass
self.host_copy = None
def _copy_to_host(self) -> None:
if self.size == 0:
return
byte_count = self.size * self.value_type.sizeof
self.host_copy = gdb.parse_and_eval(f"(void*)malloc({byte_count})")
host_address = int(self.host_copy)
status = gdb.parse_and_eval(
"(int)cudaMemcpy((void*)"
f"{host_address:#x}, (const void*){self.data_address:#x}, {byte_count}, "
f"{_CUDA_MEMCPY_DEFAULT})"
)
if int(status) != 0:
self.clear()
def children(self) -> Iterator[tuple[str, gdb.Value]]:
if self.host_copy is None:
return
pointer = self.host_copy.cast(self.value_type.pointer())
for index in range(self.size):
yield f"[{index}]", (pointer + index).dereference()
def to_string(self) -> str:
resource = memory_resource.memory_resource_description(self.memory_resource)
return (
f"{self.type_name} mr={resource}, stream={self.stream:#x}, "
f"size={self.size}, align={self.alignment}, "
f"data={self.data_address:#x} ({self.accessibility})"
)
class BufferPrinterLookup(gdb.printing.PrettyPrinter):
"""Select the cuda::buffer printer by its public class name."""
def __init__(self) -> None:
super().__init__("cuda::buffer")
def __call__(self, value: gdb.Value) -> BufferPrinter | None:
if _is_cuda_buffer(value.type):
return BufferPrinter(value)
return None
def register(objfile: ModuleType) -> None:
"""Register the cuda::buffer printer with GDB."""
gdb.printing.register_pretty_printer(objfile, BufferPrinterLookup(), replace=True)

View File

@@ -0,0 +1,76 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""GDB pretty printer for CUDA type-erased memory resources."""
from __future__ import annotations
import re
from types import ModuleType
import gdb
import gdb.printing
_ABI_NAMESPACE_PATTERN = re.compile(r"::__(?:\d+|version_bump_ver\d+_)(?=::)")
_RESOURCE_NAMES = frozenset(
{"any_resource", "any_synchronous_resource", "basic_any_resource"}
)
def public_type_name(value_type: gdb.Type) -> str:
"""Return a type name without CUDA ABI inline namespaces."""
return _ABI_NAMESPACE_PATTERN.sub("", str(value_type))
def _template_name(value_type: gdb.Type) -> str:
return str(value_type).split("<", 1)[0]
def _is_memory_resource(value_type: gdb.Type) -> bool:
value_type = value_type.strip_typedefs().unqualified()
type_name = public_type_name(value_type)
template_name = _template_name(value_type)
return (
type_name.startswith("cuda::mr::")
and template_name.rsplit("::", 1)[-1] in _RESOURCE_NAMES
)
def memory_resource_description(value: gdb.Value) -> str:
value_type = value.type.strip_typedefs().unqualified()
type_name = public_type_name(value_type)
try:
address = int(value.address)
except (gdb.error, TypeError):
return type_name
return f"{type_name} @ {address:#x}"
class MemoryResourcePrinter:
"""Summarize a CUDA type-erased memory resource."""
def __init__(self, value: gdb.Value) -> None:
self.value = value
def to_string(self) -> str:
return memory_resource_description(self.value)
class MemoryResourcePrinterLookup(gdb.printing.PrettyPrinter):
"""Select printers for public CUDA type-erased resource types."""
def __init__(self) -> None:
super().__init__("cuda::mr::any_resource")
def __call__(self, value: gdb.Value) -> MemoryResourcePrinter | None:
if _is_memory_resource(value.type):
return MemoryResourcePrinter(value)
return None
def register(objfile: ModuleType) -> None:
"""Register CUDA memory-resource formatters with GDB."""
gdb.printing.register_pretty_printer(
objfile, MemoryResourcePrinterLookup(), replace=True
)

View File

@@ -0,0 +1,63 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""GDB pretty printer for cuda::std::array."""
from __future__ import annotations
from collections.abc import Iterator
from types import ModuleType
import memory_resource
import gdb
import gdb.printing
def _template_name(value_type: gdb.Type) -> str:
return str(value_type).split("<", 1)[0]
def _is_cuda_array(value_type: gdb.Type) -> bool:
value_type = value_type.strip_typedefs().unqualified()
template_name = _template_name(value_type)
return (
template_name.startswith("cuda::std::")
and template_name.rsplit("::", 1)[-1] == "array"
)
class ArrayPrinter:
"""Expose cuda::std::array metadata and elements to GDB."""
def __init__(self, value: gdb.Value) -> None:
self.value = value
self.type = value.type.strip_typedefs().unqualified()
self.type_name = memory_resource.public_type_name(self.type)
self.size = int(self.type.template_argument(1))
def children(self) -> Iterator[tuple[str, gdb.Value]]:
elems = self.value["__elems_"]
for index in range(self.size):
yield f"[{index}]", elems[index]
def to_string(self) -> str:
return self.type_name
class ArrayPrinterLookup(gdb.printing.PrettyPrinter):
"""Select the cuda::std::array printer by its public class name."""
def __init__(self) -> None:
super().__init__("cuda::std::array")
def __call__(self, value: gdb.Value) -> ArrayPrinter | None:
if _is_cuda_array(value.type):
return ArrayPrinter(value)
return None
def register(objfile: ModuleType) -> None:
"""Register the cuda::std::array printer with GDB."""
gdb.printing.register_pretty_printer(objfile, ArrayPrinterLookup(), replace=True)

View File

@@ -0,0 +1,28 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""LLDB entry point for CUDA C++ Core Libraries pretty printers.
Requires Python 3.12 or newer.
"""
from __future__ import annotations
import buffer
import memory_resource
import std_array
import lldb
_CATEGORY = "cccl"
_FORMATTERS = (memory_resource, buffer, std_array)
InternalDict = dict[str, object]
def __lldb_init_module(debugger: lldb.SBDebugger, _internal_dict: InternalDict) -> None:
debugger.HandleCommand(f"type category define {_CATEGORY}")
for formatter in _FORMATTERS:
module = f"{__name__}.{formatter.__name__.rsplit('.', 1)[-1]}"
formatter.register(debugger, _CATEGORY, module)
debugger.HandleCommand(f"type category enable {_CATEGORY}")

View File

@@ -0,0 +1,219 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""LLDB pretty printer for cuda::buffer."""
from __future__ import annotations
import re
from typing import NamedTuple
import memory_resource
import lldb
_BUFFER_PATTERN = re.compile(r"^cuda::buffer<.+>$")
# LLDB sees cudaMemcpyKind through cudaMemcpy's declaration, but its expression
# parser does not necessarily import the enum constants. This is the value of
# cudaMemcpyDefault from the CUDA Runtime API.
_CUDA_MEMCPY_DEFAULT = 4
InternalDict = dict[str, object]
class BufferInfo(NamedTuple):
size: int
data_address: int
value_type: lldb.SBType
accessibility: str
memory_resource: lldb.SBValue
stream: lldb.SBValue
alignment: lldb.SBValue
def is_cuda_buffer(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
type_name = (
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
)
return _BUFFER_PATTERN.fullmatch(type_name) is not None
def _buffer_info(value: lldb.SBValue) -> BufferInfo | None:
value = value.GetNonSyntheticValue()
storage = value.GetChildMemberWithName("__buf_")
if not storage.IsValid():
return None
count = storage.GetChildMemberWithName("__count_")
memory_resource = storage.GetChildMemberWithName("__mr_")
stream_ref = storage.GetChildMemberWithName("__stream_")
stream = stream_ref.GetChildMemberWithName("__stream")
alignment = storage.GetChildMemberWithName("__alignment_")
allocation = storage.GetChildMemberWithName("__buf_")
if not all(
child.IsValid()
for child in (count, memory_resource, stream, alignment, allocation)
):
return None
# A source-level alias can hide the accessibility properties from
# GetTypeName(), so use the canonical public type for all property checks.
buffer_type = value.GetType().GetCanonicalType().GetUnqualifiedType()
value_type = buffer_type.GetTemplateArgumentType(0)
if not value_type.IsValid():
return None
type_name = buffer_type.GetDisplayTypeName() or ""
host_accessible = "host_accessible" in type_name
device_accessible = "device_accessible" in type_name
if host_accessible and device_accessible:
accessibility = "host/device"
elif device_accessible:
accessibility = "device"
elif host_accessible:
accessibility = "host"
else:
accessibility = "unknown"
size = count.GetValueAsUnsigned(0)
align = alignment.GetValueAsUnsigned(1)
raw_address = allocation.GetValueAsUnsigned(0)
data_address = (raw_address + align - 1) & ~(align - 1)
return BufferInfo(
size,
data_address,
value_type,
accessibility,
memory_resource,
stream,
alignment,
)
def buffer_summary(value: lldb.SBValue, _internal_dict: InternalDict) -> str | None:
info = _buffer_info(value)
if info is None:
return None
resource = memory_resource.memory_resource_description(info.memory_resource)
stream = info.stream.GetValueAsUnsigned(0)
alignment = info.alignment.GetValueAsUnsigned(0)
return (
f"mr={resource}, stream={stream:#x}, size={info.size}, align={alignment}, "
f"data={info.data_address:#x} ({info.accessibility})"
)
class BufferSyntheticProvider:
"""Expose cuda::buffer elements as LLDB synthetic children."""
def __init__(self, value: lldb.SBValue, _internal_dict: InternalDict) -> None:
self.value = value.GetNonSyntheticValue()
self.host_copy = lldb.SBValue()
self.clear()
self.update()
def __del__(self) -> None:
self.clear()
def _evaluate(self, expression: str) -> lldb.SBValue:
frame = self.value.GetFrame()
if not frame.IsValid():
return lldb.SBValue()
options = lldb.SBExpressionOptions()
options.SetIgnoreBreakpoints(True)
options.SetUnwindOnError(True)
return frame.EvaluateExpression(expression, options)
def clear(self) -> None:
"""Release the staged copy and reset all cached buffer information."""
if self.host_copy.IsValid():
address = self.host_copy.GetValueAsUnsigned(0)
if address:
self._evaluate(f"(void)free((void*){address:#x})")
self.host_copy = lldb.SBValue()
self.size = 0
self.data_address = 0
self.value_type = lldb.SBType()
self.value_size = 0
def _copy_to_host(self) -> bool:
if self.size == 0:
return True
byte_count = self.size * self.value_size
self.host_copy = self._evaluate(f"(void*)malloc({byte_count})")
if not self.host_copy.IsValid() or self.host_copy.GetError().Fail():
return False
host_address = self.host_copy.GetValueAsUnsigned(0)
result = self._evaluate(
"(int)cudaMemcpy((void*)"
f"{host_address:#x}, (const void*){self.data_address:#x}, {byte_count}, "
f"{_CUDA_MEMCPY_DEFAULT})"
)
if (
not result.IsValid()
or result.GetError().Fail()
or result.GetValueAsSigned(-1) != 0
):
self.clear()
return False
return True
def update(self) -> bool:
self.clear()
info = _buffer_info(self.value)
if info is None:
return False
self.size = info.size
self.data_address = info.data_address
self.value_type = info.value_type
self.value_size = self.value_type.GetByteSize()
self._copy_to_host()
return True
def num_children(self) -> int:
return self.size
def has_children(self) -> bool:
return self.size != 0
def get_type_name(self) -> str:
# STL element access can preserve an alloc_traits::value_type typedef.
# Report the canonical display name so LLDB shows cuda::buffer instead.
return (
self.value.GetType()
.GetCanonicalType()
.GetUnqualifiedType()
.GetDisplayTypeName()
or ""
)
def get_child_index(self, name: str) -> int:
if name.startswith("[") and name.endswith("]"):
try:
return int(name[1:-1])
except ValueError:
pass
return -1
def get_child_at_index(self, index: int) -> lldb.SBValue | None:
if index < 0:
return None
if index >= self.size:
return None
offset = index * self.value_size
return self.host_copy.CreateChildAtOffset(f"[{index}]", offset, self.value_type)
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
"""Register the cuda::buffer formatter in an LLDB category."""
debugger.HandleCommand(
f"type summary add --category {category} --expand --python-function {module}.buffer_summary "
f"--recognizer-function {module}.is_cuda_buffer"
)
debugger.HandleCommand(
f"type synthetic add --category {category} --python-class {module}.BufferSyntheticProvider "
f"--recognizer-function {module}.is_cuda_buffer"
)

View File

@@ -0,0 +1,50 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""LLDB pretty printer for CUDA type-erased memory resources."""
from __future__ import annotations
import re
import lldb
_RESOURCE_PATTERN = re.compile(
r"^cuda::mr::(?:basic_any_resource|any_resource|any_synchronous_resource)<.+>$"
)
InternalDict = dict[str, object]
def is_memory_resource(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
type_name = (
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
)
return _RESOURCE_PATTERN.fullmatch(type_name) is not None
def memory_resource_description(value: lldb.SBValue) -> str:
"""Describe a type-erased resource using only public type information."""
type_name = (
value.GetType().GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName()
)
if not type_name:
type_name = "type-erased resource"
address = value.GetLoadAddress()
if address == lldb.LLDB_INVALID_ADDRESS:
return type_name
return f"{type_name} @ {address:#x}"
def memory_resource_summary(value: lldb.SBValue, _internal_dict: InternalDict) -> str:
return memory_resource_description(value)
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
"""Register CUDA memory-resource formatters in an LLDB category."""
debugger.HandleCommand(
f"type summary add --category {category} --python-function "
f"{module}.memory_resource_summary --recognizer-function "
f"{module}.is_memory_resource"
)

View File

@@ -0,0 +1,78 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""LLDB pretty printer for cuda::std::array."""
from __future__ import annotations
import re
import lldb
_ARRAY_PATTERN = re.compile(r"^cuda::std::array<.+,\s*(\d+)>$")
InternalDict = dict[str, object]
def is_cuda_array(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
type_name = (
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
)
return _ARRAY_PATTERN.fullmatch(type_name) is not None
class ArraySyntheticProvider:
"""Expose cuda::std::array elements as LLDB synthetic children."""
def __init__(self, value: lldb.SBValue, _internal_dict: InternalDict) -> None:
self.value = value.GetNonSyntheticValue()
self.update()
def update(self) -> bool:
type_name = (
self.value.GetType()
.GetCanonicalType()
.GetUnqualifiedType()
.GetDisplayTypeName()
or ""
)
self.type_name = type_name
match = _ARRAY_PATTERN.fullmatch(type_name)
self.elems = self.value.GetChildMemberWithName("__elems_")
self.size = 0
if not self.elems.IsValid() or not match:
return False
self.size = int(match.group(1))
return True
def num_children(self) -> int:
return self.size
def has_children(self) -> bool:
return self.size != 0
def get_type_name(self) -> str:
return self.type_name
def get_child_index(self, name: str) -> int:
if name.startswith("[") and name.endswith("]"):
try:
return int(name[1:-1])
except ValueError:
pass
return -1
def get_child_at_index(self, index: int) -> lldb.SBValue | None:
if index < 0:
return None
if index >= self.size:
return None
return self.elems.GetChildAtIndex(index)
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
"""Register the cuda::std::array formatter in an LLDB category."""
debugger.HandleCommand(
f"type synthetic add --category {category} --python-class {module}.ArraySyntheticProvider "
f"--recognizer-function {module}.is_cuda_array"
)