[MUH] Bootstrap muh toolchain — extract/parse/gen_yaml/gen_patch + baseline.muh
Pipeline: 1. extract.py: Parses all 26 CCCL tuning_*.cuh → 26 YAML schemas in muh/schema/ 2. parse.py: .muh file parser with extends-inheritance + schema validation 3. gen_yaml.py: .muh → computility-run.yaml (verified: matches competition reference) 4. gen_patch.py: .muh → vllm kernel unified diff patches (6 algorithm mappings) 5. baseline.muh: Competition reference config, all tuning values pending BI-V100 benchmarks Schemas extracted: 26 algorithms, 8-19 params each, SM75/80/90/100 reference tunings Priority mapping: reduce→attention, topk→sampling, scan→paged_attention, transform→activations, batch_memcpy→KV_cache, for→RoPE Tested: extract→parse→validate→gen_yaml→gen_patch full pipeline passes
This commit is contained in:
1
.gitignore
vendored
Normal file
1
.gitignore
vendored
Normal file
@@ -0,0 +1 @@
|
||||
muh/__pycache__/
|
||||
122
baseline.muh
Normal file
122
baseline.muh
Normal file
@@ -0,0 +1,122 @@
|
||||
# baseline.muh — Competition reference configuration
|
||||
# Corresponds to: dev.modelhub.org.cn EngineX-Iluvatar/enginex-vllm-bi100-qwen36
|
||||
#
|
||||
# This is the starting point. All tuning values are pending BI-V100 benchmarks.
|
||||
# Child .muh files use 'extends: baseline.muh' to override specific algorithms.
|
||||
|
||||
# --- Hardware description ---
|
||||
hardware:
|
||||
name: Iluvatar-BI-V100-50c-200G
|
||||
gpu_count: 4
|
||||
# These need to be confirmed on actual hardware:
|
||||
warp_size: 32
|
||||
max_threads_per_block: 1024
|
||||
max_shared_memory_per_block: 49152
|
||||
max_registers_per_thread: 255
|
||||
l2_cache_size_bytes: 6291456
|
||||
memory_bandwidth_gbps: 900
|
||||
compute_capability: iluvatar_bi100
|
||||
|
||||
# --- vllm launch configuration ---
|
||||
# Maps directly to computility-run.yaml command
|
||||
vllm:
|
||||
model_path: /model
|
||||
served_model_name: llm
|
||||
max_model_len: 100000
|
||||
gpu_memory_utilization: 0.9
|
||||
tensor_parallel: 4
|
||||
max_num_seqs: 1
|
||||
max_num_batched_tokens: 8192
|
||||
max_seq_len_to_capture: 32768
|
||||
trust_remote_code: true
|
||||
disable_log_requests: true
|
||||
disable_frontend_multiprocessing: true
|
||||
enable_chunked_prefill: true
|
||||
enable_auto_tool_choice: true
|
||||
tool_call_parser: qwen3_coder
|
||||
reasoning_parser: qwen3
|
||||
enable_prefix_caching: true
|
||||
|
||||
# --- Concurrency ---
|
||||
concurrency: 1
|
||||
|
||||
# --- Environment ---
|
||||
env:
|
||||
VLLM_ENGINE_ITERATION_TIMEOUT_S: 3600
|
||||
|
||||
# --- Tuning overrides (per CCCL algorithm) ---
|
||||
# Each key corresponds to a tuning_*.cuh schema in muh/schema/
|
||||
# Values are TBD until we run benchmarks on BI-V100
|
||||
#
|
||||
# Priority order (by competition score impact):
|
||||
# 1. reduce — attention reduction (Output TPS × 16.796)
|
||||
# 2. topk — sampling top-k/top-p (Output TPS × 16.796)
|
||||
# 3. scan — prefix scan in paged attention
|
||||
# 4. transform — activation kernels (SiLU, GELU)
|
||||
# 5. batch_memcpy — KV cache management (Cache TPS × 0.56)
|
||||
# 6. for — RoPE position encoding
|
||||
|
||||
tuning:
|
||||
reduce:
|
||||
_priority: P0
|
||||
_vllm_impact: attention_reduction
|
||||
_score_weight: Output TPS × 16.796
|
||||
# CCCL SM90 reference: threads=128, items=24, vec_size=4
|
||||
# CCCL SM100 reference: threads varies by accum_size
|
||||
threads_per_block: null
|
||||
items_per_thread: null
|
||||
vec_size: null
|
||||
|
||||
topk:
|
||||
_priority: P0
|
||||
_vllm_impact: sampling_decode
|
||||
_score_weight: Output TPS × 16.796
|
||||
# CCCL reference: threads=512, items=4, bits_per_pass=11
|
||||
threads_per_block: null
|
||||
items_per_thread: null
|
||||
bits_per_pass: null
|
||||
|
||||
scan:
|
||||
_priority: P0
|
||||
_vllm_impact: paged_attention_prefix_scan
|
||||
_score_weight: Input TPS × 2.799
|
||||
# CCCL SM90 lookback: threads=128, items=24, delay=fixed(688, 1140) for float32
|
||||
# CCCL SM100 lookback: threads=384, items=22, delay=exponential_backon(1904, 830)
|
||||
# CCCL SM100 lookahead: warps=4, items=80-1, lookahead_items=3
|
||||
threads_per_block: null
|
||||
items_per_thread: null
|
||||
load_algorithm: null
|
||||
store_algorithm: null
|
||||
scan_algorithm: null
|
||||
|
||||
transform:
|
||||
_priority: P1
|
||||
_vllm_impact: activation_elementwise
|
||||
_score_weight: Output TPS × 16.796
|
||||
threads_per_block: null
|
||||
items_per_thread: null
|
||||
|
||||
batch_memcpy:
|
||||
_priority: P1
|
||||
_vllm_impact: kv_cache_copy
|
||||
_score_weight: Cache TPS × 0.56
|
||||
threads_per_block: null
|
||||
|
||||
for:
|
||||
_priority: P2
|
||||
_vllm_impact: rope_position_encoding
|
||||
threads_per_block: null
|
||||
items_per_thread: null
|
||||
|
||||
radix_sort:
|
||||
_priority: P2
|
||||
_vllm_impact: beam_search_token_sort
|
||||
threads_per_block: null
|
||||
items_per_thread: null
|
||||
radix_bits: null
|
||||
|
||||
merge:
|
||||
_priority: P2
|
||||
_vllm_impact: sequence_merging
|
||||
threads_per_block: null
|
||||
items_per_thread: null
|
||||
285
muh/extract.py
Normal file
285
muh/extract.py
Normal file
@@ -0,0 +1,285 @@
|
||||
#!/usr/bin/env python3
|
||||
"""muh/extract.py — Extract CCCL tuning parameter spaces into muh/schema/*.yaml
|
||||
|
||||
Reads cccl_upstream/cub/cub/device/dispatch/tuning/tuning_*.cuh,
|
||||
parses policy struct fields and SM-specific tuning values,
|
||||
outputs one YAML file per algorithm under muh/schema/.
|
||||
|
||||
Usage:
|
||||
python3 muh/extract.py [--cccl-root cccl_upstream] [--out-dir muh/schema]
|
||||
"""
|
||||
|
||||
import re
|
||||
import os
|
||||
import sys
|
||||
import glob
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
from collections import OrderedDict
|
||||
|
||||
# --- Enum value sets (from CCCL headers) ---
|
||||
|
||||
BLOCK_LOAD_ALGORITHMS = [
|
||||
"BLOCK_LOAD_DIRECT",
|
||||
"BLOCK_LOAD_VECTORIZE",
|
||||
"BLOCK_LOAD_TRANSPOSE",
|
||||
"BLOCK_LOAD_WARP_TRANSPOSE",
|
||||
"BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED",
|
||||
"BLOCK_LOAD_STRIPED",
|
||||
]
|
||||
|
||||
BLOCK_STORE_ALGORITHMS = [
|
||||
"BLOCK_STORE_DIRECT",
|
||||
"BLOCK_STORE_WARP_TRANSPOSE",
|
||||
"BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED",
|
||||
"BLOCK_STORE_STRIPED",
|
||||
]
|
||||
|
||||
BLOCK_REDUCE_ALGORITHMS = [
|
||||
"BLOCK_REDUCE_RAKING",
|
||||
"BLOCK_REDUCE_RAKING_COMMUTATIVE_ONLY",
|
||||
"BLOCK_REDUCE_WARP_REDUCTIONS",
|
||||
]
|
||||
|
||||
BLOCK_SCAN_ALGORITHMS = [
|
||||
"BLOCK_SCAN_RAKING",
|
||||
"BLOCK_SCAN_RAKING_MEMOIZE",
|
||||
"BLOCK_SCAN_WARP_SCANS",
|
||||
]
|
||||
|
||||
CACHE_LOAD_MODIFIERS = [
|
||||
"LOAD_DEFAULT",
|
||||
"LOAD_CA",
|
||||
"LOAD_CG",
|
||||
"LOAD_CS",
|
||||
"LOAD_CV",
|
||||
"LOAD_LDG",
|
||||
]
|
||||
|
||||
LOOKBACK_DELAY_ALGORITHMS = [
|
||||
"no_delay",
|
||||
"fixed_delay",
|
||||
"exponential_backoff",
|
||||
"exponential_backoff_jitter",
|
||||
"exponential_backoff_jitter_window",
|
||||
"exponential_backon_jitter_window",
|
||||
"exponential_backon_jitter",
|
||||
"exponential_backon",
|
||||
]
|
||||
|
||||
# --- Field type → range/enum mapping ---
|
||||
|
||||
FIELD_TYPES = {
|
||||
"threads_per_block": {"type": "int", "range": [32, 1024], "step": 32},
|
||||
"items_per_thread": {"type": "int", "range": [1, 32], "step": 1},
|
||||
"vec_size": {"type": "int", "range": [1, 8], "step": 1},
|
||||
"bits_per_pass": {"type": "int", "range": [4, 11], "step": 1},
|
||||
"radix_bits": {"type": "int", "range": [4, 8], "step": 1},
|
||||
"load_algorithm": {"type": "enum", "values": BLOCK_LOAD_ALGORITHMS},
|
||||
"store_algorithm": {"type": "enum", "values": BLOCK_STORE_ALGORITHMS},
|
||||
"reduce_algorithm": {"type": "enum", "values": BLOCK_REDUCE_ALGORITHMS},
|
||||
"scan_algorithm": {"type": "enum", "values": BLOCK_SCAN_ALGORITHMS},
|
||||
"load_modifier": {"type": "enum", "values": CACHE_LOAD_MODIFIERS},
|
||||
"lookback_delay.kind": {"type": "enum", "values": LOOKBACK_DELAY_ALGORITHMS},
|
||||
"lookback_delay.delay": {"type": "int", "range": [0, 2000], "step": 50},
|
||||
"lookback_delay.l2_write_latency": {"type": "int", "range": [0, 2000], "step": 50},
|
||||
"reduce_and_scan_warps": {"type": "int", "range": [1, 8], "step": 1},
|
||||
"lookahead_items_per_thread": {"type": "int", "range": [1, 16], "step": 1},
|
||||
}
|
||||
|
||||
|
||||
def extract_policy_fields(content, filename):
|
||||
"""Extract policy struct field names from a tuning file."""
|
||||
fields = []
|
||||
# Match lines like: int threads_per_block; or BlockLoadAlgorithm load_algorithm;
|
||||
pattern = re.compile(
|
||||
r'^\s+(?:int|BlockLoadAlgorithm|BlockStoreAlgorithm|BlockReduceAlgorithm|'
|
||||
r'BlockScanAlgorithm|CacheLoadModifier|LookbackDelayPolicy)\s+'
|
||||
r'(\w+)\s*[;=]',
|
||||
re.MULTILINE
|
||||
)
|
||||
for m in pattern.finditer(content):
|
||||
field = m.group(1)
|
||||
if field not in fields:
|
||||
fields.append(field)
|
||||
return fields
|
||||
|
||||
|
||||
def extract_sm_tunings(content):
|
||||
"""Extract SM-specific tuning values from static constexpr definitions."""
|
||||
tunings = {}
|
||||
# Match patterns like: sm80_tuning, sm90_tuning, sm100_tuning
|
||||
sm_pattern = re.compile(r'struct\s+sm(\d+)_tuning')
|
||||
for m in sm_pattern.finditer(content):
|
||||
sm = int(m.group(1))
|
||||
if sm not in tunings:
|
||||
tunings[sm] = []
|
||||
|
||||
# Extract actual parameter values from constexpr definitions
|
||||
# Pattern: static constexpr int threads = 512;
|
||||
blocks = re.split(r'(?=struct\s+sm\d+_tuning)', content)
|
||||
for block in blocks:
|
||||
sm_m = re.match(r'struct\s+sm(\d+)_tuning', block)
|
||||
if not sm_m:
|
||||
continue
|
||||
sm = int(sm_m.group(1))
|
||||
vals = {}
|
||||
for line in block.split('\n'):
|
||||
# int values
|
||||
m = re.search(r'static\s+constexpr\s+int\s+(\w+)\s*=\s*(\d+)', line)
|
||||
if m:
|
||||
vals[m.group(1)] = int(m.group(2))
|
||||
# enum values
|
||||
m = re.search(r'static\s+constexpr\s+(?:BlockLoadAlgorithm|BlockStoreAlgorithm|CacheLoadModifier)\s+(\w+)\s*=\s*(\w+)', line)
|
||||
if m:
|
||||
vals[m.group(1)] = m.group(2)
|
||||
if vals:
|
||||
tunings.setdefault(sm, []).append(vals)
|
||||
|
||||
return tunings
|
||||
|
||||
|
||||
def extract_inline_tunings(content):
|
||||
"""Extract inline tuning values from make_mem_scaled_lookback_scan_policy calls and similar."""
|
||||
inline = []
|
||||
# Pattern: threads_per_block, items_per_thread in constructor-style calls
|
||||
pattern = re.compile(
|
||||
r'(?:topk_policy|ReducePassPolicy|ScanLookbackPolicy)\s*\{'
|
||||
r'\s*(\d+)\s*,\s*(\d+)',
|
||||
re.MULTILINE
|
||||
)
|
||||
for m in pattern.finditer(content):
|
||||
inline.append({
|
||||
"threads_per_block": int(m.group(1)),
|
||||
"items_per_thread": int(m.group(2)),
|
||||
})
|
||||
return inline
|
||||
|
||||
|
||||
def algo_name_from_filename(filename):
|
||||
"""tuning_topk.cuh → topk"""
|
||||
base = os.path.basename(filename)
|
||||
return base.replace("tuning_", "").replace(".cuh", "")
|
||||
|
||||
|
||||
def build_schema(algo, fields, sm_tunings, inline_tunings):
|
||||
"""Build YAML-serializable schema dict for one algorithm."""
|
||||
schema = OrderedDict()
|
||||
schema["algorithm"] = algo
|
||||
schema["source"] = f"cub/cub/device/dispatch/tuning/tuning_{algo}.cuh"
|
||||
|
||||
# Parameter space
|
||||
params = OrderedDict()
|
||||
for field in fields:
|
||||
if field in FIELD_TYPES:
|
||||
params[field] = dict(FIELD_TYPES[field])
|
||||
elif field == "lookback_delay":
|
||||
# Expand to sub-fields
|
||||
for sub in ["lookback_delay.kind", "lookback_delay.delay", "lookback_delay.l2_write_latency"]:
|
||||
params[sub] = dict(FIELD_TYPES[sub])
|
||||
else:
|
||||
params[field] = {"type": "int", "range": [1, 1024], "step": 1, "note": "unknown_range"}
|
||||
schema["parameters"] = dict(params)
|
||||
|
||||
# Known SM tunings (for reference when tuning BI-V100)
|
||||
if sm_tunings:
|
||||
ref = OrderedDict()
|
||||
for sm, vals_list in sorted(sm_tunings.items()):
|
||||
ref[f"sm{sm}"] = vals_list
|
||||
schema["reference_tunings"] = dict(ref)
|
||||
|
||||
# BI-V100 placeholder
|
||||
schema["bi_v100"] = {
|
||||
"status": "pending_benchmark",
|
||||
"note": "Run muh benchmark on Iluvatar BI-V100 to fill these values",
|
||||
"threads_per_block": "TBD",
|
||||
"items_per_thread": "TBD",
|
||||
}
|
||||
|
||||
return schema
|
||||
|
||||
|
||||
def yaml_dump(data, indent=0):
|
||||
"""Simple YAML serializer (no dependency on pyyaml)."""
|
||||
lines = []
|
||||
prefix = " " * indent
|
||||
if isinstance(data, dict):
|
||||
for k, v in data.items():
|
||||
if isinstance(v, (dict, list)):
|
||||
lines.append(f"{prefix}{k}:")
|
||||
lines.append(yaml_dump(v, indent + 1))
|
||||
else:
|
||||
lines.append(f"{prefix}{k}: {v}")
|
||||
elif isinstance(data, list):
|
||||
for item in data:
|
||||
if isinstance(item, dict):
|
||||
lines.append(f"{prefix}-")
|
||||
lines.append(yaml_dump(item, indent + 1))
|
||||
else:
|
||||
lines.append(f"{prefix}- {item}")
|
||||
else:
|
||||
lines.append(f"{prefix}{data}")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Extract CCCL tuning params to muh schema")
|
||||
parser.add_argument("--cccl-root", default="cccl_upstream",
|
||||
help="Path to CCCL root (default: cccl_upstream)")
|
||||
parser.add_argument("--out-dir", default="muh/schema",
|
||||
help="Output directory for YAML schemas (default: muh/schema)")
|
||||
args = parser.parse_args()
|
||||
|
||||
tuning_dir = os.path.join(args.cccl_root, "cub", "cub", "device", "dispatch", "tuning")
|
||||
out_dir = args.out_dir
|
||||
os.makedirs(out_dir, exist_ok=True)
|
||||
|
||||
tuning_files = sorted(glob.glob(os.path.join(tuning_dir, "tuning_*.cuh")))
|
||||
if not tuning_files:
|
||||
print(f"ERROR: No tuning_*.cuh files found in {tuning_dir}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Found {len(tuning_files)} tuning files in {tuning_dir}")
|
||||
|
||||
all_algos = []
|
||||
for filepath in tuning_files:
|
||||
if os.path.basename(filepath) == "common.cuh":
|
||||
continue
|
||||
|
||||
with open(filepath, "r") as f:
|
||||
content = f.read()
|
||||
|
||||
algo = algo_name_from_filename(filepath)
|
||||
fields = extract_policy_fields(content, filepath)
|
||||
sm_tunings = extract_sm_tunings(content)
|
||||
inline_tunings = extract_inline_tunings(content)
|
||||
|
||||
schema = build_schema(algo, fields, sm_tunings, inline_tunings)
|
||||
|
||||
out_path = os.path.join(out_dir, f"{algo}.yaml")
|
||||
with open(out_path, "w") as f:
|
||||
f.write(f"# muh schema for {algo}\n")
|
||||
f.write(f"# Auto-extracted from {schema['source']}\n")
|
||||
f.write(f"# Generated by muh/extract.py\n\n")
|
||||
f.write(yaml_dump(dict(schema)))
|
||||
f.write("\n")
|
||||
|
||||
all_algos.append(algo)
|
||||
print(f" {algo}: {len(fields)} params, {len(sm_tunings)} SM tunings → {out_path}")
|
||||
|
||||
# Write index
|
||||
index_path = os.path.join(out_dir, "_index.yaml")
|
||||
with open(index_path, "w") as f:
|
||||
f.write("# muh schema index — all extracted CCCL tuning algorithms\n\n")
|
||||
f.write("algorithms:\n")
|
||||
for algo in all_algos:
|
||||
f.write(f" - {algo}\n")
|
||||
f.write(f"\ntotal: {len(all_algos)}\n")
|
||||
f.write(f"source: cccl_upstream/cub/cub/device/dispatch/tuning/\n")
|
||||
|
||||
print(f"\nDone: {len(all_algos)} schemas → {out_dir}/")
|
||||
print(f"Index: {index_path}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
244
muh/gen_patch.py
Normal file
244
muh/gen_patch.py
Normal file
@@ -0,0 +1,244 @@
|
||||
#!/usr/bin/env python3
|
||||
"""muh/gen_patch.py — Generate vllm kernel patches from .muh tuning configuration
|
||||
|
||||
Given a .muh file with tuning overrides for BI-V100, generates unified diff
|
||||
patches that can be applied to the vllm source tree to inject optimized
|
||||
kernel parameters.
|
||||
|
||||
The key insight: vllm's CUDA kernels (attention, sampling, layernorm) have
|
||||
hardcoded launch configs. This script generates patches that replace those
|
||||
hardcodes with values tuned for Iluvatar BI-V100 via CCCL benchmark data.
|
||||
|
||||
Usage:
|
||||
python3 muh/gen_patch.py baseline.muh [-o patches/] [--vllm-root /path/to/vllm]
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import argparse
|
||||
from datetime import datetime
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from parse import load_muh
|
||||
|
||||
|
||||
# --- Kernel location mapping ---
|
||||
# Maps CCCL algorithm names to vllm source files and the specific
|
||||
# constants/defines that control kernel launch parameters.
|
||||
|
||||
VLLM_KERNEL_MAP = {
|
||||
"reduce": {
|
||||
"description": "Attention score reduction in multi-head attention",
|
||||
"files": [
|
||||
"csrc/attention/attention_kernels.cu",
|
||||
"csrc/attention/paged_attention_v2.cu",
|
||||
],
|
||||
"params": {
|
||||
"threads_per_block": {
|
||||
"pattern": "NUM_THREADS",
|
||||
"default": 128,
|
||||
"locations": ["#define NUM_THREADS 128"],
|
||||
},
|
||||
"items_per_thread": {
|
||||
"pattern": "NUM_ITEMS_PER_THREAD",
|
||||
"default": 8,
|
||||
},
|
||||
"vec_size": {
|
||||
"pattern": "VEC_SIZE",
|
||||
"default": 4,
|
||||
},
|
||||
},
|
||||
},
|
||||
"topk": {
|
||||
"description": "Top-k / top-p sampling in decode stage",
|
||||
"files": [
|
||||
"csrc/sampling/sampling_kernels.cu",
|
||||
],
|
||||
"params": {
|
||||
"threads_per_block": {
|
||||
"pattern": "SAMPLING_BLOCK_SIZE",
|
||||
"default": 256,
|
||||
},
|
||||
"bits_per_pass": {
|
||||
"pattern": "RADIX_BITS",
|
||||
"default": 8,
|
||||
},
|
||||
},
|
||||
},
|
||||
"scan": {
|
||||
"description": "Prefix scan in paged attention block table lookup",
|
||||
"files": [
|
||||
"csrc/attention/paged_attention_v1.cu",
|
||||
],
|
||||
"params": {
|
||||
"threads_per_block": {
|
||||
"pattern": "SCAN_BLOCK_SIZE",
|
||||
"default": 128,
|
||||
},
|
||||
},
|
||||
},
|
||||
"transform": {
|
||||
"description": "Elementwise activation kernels (SiLU, GELU, RMSNorm)",
|
||||
"files": [
|
||||
"csrc/activation_kernels.cu",
|
||||
"csrc/layernorm_kernels.cu",
|
||||
],
|
||||
"params": {
|
||||
"threads_per_block": {
|
||||
"pattern": "ACTIVATION_BLOCK_SIZE",
|
||||
"default": 512,
|
||||
},
|
||||
},
|
||||
},
|
||||
"batch_memcpy": {
|
||||
"description": "KV cache block copy between GPU memory regions",
|
||||
"files": [
|
||||
"csrc/cache_kernels.cu",
|
||||
],
|
||||
"params": {
|
||||
"threads_per_block": {
|
||||
"pattern": "COPY_BLOCK_SIZE",
|
||||
"default": 256,
|
||||
},
|
||||
},
|
||||
},
|
||||
"for": {
|
||||
"description": "Elementwise for-each kernels (position embeddings, rope)",
|
||||
"files": [
|
||||
"csrc/pos_encoding_kernels.cu",
|
||||
],
|
||||
"params": {
|
||||
"threads_per_block": {
|
||||
"pattern": "ROPE_BLOCK_SIZE",
|
||||
"default": 512,
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def generate_define_patch(algo, param_name, old_value, new_value, define_name, filepath):
|
||||
"""Generate a unified diff snippet for a #define change."""
|
||||
lines = []
|
||||
lines.append(f"--- a/{filepath}")
|
||||
lines.append(f"+++ b/{filepath}")
|
||||
lines.append(f"@@ -1,1 +1,1 @@")
|
||||
lines.append(f"-#define {define_name} {old_value}")
|
||||
lines.append(f"+#define {define_name} {new_value} // muh: tuned for BI-V100 ({algo}.{param_name})")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def generate_patches(config, vllm_root=None):
|
||||
"""Generate all patches from tuning config."""
|
||||
tuning = config.get("tuning", {})
|
||||
patches = []
|
||||
summary = []
|
||||
|
||||
for algo, algo_params in tuning.items():
|
||||
if not isinstance(algo_params, dict):
|
||||
continue
|
||||
|
||||
mapping = VLLM_KERNEL_MAP.get(algo)
|
||||
if mapping is None:
|
||||
summary.append(f"SKIP {algo}: no vllm kernel mapping defined")
|
||||
continue
|
||||
|
||||
for param_name, new_value in algo_params.items():
|
||||
if param_name.startswith("_"):
|
||||
continue
|
||||
if new_value is None:
|
||||
continue
|
||||
|
||||
param_spec = mapping.get("params", {}).get(param_name)
|
||||
if param_spec is None:
|
||||
continue
|
||||
|
||||
old_value = param_spec.get("default")
|
||||
define_name = param_spec.get("pattern", param_name.upper())
|
||||
|
||||
for filepath in mapping.get("files", []):
|
||||
patch = generate_define_patch(
|
||||
algo, param_name, old_value, new_value, define_name, filepath
|
||||
)
|
||||
patches.append({
|
||||
"algo": algo,
|
||||
"param": param_name,
|
||||
"file": filepath,
|
||||
"old": old_value,
|
||||
"new": new_value,
|
||||
"diff": patch,
|
||||
})
|
||||
summary.append(
|
||||
f"PATCH {filepath}: {define_name} {old_value} → {new_value} "
|
||||
f"(from {algo}.{param_name})"
|
||||
)
|
||||
|
||||
return patches, summary
|
||||
|
||||
|
||||
def write_patches(patches, out_dir):
|
||||
"""Write patches to individual .patch files."""
|
||||
os.makedirs(out_dir, exist_ok=True)
|
||||
|
||||
# Combined patch
|
||||
combined_path = os.path.join(out_dir, "muh_bi100_tuning.patch")
|
||||
with open(combined_path, 'w') as f:
|
||||
f.write(f"# muh kernel tuning patch for Iluvatar BI-V100\n")
|
||||
f.write(f"# Generated: {datetime.now().isoformat()}\n")
|
||||
f.write(f"# Algorithms patched: {len(set(p['algo'] for p in patches))}\n")
|
||||
f.write(f"# Total changes: {len(patches)}\n\n")
|
||||
for p in patches:
|
||||
f.write(p["diff"])
|
||||
f.write("\n\n")
|
||||
|
||||
# Per-algorithm patches
|
||||
by_algo = {}
|
||||
for p in patches:
|
||||
by_algo.setdefault(p["algo"], []).append(p)
|
||||
|
||||
for algo, algo_patches in by_algo.items():
|
||||
algo_path = os.path.join(out_dir, f"{algo}.patch")
|
||||
with open(algo_path, 'w') as f:
|
||||
f.write(f"# muh tuning patch: {algo} for BI-V100\n\n")
|
||||
for p in algo_patches:
|
||||
f.write(p["diff"])
|
||||
f.write("\n\n")
|
||||
|
||||
return combined_path
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Generate vllm kernel patches from .muh")
|
||||
parser.add_argument("muh_file", help="Path to .muh file")
|
||||
parser.add_argument("-o", "--output-dir", default="patches",
|
||||
help="Output directory for patches (default: patches)")
|
||||
parser.add_argument("--vllm-root", default=None,
|
||||
help="Path to vllm source tree (for verification)")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="Print patches to stdout instead of writing files")
|
||||
args = parser.parse_args()
|
||||
|
||||
config = load_muh(args.muh_file)
|
||||
patches, summary = generate_patches(config, args.vllm_root)
|
||||
|
||||
print(f"muh gen_patch: {len(patches)} patches from {args.muh_file}\n")
|
||||
for s in summary:
|
||||
print(f" {s}")
|
||||
|
||||
if not patches:
|
||||
print("\nNo patches generated. Add tuning overrides to your .muh file.")
|
||||
return
|
||||
|
||||
if args.dry_run:
|
||||
print("\n--- Patches ---\n")
|
||||
for p in patches:
|
||||
print(p["diff"])
|
||||
print()
|
||||
else:
|
||||
combined = write_patches(patches, args.output_dir)
|
||||
print(f"\nWritten to {args.output_dir}/")
|
||||
print(f"Combined: {combined}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
130
muh/gen_yaml.py
Normal file
130
muh/gen_yaml.py
Normal file
@@ -0,0 +1,130 @@
|
||||
#!/usr/bin/env python3
|
||||
"""muh/gen_yaml.py — Generate computility-run.yaml from a .muh configuration
|
||||
|
||||
Reads a .muh file (via parse.py), extracts the 'vllm' and 'env' sections,
|
||||
and outputs a computility-run.yaml compatible with ModelHub XC platform.
|
||||
|
||||
Usage:
|
||||
python3 muh/gen_yaml.py baseline.muh [-o computility-run.yaml]
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import argparse
|
||||
|
||||
# Import from sibling module
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from parse import load_muh
|
||||
|
||||
|
||||
# Mapping from .muh vllm config keys to CLI arguments
|
||||
VLLM_ARG_MAP = {
|
||||
"model_path": "--model",
|
||||
"served_model_name": "--served-model-name",
|
||||
"max_model_len": "--max-model-len",
|
||||
"gpu_memory_utilization": "--gpu-memory-utilization",
|
||||
"tensor_parallel": "-tp",
|
||||
"max_num_seqs": "--max-num-seqs",
|
||||
"max_num_batched_tokens": "--max-num-batched-tokens",
|
||||
"max_seq_len_to_capture": "--max-seq-len-to-capture",
|
||||
"tool_call_parser": "--tool-call-parser",
|
||||
"reasoning_parser": "--reasoning-parser",
|
||||
}
|
||||
|
||||
# Boolean flags (presence = enabled, no value needed)
|
||||
VLLM_FLAG_MAP = {
|
||||
"trust_remote_code": "--trust-remote-code",
|
||||
"disable_log_requests": "--disable-log-requests",
|
||||
"disable_frontend_multiprocessing": "--disable-frontend-multiprocessing",
|
||||
"enable_chunked_prefill": "--enable-chunked-prefill",
|
||||
"enable_auto_tool_choice": "--enable-auto-tool-choice",
|
||||
"enable_prefix_caching": "--enable-prefix-caching",
|
||||
}
|
||||
|
||||
|
||||
def build_command(vllm_config):
|
||||
"""Build the command list for computility-run.yaml from vllm config."""
|
||||
cmd = [
|
||||
"python3",
|
||||
"-m",
|
||||
"vllm.entrypoints.openai.api_server",
|
||||
]
|
||||
|
||||
# Model path (required)
|
||||
model_path = vllm_config.get("model_path", "/model")
|
||||
cmd.extend(["--model", model_path])
|
||||
|
||||
# Named arguments
|
||||
for muh_key, cli_arg in VLLM_ARG_MAP.items():
|
||||
if muh_key == "model_path":
|
||||
continue # already handled
|
||||
val = vllm_config.get(muh_key)
|
||||
if val is not None:
|
||||
cmd.extend([cli_arg, str(val)])
|
||||
|
||||
# Boolean flags
|
||||
for muh_key, cli_flag in VLLM_FLAG_MAP.items():
|
||||
if vllm_config.get(muh_key, False):
|
||||
cmd.append(cli_flag)
|
||||
|
||||
# Extra raw args (pass-through)
|
||||
extra = vllm_config.get("extra_args", [])
|
||||
if isinstance(extra, list):
|
||||
cmd.extend([str(a) for a in extra])
|
||||
|
||||
return cmd
|
||||
|
||||
|
||||
def build_env(env_config):
|
||||
"""Build environment variable list."""
|
||||
env_list = []
|
||||
for name, value in env_config.items():
|
||||
env_list.append({"name": name, "value": str(value)})
|
||||
return env_list
|
||||
|
||||
|
||||
def yaml_serialize_computility(concurrency, command, env_list):
|
||||
"""Serialize to computility-run.yaml format (no pyyaml dependency)."""
|
||||
lines = []
|
||||
lines.append(f"concurrency: {concurrency}")
|
||||
lines.append("command:")
|
||||
for item in command:
|
||||
lines.append(f" - '{item}'" if ' ' in str(item) else f" - {item}")
|
||||
if env_list:
|
||||
lines.append("env:")
|
||||
for e in env_list:
|
||||
lines.append(f" - name: {e['name']}")
|
||||
lines.append(f" value: {e['value']}")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Generate computility-run.yaml from .muh")
|
||||
parser.add_argument("muh_file", help="Path to .muh file")
|
||||
parser.add_argument("-o", "--output", default="computility-run.yaml",
|
||||
help="Output file path (default: computility-run.yaml)")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="Print to stdout instead of writing file")
|
||||
args = parser.parse_args()
|
||||
|
||||
config = load_muh(args.muh_file)
|
||||
|
||||
vllm_config = config.get("vllm", {})
|
||||
env_config = config.get("env", {})
|
||||
concurrency = config.get("concurrency", 1)
|
||||
|
||||
command = build_command(vllm_config)
|
||||
env_list = build_env(env_config)
|
||||
|
||||
output = yaml_serialize_computility(concurrency, command, env_list)
|
||||
|
||||
if args.dry_run:
|
||||
print(output)
|
||||
else:
|
||||
with open(args.output, 'w') as f:
|
||||
f.write(output)
|
||||
print(f"Generated {args.output} ({len(command)} command args, {len(env_list)} env vars)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
229
muh/parse.py
Normal file
229
muh/parse.py
Normal file
@@ -0,0 +1,229 @@
|
||||
#!/usr/bin/env python3
|
||||
"""muh/parse.py — Parse .muh configuration files
|
||||
|
||||
A .muh file is YAML with these semantics:
|
||||
- 'extends': inherit from another .muh file (deep merge, child overrides parent)
|
||||
- 'hardware': target hardware description (warp_size, smem, registers, etc.)
|
||||
- 'tuning': per-algorithm parameter overrides
|
||||
- 'vllm': vllm-specific launch config (maps to computility-run.yaml)
|
||||
- 'env': environment variable overrides
|
||||
|
||||
Schema validation against muh/schema/*.yaml ensures parameter names and
|
||||
value ranges are legal.
|
||||
|
||||
Usage:
|
||||
python3 muh/parse.py baseline.muh [--schema-dir muh/schema] [--validate]
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import copy
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def yaml_load_simple(text):
|
||||
"""Minimal YAML parser — handles flat dicts, nested dicts, lists, strings, numbers.
|
||||
No dependency on PyYAML. Sufficient for .muh files."""
|
||||
result = {}
|
||||
stack = [(result, -1)] # (dict, indent_level)
|
||||
current_list_key = None
|
||||
|
||||
for line in text.split('\n'):
|
||||
stripped = line.strip()
|
||||
if not stripped or stripped.startswith('#'):
|
||||
continue
|
||||
|
||||
indent = len(line) - len(line.lstrip())
|
||||
|
||||
# Pop stack to find parent at correct indent
|
||||
while len(stack) > 1 and stack[-1][1] >= indent:
|
||||
stack.pop()
|
||||
|
||||
parent, _ = stack[-1]
|
||||
|
||||
# List item
|
||||
if stripped.startswith('- '):
|
||||
val = stripped[2:].strip()
|
||||
if current_list_key and current_list_key in parent:
|
||||
if not isinstance(parent[current_list_key], list):
|
||||
parent[current_list_key] = []
|
||||
parent[current_list_key].append(_parse_value(val))
|
||||
continue
|
||||
|
||||
if ':' in stripped:
|
||||
key, _, val = stripped.partition(':')
|
||||
key = key.strip()
|
||||
val = val.strip()
|
||||
|
||||
if val == '' or val == '|':
|
||||
# Nested dict or upcoming list
|
||||
parent[key] = {}
|
||||
stack.append((parent[key], indent))
|
||||
current_list_key = key
|
||||
elif val.startswith('[') and val.endswith(']'):
|
||||
# Inline list
|
||||
items = [_parse_value(v.strip()) for v in val[1:-1].split(',') if v.strip()]
|
||||
parent[key] = items
|
||||
current_list_key = None
|
||||
else:
|
||||
parent[key] = _parse_value(val)
|
||||
current_list_key = key if val == '' else None
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def _parse_value(v):
|
||||
"""Parse a YAML scalar value."""
|
||||
if v in ('true', 'True', 'yes'):
|
||||
return True
|
||||
if v in ('false', 'False', 'no'):
|
||||
return False
|
||||
if v in ('null', 'None', '~', 'TBD'):
|
||||
return None
|
||||
# Strip quotes
|
||||
if (v.startswith('"') and v.endswith('"')) or (v.startswith("'") and v.endswith("'")):
|
||||
return v[1:-1]
|
||||
# Try int
|
||||
try:
|
||||
return int(v)
|
||||
except ValueError:
|
||||
pass
|
||||
# Try float
|
||||
try:
|
||||
return float(v)
|
||||
except ValueError:
|
||||
pass
|
||||
return v
|
||||
|
||||
|
||||
def deep_merge(base, override):
|
||||
"""Deep merge two dicts; override wins on conflicts."""
|
||||
result = copy.deepcopy(base)
|
||||
for k, v in override.items():
|
||||
if k in result and isinstance(result[k], dict) and isinstance(v, dict):
|
||||
result[k] = deep_merge(result[k], v)
|
||||
else:
|
||||
result[k] = copy.deepcopy(v)
|
||||
return result
|
||||
|
||||
|
||||
def load_muh(filepath, search_dirs=None):
|
||||
"""Load a .muh file, resolving 'extends' chain."""
|
||||
if search_dirs is None:
|
||||
search_dirs = [os.path.dirname(filepath), '.']
|
||||
|
||||
with open(filepath, 'r') as f:
|
||||
data = yaml_load_simple(f.read())
|
||||
|
||||
# Resolve extends
|
||||
if 'extends' in data:
|
||||
parent_name = data.pop('extends')
|
||||
parent_path = None
|
||||
for d in search_dirs:
|
||||
candidate = os.path.join(d, parent_name)
|
||||
if os.path.exists(candidate):
|
||||
parent_path = candidate
|
||||
break
|
||||
if parent_path is None:
|
||||
raise FileNotFoundError(f"Cannot find parent .muh file: {parent_name} (searched {search_dirs})")
|
||||
parent_data = load_muh(parent_path, search_dirs)
|
||||
data = deep_merge(parent_data, data)
|
||||
|
||||
return data
|
||||
|
||||
|
||||
def load_schema(schema_dir, algo):
|
||||
"""Load a schema YAML for validation."""
|
||||
path = os.path.join(schema_dir, f"{algo}.yaml")
|
||||
if not os.path.exists(path):
|
||||
return None
|
||||
with open(path, 'r') as f:
|
||||
return yaml_load_simple(f.read())
|
||||
|
||||
|
||||
def validate_tuning(config, schema_dir):
|
||||
"""Validate tuning parameters against extracted schemas."""
|
||||
errors = []
|
||||
tuning = config.get('tuning', {})
|
||||
|
||||
for algo, params in tuning.items():
|
||||
if not isinstance(params, dict):
|
||||
continue
|
||||
schema = load_schema(schema_dir, algo)
|
||||
if schema is None:
|
||||
errors.append(f"WARNING: No schema for algorithm '{algo}' — skipping validation")
|
||||
continue
|
||||
|
||||
schema_params = schema.get('parameters', {})
|
||||
for param_name, param_value in params.items():
|
||||
if param_name.startswith('_'): # metadata keys
|
||||
continue
|
||||
if param_name not in schema_params:
|
||||
errors.append(f"{algo}.{param_name}: unknown parameter (not in schema)")
|
||||
continue
|
||||
|
||||
spec = schema_params[param_name]
|
||||
if spec.get('type') == 'int' and isinstance(param_value, (int, float)):
|
||||
rng = spec.get('range', [0, 99999])
|
||||
if not (rng[0] <= param_value <= rng[1]):
|
||||
errors.append(
|
||||
f"{algo}.{param_name}: value {param_value} outside range {rng}"
|
||||
)
|
||||
elif spec.get('type') == 'enum' and isinstance(param_value, str):
|
||||
valid = spec.get('values', [])
|
||||
if param_value not in valid:
|
||||
errors.append(
|
||||
f"{algo}.{param_name}: '{param_value}' not in {valid}"
|
||||
)
|
||||
|
||||
return errors
|
||||
|
||||
|
||||
def print_config(config, indent=0):
|
||||
"""Pretty-print a parsed .muh config."""
|
||||
prefix = " " * indent
|
||||
for k, v in config.items():
|
||||
if isinstance(v, dict):
|
||||
print(f"{prefix}{k}:")
|
||||
print_config(v, indent + 1)
|
||||
elif isinstance(v, list):
|
||||
print(f"{prefix}{k}:")
|
||||
for item in v:
|
||||
print(f"{prefix} - {item}")
|
||||
else:
|
||||
print(f"{prefix}{k}: {v}")
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Parse and validate .muh files")
|
||||
parser.add_argument("muh_file", help="Path to .muh file")
|
||||
parser.add_argument("--schema-dir", default="muh/schema",
|
||||
help="Schema directory (default: muh/schema)")
|
||||
parser.add_argument("--validate", action="store_true",
|
||||
help="Validate tuning params against schemas")
|
||||
parser.add_argument("--json", action="store_true",
|
||||
help="Output as JSON instead of pretty-print")
|
||||
args = parser.parse_args()
|
||||
|
||||
config = load_muh(args.muh_file)
|
||||
|
||||
if args.validate:
|
||||
errors = validate_tuning(config, args.schema_dir)
|
||||
if errors:
|
||||
print("Validation errors:", file=sys.stderr)
|
||||
for e in errors:
|
||||
print(f" ✗ {e}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
else:
|
||||
print("Validation passed ✓", file=sys.stderr)
|
||||
|
||||
if args.json:
|
||||
import json
|
||||
print(json.dumps(config, indent=2, ensure_ascii=False))
|
||||
else:
|
||||
print_config(config)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
32
muh/schema/_index.yaml
Normal file
32
muh/schema/_index.yaml
Normal file
@@ -0,0 +1,32 @@
|
||||
# muh schema index — all extracted CCCL tuning algorithms
|
||||
|
||||
algorithms:
|
||||
- adjacent_difference
|
||||
- batch_memcpy
|
||||
- batched_topk
|
||||
- find
|
||||
- find_bound_sorted_values
|
||||
- for
|
||||
- histogram
|
||||
- merge
|
||||
- merge_sort
|
||||
- radix_sort
|
||||
- reduce
|
||||
- reduce_by_key
|
||||
- rle_encode
|
||||
- rle_non_trivial_runs
|
||||
- scan
|
||||
- scan_by_key
|
||||
- segmented_radix_sort
|
||||
- segmented_reduce
|
||||
- segmented_scan
|
||||
- segmented_sort
|
||||
- select_if
|
||||
- three_way_partition
|
||||
- topk
|
||||
- transform
|
||||
- transform_tile
|
||||
- unique_by_key
|
||||
|
||||
total: 26
|
||||
source: cccl_upstream/cub/cub/device/dispatch/tuning/
|
||||
56
muh/schema/adjacent_difference.yaml
Normal file
56
muh/schema/adjacent_difference.yaml
Normal file
@@ -0,0 +1,56 @@
|
||||
# muh schema for adjacent_difference
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_adjacent_difference.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: adjacent_difference
|
||||
source: cub/cub/device/dispatch/tuning/tuning_adjacent_difference.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
value_type_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
67
muh/schema/batch_memcpy.yaml
Normal file
67
muh/schema/batch_memcpy.yaml
Normal file
@@ -0,0 +1,67 @@
|
||||
# muh schema for batch_memcpy
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_batch_memcpy.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: batch_memcpy
|
||||
source: cub/cub/device/dispatch/tuning/tuning_batch_memcpy.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
buffers_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bytes_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
block_level_tile_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
warp_level_threshold:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
block_level_threshold:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
buffer_lookback_delay:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
block_lookback_delay:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
46
muh/schema/batched_topk.yaml
Normal file
46
muh/schema/batched_topk.yaml
Normal file
@@ -0,0 +1,46 @@
|
||||
# muh schema for batched_topk
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_batched_topk.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: batched_topk
|
||||
source: cub/cub/device/dispatch/tuning/tuning_batched_topk.cuh
|
||||
parameters:
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
46
muh/schema/find.yaml
Normal file
46
muh/schema/find.yaml
Normal file
@@ -0,0 +1,46 @@
|
||||
# muh schema for find
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_find.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: find
|
||||
source: cub/cub/device/dispatch/tuning/tuning_find.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
input_type_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
47
muh/schema/find_bound_sorted_values.yaml
Normal file
47
muh/schema/find_bound_sorted_values.yaml
Normal file
@@ -0,0 +1,47 @@
|
||||
# muh schema for find_bound_sorted_values
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_find_bound_sorted_values.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: find_bound_sorted_values
|
||||
source: cub/cub/device/dispatch/tuning/tuning_find_bound_sorted_values.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
range_type_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
values_type_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
24
muh/schema/for.yaml
Normal file
24
muh/schema/for.yaml
Normal file
@@ -0,0 +1,24 @@
|
||||
# muh schema for for
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_for.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: for
|
||||
source: cub/cub/device/dispatch/tuning/tuning_for.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
117
muh/schema/histogram.yaml
Normal file
117
muh/schema/histogram.yaml
Normal file
@@ -0,0 +1,117 @@
|
||||
# muh schema for histogram
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_histogram.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: histogram
|
||||
source: cub/cub/device/dispatch/tuning/tuning_histogram.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
pixels_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
init_kernel_pdl_trigger_max_bins:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
sample_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
counter_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
sample_size_bytes:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
num_channels:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
num_active_channels:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm90:
|
||||
-
|
||||
threads: 768
|
||||
items: 12
|
||||
load_modifier: LOAD_LDG
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 960
|
||||
items: 10
|
||||
load_modifier: LOAD_DEFAULT
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
sm100:
|
||||
-
|
||||
items: 12
|
||||
threads: 928
|
||||
load_modifier: LOAD_CA
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
vec_size: 1
|
||||
-
|
||||
items: 12
|
||||
threads: 448
|
||||
load_modifier: LOAD_LDG
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
vec_size: 1
|
||||
init_kernel_pdl_trigger_max_bins: 2048
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
89
muh/schema/merge.yaml
Normal file
89
muh/schema/merge.yaml
Normal file
@@ -0,0 +1,89 @@
|
||||
# muh schema for merge
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_merge.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: merge
|
||||
source: cub/cub/device/dispatch/tuning/tuning_merge.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_align:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_is_trivially_relocatable:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_align:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_is_trivially_relocatable:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
63
muh/schema/merge_sort.yaml
Normal file
63
muh/schema/merge_sort.yaml
Normal file
@@ -0,0 +1,63 @@
|
||||
# muh schema for merge_sort
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_merge_sort.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: merge_sort
|
||||
source: cub/cub/device/dispatch/tuning/tuning_merge_sort.cuh
|
||||
parameters:
|
||||
ItemsPerThread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
103
muh/schema/radix_sort.yaml
Normal file
103
muh/schema/radix_sort.yaml
Normal file
@@ -0,0 +1,103 @@
|
||||
# muh schema for radix_sort
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_radix_sort.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: radix_sort
|
||||
source: cub/cub/device/dispatch/tuning/tuning_radix_sort.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
private_partitions:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
radix_bits:
|
||||
type: int
|
||||
range:
|
||||
- 4
|
||||
- 8
|
||||
step: 1
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
rank_private_partitions:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
threads:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
items:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
102
muh/schema/reduce.yaml
Normal file
102
muh/schema/reduce.yaml
Normal file
@@ -0,0 +1,102 @@
|
||||
# muh schema for reduce
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_reduce.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: reduce
|
||||
source: cub/cub/device/dispatch/tuning/tuning_reduce.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
reduce_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_REDUCE_RAKING
|
||||
- BLOCK_REDUCE_RAKING_COMMUTATIVE_ONLY
|
||||
- BLOCK_REDUCE_WARP_REDUCTIONS
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
items:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
threads:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
items_per_vec_load:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm100:
|
||||
-
|
||||
items: 15
|
||||
threads: 512
|
||||
items_per_vec_load: 2
|
||||
-
|
||||
items: 15
|
||||
threads: 512
|
||||
items_per_vec_load: 1
|
||||
-
|
||||
items: 16
|
||||
threads: 512
|
||||
items_per_vec_load: 2
|
||||
-
|
||||
items: 16
|
||||
threads: 640
|
||||
items_per_vec_load: 1
|
||||
-
|
||||
threads_per_block: 256
|
||||
items_per_thread: 16
|
||||
items_per_vec_load: 4
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
390
muh/schema/reduce_by_key.yaml
Normal file
390
muh/schema/reduce_by_key.yaml
Normal file
@@ -0,0 +1,390 @@
|
||||
# muh schema for reduce_by_key
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_reduce_by_key.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: reduce_by_key
|
||||
source: cub/cub/device/dispatch/tuning/tuning_reduce_by_key.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 160
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 160
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 320
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
nominal_4B_items_per_thread: 6
|
||||
sm100:
|
||||
-
|
||||
items: 13
|
||||
threads: 576
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 10
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 128
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 19
|
||||
threads: 128
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 14
|
||||
threads: 128
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 11
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 10
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 10
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 11
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 10
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 9
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 11
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 9
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 9
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
155
muh/schema/rle_encode.yaml
Normal file
155
muh/schema/rle_encode.yaml
Normal file
@@ -0,0 +1,155 @@
|
||||
# muh schema for rle_encode
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: rle_encode
|
||||
source: cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
length_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
nominal_4B_items_per_thread: 6
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
155
muh/schema/rle_non_trivial_runs.yaml
Normal file
155
muh/schema/rle_non_trivial_runs.yaml
Normal file
@@ -0,0 +1,155 @@
|
||||
# muh schema for rle_non_trivial_runs
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_rle_non_trivial_runs.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: rle_non_trivial_runs
|
||||
source: cub/cub/device/dispatch/tuning/tuning_rle_non_trivial_runs.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
length_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 192
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
threads: 224
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 224
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 224
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 288
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
nominal_4B_items_per_thread: 15
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
255
muh/schema/scan.yaml
Normal file
255
muh/schema/scan.yaml
Normal file
@@ -0,0 +1,255 @@
|
||||
# muh schema for scan
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_scan.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: scan
|
||||
source: cub/cub/device/dispatch/tuning/tuning_scan.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
reduce_and_scan_warps:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
lookahead_items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 16
|
||||
step: 1
|
||||
lookahead_stages:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
block_idx_stages:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
delay_constructor:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
input_value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
input_value_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
output_value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
output_value_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm75:
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
sm80:
|
||||
-
|
||||
threads: 320
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 352
|
||||
items: 16
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 320
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 8
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 384
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 640
|
||||
items: 24
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
sm90:
|
||||
|
||||
sm100:
|
||||
-
|
||||
items: 18
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 13
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 13
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 22
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 19
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 23
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 22
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
450
muh/schema/scan_by_key.yaml
Normal file
450
muh/schema/scan_by_key.yaml
Normal file
@@ -0,0 +1,450 @@
|
||||
# muh schema for scan_by_key
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_scan_by_key.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: scan_by_key
|
||||
source: cub/cub/device/dispatch/tuning/tuning_scan_by_key.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 288
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 8
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 320
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 160
|
||||
items: 17
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 160
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 17
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 320
|
||||
items: 8
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 16
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
items: 13
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 13
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 19
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 18
|
||||
threads: 192
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 12
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 13
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 20
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 13
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 20
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 14
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 12
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 15
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 22
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 23
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
nominal_4b_items_per_thread: 9
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
26
muh/schema/segmented_radix_sort.yaml
Normal file
26
muh/schema/segmented_radix_sort.yaml
Normal file
@@ -0,0 +1,26 @@
|
||||
# muh schema for segmented_radix_sort
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_radix_sort.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: segmented_radix_sort
|
||||
source: cub/cub/device/dispatch/tuning/tuning_segmented_radix_sort.cuh
|
||||
parameters:
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
60
muh/schema/segmented_reduce.yaml
Normal file
60
muh/schema/segmented_reduce.yaml
Normal file
@@ -0,0 +1,60 @@
|
||||
# muh schema for segmented_reduce
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_reduce.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: segmented_reduce
|
||||
source: cub/cub/device/dispatch/tuning/tuning_segmented_reduce.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
threads_per_warp:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
76
muh/schema/segmented_scan.yaml
Normal file
76
muh/schema/segmented_scan.yaml
Normal file
@@ -0,0 +1,76 @@
|
||||
# muh schema for segmented_scan
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_scan.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: segmented_scan
|
||||
source: cub/cub/device/dispatch/tuning/tuning_segmented_scan.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
max_segments:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_align:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
82
muh/schema/segmented_sort.yaml
Normal file
82
muh/schema/segmented_sort.yaml
Normal file
@@ -0,0 +1,82 @@
|
||||
# muh schema for segmented_sort
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_sort.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: segmented_sort
|
||||
source: cub/cub/device/dispatch/tuning/tuning_segmented_sort.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
radix_bits:
|
||||
type: int
|
||||
range:
|
||||
- 4
|
||||
- 8
|
||||
step: 1
|
||||
threads_per_warp:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
partitioning_threshold:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
508
muh/schema/select_if.yaml
Normal file
508
muh/schema/select_if.yaml
Normal file
@@ -0,0 +1,508 @@
|
||||
# muh schema for select_if
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_select_if.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: select_if
|
||||
source: cub/cub/device/dispatch/tuning/tuning_select_if.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
input_size_bytes:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
flag_size_bytes:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size_bytes:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 992
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 576
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 384
|
||||
items: 4
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 320
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 6
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 512
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 384
|
||||
items: 17
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 384
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 512
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 448
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 448
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
items: 3
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 320
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 6
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 160
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
sm100:
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 896
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 1024
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
nominal_4b_items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 320
|
||||
nominal_4b_items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 21
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 3
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 3
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
nominal_4b_items: 15
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 576
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 18
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 14
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 23
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 23
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 448
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 18
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 21
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 448
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
nominal_4b_items: 14
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 14
|
||||
threads: 640
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 24
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 18
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 11
|
||||
threads: 448
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 12
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 12
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 14
|
||||
threads: 352
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 11
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
nominal_4B_items_per_thread: 10
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
185
muh/schema/three_way_partition.yaml
Normal file
185
muh/schema/three_way_partition.yaml
Normal file
@@ -0,0 +1,185 @@
|
||||
# muh schema for three_way_partition
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_three_way_partition.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: three_way_partition
|
||||
source: cub/cub/device/dispatch/tuning/tuning_three_way_partition.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
input_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 320
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 24
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 640
|
||||
items: 24
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
items: 12
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
items: 14
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 11
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
items: 10
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 20
|
||||
threads: 768
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 20
|
||||
threads: 768
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 15
|
||||
threads: 768
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 14
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
52
muh/schema/topk.yaml
Normal file
52
muh/schema/topk.yaml
Normal file
@@ -0,0 +1,52 @@
|
||||
# muh schema for topk
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_topk.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: topk
|
||||
source: cub/cub/device/dispatch/tuning/tuning_topk.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
bits_per_pass:
|
||||
type: int
|
||||
range:
|
||||
- 4
|
||||
- 11
|
||||
step: 1
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
107
muh/schema/transform.yaml
Normal file
107
muh/schema/transform.yaml
Normal file
@@ -0,0 +1,107 @@
|
||||
# muh schema for transform
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_transform.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: transform
|
||||
source: cub/cub/device/dispatch/tuning/tuning_transform.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread_no_input:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
min_items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
max_items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
prefetch_byte_stride:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
unroll_factor:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
store_vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
min_bytes_in_flight:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
copy_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
smem_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
tile_padding:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
max_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
19
muh/schema/transform_tile.yaml
Normal file
19
muh/schema/transform_tile.yaml
Normal file
@@ -0,0 +1,19 @@
|
||||
# muh schema for transform_tile
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_transform_tile.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: transform_tile
|
||||
source: cub/cub/device/dispatch/tuning/tuning_transform_tile.cuh
|
||||
parameters:
|
||||
items:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
379
muh/schema/unique_by_key.yaml
Normal file
379
muh/schema/unique_by_key.yaml
Normal file
@@ -0,0 +1,379 @@
|
||||
# muh schema for unique_by_key
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_unique_by_key.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: unique_by_key
|
||||
source: cub/cub/device/dispatch/tuning/tuning_unique_by_key.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 320
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 192
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 288
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 288
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 640
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
sm100:
|
||||
-
|
||||
threads: 512
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 288
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 576
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
Reference in New Issue
Block a user