[MUH] Bootstrap muh toolchain — extract/parse/gen_yaml/gen_patch + baseline.muh

Pipeline:
  1. extract.py: Parses all 26 CCCL tuning_*.cuh → 26 YAML schemas in muh/schema/
  2. parse.py: .muh file parser with extends-inheritance + schema validation
  3. gen_yaml.py: .muh → computility-run.yaml (verified: matches competition reference)
  4. gen_patch.py: .muh → vllm kernel unified diff patches (6 algorithm mappings)
  5. baseline.muh: Competition reference config, all tuning values pending BI-V100 benchmarks

Schemas extracted:
  26 algorithms, 8-19 params each, SM75/80/90/100 reference tunings
  Priority mapping: reduce→attention, topk→sampling, scan→paged_attention,
  transform→activations, batch_memcpy→KV_cache, for→RoPE

Tested: extract→parse→validate→gen_yaml→gen_patch full pipeline passes
This commit is contained in:
dylanyunlon
2026-07-30 10:39:06 +00:00
parent 70e80c5810
commit 9b21a13119
33 changed files with 4702 additions and 0 deletions

32
muh/schema/_index.yaml Normal file
View File

@@ -0,0 +1,32 @@
# muh schema index — all extracted CCCL tuning algorithms
algorithms:
- adjacent_difference
- batch_memcpy
- batched_topk
- find
- find_bound_sorted_values
- for
- histogram
- merge
- merge_sort
- radix_sort
- reduce
- reduce_by_key
- rle_encode
- rle_non_trivial_runs
- scan
- scan_by_key
- segmented_radix_sort
- segmented_reduce
- segmented_scan
- segmented_sort
- select_if
- three_way_partition
- topk
- transform
- transform_tile
- unique_by_key
total: 26
source: cccl_upstream/cub/cub/device/dispatch/tuning/

View File

@@ -0,0 +1,56 @@
# muh schema for adjacent_difference
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_adjacent_difference.cuh
# Generated by muh/extract.py
algorithm: adjacent_difference
source: cub/cub/device/dispatch/tuning/tuning_adjacent_difference.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
store_algorithm:
type: enum
values:
- BLOCK_STORE_DIRECT
- BLOCK_STORE_WARP_TRANSPOSE
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
- BLOCK_STORE_STRIPED
value_type_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,67 @@
# muh schema for batch_memcpy
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_batch_memcpy.cuh
# Generated by muh/extract.py
algorithm: batch_memcpy
source: cub/cub/device/dispatch/tuning/tuning_batch_memcpy.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
buffers_per_thread:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bytes_per_thread:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
block_level_tile_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
warp_level_threshold:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
block_level_threshold:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
buffer_lookback_delay:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
block_lookback_delay:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,46 @@
# muh schema for batched_topk
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_batched_topk.cuh
# Generated by muh/extract.py
algorithm: batched_topk
source: cub/cub/device/dispatch/tuning/tuning_batched_topk.cuh
parameters:
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
store_algorithm:
type: enum
values:
- BLOCK_STORE_DIRECT
- BLOCK_STORE_WARP_TRANSPOSE
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
- BLOCK_STORE_STRIPED
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

46
muh/schema/find.yaml Normal file
View File

@@ -0,0 +1,46 @@
# muh schema for find
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_find.cuh
# Generated by muh/extract.py
algorithm: find
source: cub/cub/device/dispatch/tuning/tuning_find.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
vec_size:
type: int
range:
- 1
- 8
step: 1
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
input_type_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,47 @@
# muh schema for find_bound_sorted_values
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_find_bound_sorted_values.cuh
# Generated by muh/extract.py
algorithm: find_bound_sorted_values
source: cub/cub/device/dispatch/tuning/tuning_find_bound_sorted_values.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
range_type_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
values_type_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

24
muh/schema/for.yaml Normal file
View File

@@ -0,0 +1,24 @@
# muh schema for for
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_for.cuh
# Generated by muh/extract.py
algorithm: for
source: cub/cub/device/dispatch/tuning/tuning_for.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

117
muh/schema/histogram.yaml Normal file
View File

@@ -0,0 +1,117 @@
# muh schema for histogram
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_histogram.cuh
# Generated by muh/extract.py
algorithm: histogram
source: cub/cub/device/dispatch/tuning/tuning_histogram.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
pixels_per_thread:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
vec_size:
type: int
range:
- 1
- 8
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
init_kernel_pdl_trigger_max_bins:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
sample_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
counter_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
sample_size_bytes:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
num_channels:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
num_active_channels:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm90:
-
threads: 768
items: 12
load_modifier: LOAD_LDG
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 960
items: 10
load_modifier: LOAD_DEFAULT
load_algorithm: BLOCK_LOAD_DIRECT
sm100:
-
items: 12
threads: 928
load_modifier: LOAD_CA
load_algorithm: BLOCK_LOAD_DIRECT
vec_size: 1
-
items: 12
threads: 448
load_modifier: LOAD_LDG
load_algorithm: BLOCK_LOAD_DIRECT
vec_size: 1
init_kernel_pdl_trigger_max_bins: 2048
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

89
muh/schema/merge.yaml Normal file
View File

@@ -0,0 +1,89 @@
# muh schema for merge
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_merge.cuh
# Generated by muh/extract.py
algorithm: merge
source: cub/cub/device/dispatch/tuning/tuning_merge.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
store_algorithm:
type: enum
values:
- BLOCK_STORE_DIRECT
- BLOCK_STORE_WARP_TRANSPOSE
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
- BLOCK_STORE_STRIPED
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
key_align:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
key_is_trivially_relocatable:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
value_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
value_align:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
value_is_trivially_relocatable:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
offset_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,63 @@
# muh schema for merge_sort
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_merge_sort.cuh
# Generated by muh/extract.py
algorithm: merge_sort
source: cub/cub/device/dispatch/tuning/tuning_merge_sort.cuh
parameters:
ItemsPerThread:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
store_algorithm:
type: enum
values:
- BLOCK_STORE_DIRECT
- BLOCK_STORE_WARP_TRANSPOSE
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
- BLOCK_STORE_STRIPED
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

103
muh/schema/radix_sort.yaml Normal file
View File

@@ -0,0 +1,103 @@
# muh schema for radix_sort
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_radix_sort.cuh
# Generated by muh/extract.py
algorithm: radix_sort
source: cub/cub/device/dispatch/tuning/tuning_radix_sort.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
private_partitions:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
radix_bits:
type: int
range:
- 4
- 8
step: 1
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
rank_private_partitions:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
threads:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
items:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
value_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
offset_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

102
muh/schema/reduce.yaml Normal file
View File

@@ -0,0 +1,102 @@
# muh schema for reduce
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_reduce.cuh
# Generated by muh/extract.py
algorithm: reduce
source: cub/cub/device/dispatch/tuning/tuning_reduce.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
vec_size:
type: int
range:
- 1
- 8
step: 1
reduce_algorithm:
type: enum
values:
- BLOCK_REDUCE_RAKING
- BLOCK_REDUCE_RAKING_COMMUTATIVE_ONLY
- BLOCK_REDUCE_WARP_REDUCTIONS
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
items:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
threads:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
items_per_vec_load:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
offset_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
accum_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm100:
-
items: 15
threads: 512
items_per_vec_load: 2
-
items: 15
threads: 512
items_per_vec_load: 1
-
items: 16
threads: 512
items_per_vec_load: 2
-
items: 16
threads: 640
items_per_vec_load: 1
-
threads_per_block: 256
items_per_thread: 16
items_per_vec_load: 4
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,390 @@
# muh schema for reduce_by_key
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_reduce_by_key.cuh
# Generated by muh/extract.py
algorithm: reduce_by_key
source: cub/cub/device/dispatch/tuning/tuning_reduce_by_key.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
lookback_delay.kind:
type: enum
values:
- no_delay
- fixed_delay
- exponential_backoff
- exponential_backoff_jitter
- exponential_backoff_jitter_window
- exponential_backon_jitter_window
- exponential_backon_jitter
- exponential_backon
lookback_delay.delay:
type: int
range:
- 0
- 2000
step: 50
lookback_delay.l2_write_latency:
type: int
range:
- 0
- 2000
step: 50
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
accum_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm80:
-
threads: 256
items: 13
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 224
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 128
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 11
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 224
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 160
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 288
items: 11
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 192
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 15
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 160
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 224
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 384
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 128
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 192
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
sm90:
-
threads: 256
items: 13
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 320
items: 23
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 192
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 13
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 23
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 11
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 19
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 13
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 18
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 18
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 13
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 9
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 128
items: 13
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 23
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
nominal_4B_items_per_thread: 6
sm100:
-
items: 13
threads: 576
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_CA
-
items: 10
threads: 224
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
items: 14
threads: 128
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
items: 19
threads: 128
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
items: 14
threads: 128
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 14
threads: 256
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 11
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
items: 10
threads: 160
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
items: 10
threads: 224
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
items: 11
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
items: 14
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 10
threads: 256
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
items: 9
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 11
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 9
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 9
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

155
muh/schema/rle_encode.yaml Normal file
View File

@@ -0,0 +1,155 @@
# muh schema for rle_encode
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh
# Generated by muh/extract.py
algorithm: rle_encode
source: cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
lookback_delay.kind:
type: enum
values:
- no_delay
- fixed_delay
- exponential_backoff
- exponential_backoff_jitter
- exponential_backoff_jitter_window
- exponential_backon_jitter_window
- exponential_backon_jitter
- exponential_backon
lookback_delay.delay:
type: int
range:
- 0
- 2000
step: 50
lookback_delay.l2_write_latency:
type: int
range:
- 0
- 2000
step: 50
length_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm80:
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 13
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 13
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
sm90:
-
threads: 256
items: 13
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 128
items: 22
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 192
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 19
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
sm100:
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_CA
-
threads: 224
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_CA
-
threads: 224
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
nominal_4B_items_per_thread: 6
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,155 @@
# muh schema for rle_non_trivial_runs
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_rle_non_trivial_runs.cuh
# Generated by muh/extract.py
algorithm: rle_non_trivial_runs
source: cub/cub/device/dispatch/tuning/tuning_rle_non_trivial_runs.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
lookback_delay.kind:
type: enum
values:
- no_delay
- fixed_delay
- exponential_backoff
- exponential_backoff_jitter
- exponential_backoff_jitter_window
- exponential_backon_jitter_window
- exponential_backon_jitter
- exponential_backon
lookback_delay.delay:
type: int
range:
- 0
- 2000
step: 50
lookback_delay.l2_write_latency:
type: int
range:
- 0
- 2000
step: 50
length_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm80:
-
threads: 192
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 192
items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 224
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 13
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 13
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
sm90:
-
threads: 256
items: 18
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 18
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 288
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
sm100:
-
threads: 224
items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
threads: 224
items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 224
items: 13
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 288
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
nominal_4B_items_per_thread: 15
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

255
muh/schema/scan.yaml Normal file
View File

@@ -0,0 +1,255 @@
# muh schema for scan
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_scan.cuh
# Generated by muh/extract.py
algorithm: scan
source: cub/cub/device/dispatch/tuning/tuning_scan.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
store_algorithm:
type: enum
values:
- BLOCK_STORE_DIRECT
- BLOCK_STORE_WARP_TRANSPOSE
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
- BLOCK_STORE_STRIPED
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
lookback_delay.kind:
type: enum
values:
- no_delay
- fixed_delay
- exponential_backoff
- exponential_backoff_jitter
- exponential_backoff_jitter_window
- exponential_backon_jitter_window
- exponential_backon_jitter
- exponential_backon
lookback_delay.delay:
type: int
range:
- 0
- 2000
step: 50
lookback_delay.l2_write_latency:
type: int
range:
- 0
- 2000
step: 50
reduce_and_scan_warps:
type: int
range:
- 1
- 8
step: 1
lookahead_items_per_thread:
type: int
range:
- 1
- 16
step: 1
lookahead_stages:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
block_idx_stages:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
delay_constructor:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
input_value_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
input_value_alignment:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
output_value_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
output_value_alignment:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
accum_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
accum_alignment:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
offset_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm75:
-
threads: 128
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
sm80:
-
threads: 320
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 352
items: 16
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 320
items: 12
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 288
items: 22
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 288
items: 8
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 384
items: 12
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 640
items: 24
load_algorithm: BLOCK_LOAD_DIRECT
store_algorithm: BLOCK_STORE_DIRECT
sm90:
sm100:
-
items: 18
threads: 512
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 14
threads: 384
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 13
threads: 512
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 13
threads: 288
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 22
threads: 384
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 19
threads: 416
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 23
threads: 416
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 22
threads: 320
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

450
muh/schema/scan_by_key.yaml Normal file
View File

@@ -0,0 +1,450 @@
# muh schema for scan_by_key
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_scan_by_key.cuh
# Generated by muh/extract.py
algorithm: scan_by_key
source: cub/cub/device/dispatch/tuning/tuning_scan_by_key.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
store_algorithm:
type: enum
values:
- BLOCK_STORE_DIRECT
- BLOCK_STORE_WARP_TRANSPOSE
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
- BLOCK_STORE_STRIPED
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
lookback_delay.kind:
type: enum
values:
- no_delay
- fixed_delay
- exponential_backoff
- exponential_backoff_jitter
- exponential_backoff_jitter_window
- exponential_backon_jitter_window
- exponential_backon_jitter
- exponential_backon
lookback_delay.delay:
type: int
range:
- 0
- 2000
step: 50
lookback_delay.l2_write_latency:
type: int
range:
- 0
- 2000
step: 50
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
value_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
accum_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm80:
-
threads: 128
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
store_algorithm: BLOCK_STORE_DIRECT
-
threads: 288
items: 12
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 256
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 192
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 19
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 256
items: 8
load_algorithm: BLOCK_LOAD_DIRECT
store_algorithm: BLOCK_STORE_DIRECT
-
threads: 320
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 256
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 160
items: 17
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 160
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
store_algorithm: BLOCK_STORE_DIRECT
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 256
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 256
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 17
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 256
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 224
items: 13
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 224
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 320
items: 8
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 192
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 192
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 256
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
sm90:
-
threads: 128
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
store_algorithm: BLOCK_STORE_DIRECT
-
threads: 256
items: 16
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 224
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 23
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
store_algorithm: BLOCK_STORE_DIRECT
-
threads: 128
items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 22
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 224
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 23
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
store_algorithm: BLOCK_STORE_DIRECT
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 288
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 224
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 23
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
store_algorithm: BLOCK_STORE_DIRECT
-
threads: 224
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 192
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 224
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 192
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 192
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 224
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 256
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 192
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
-
threads: 128
items: 23
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
sm100:
-
items: 13
threads: 288
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 13
threads: 288
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 19
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 18
threads: 192
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 12
threads: 384
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 14
threads: 160
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 14
threads: 160
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 13
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 20
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 13
threads: 288
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 20
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 14
threads: 224
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 12
threads: 160
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 15
threads: 288
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
items: 22
threads: 160
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
items: 23
threads: 256
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
nominal_4b_items_per_thread: 9
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,26 @@
# muh schema for segmented_radix_sort
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_radix_sort.cuh
# Generated by muh/extract.py
algorithm: segmented_radix_sort
source: cub/cub/device/dispatch/tuning/tuning_segmented_radix_sort.cuh
parameters:
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
value_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,60 @@
# muh schema for segmented_reduce
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_reduce.cuh
# Generated by muh/extract.py
algorithm: segmented_reduce
source: cub/cub/device/dispatch/tuning/tuning_segmented_reduce.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
threads_per_warp:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
items_per_thread:
type: int
range:
- 1
- 32
step: 1
vec_size:
type: int
range:
- 1
- 8
step: 1
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
offset_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
accum_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,76 @@
# muh schema for segmented_scan
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_scan.cuh
# Generated by muh/extract.py
algorithm: segmented_scan
source: cub/cub/device/dispatch/tuning/tuning_segmented_scan.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
store_algorithm:
type: enum
values:
- BLOCK_STORE_DIRECT
- BLOCK_STORE_WARP_TRANSPOSE
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
- BLOCK_STORE_STRIPED
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
max_segments:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
accum_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
accum_align:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,82 @@
# muh schema for segmented_sort
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_sort.cuh
# Generated by muh/extract.py
algorithm: segmented_sort
source: cub/cub/device/dispatch/tuning/tuning_segmented_sort.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
radix_bits:
type: int
range:
- 4
- 8
step: 1
threads_per_warp:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
partitioning_threshold:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
value_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

508
muh/schema/select_if.yaml Normal file
View File

@@ -0,0 +1,508 @@
# muh schema for select_if
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_select_if.cuh
# Generated by muh/extract.py
algorithm: select_if
source: cub/cub/device/dispatch/tuning/tuning_select_if.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
lookback_delay.kind:
type: enum
values:
- no_delay
- fixed_delay
- exponential_backoff
- exponential_backoff_jitter
- exponential_backoff_jitter_window
- exponential_backon_jitter_window
- exponential_backon_jitter
- exponential_backon
lookback_delay.delay:
type: int
range:
- 0
- 2000
step: 50
lookback_delay.l2_write_latency:
type: int
range:
- 0
- 2000
step: 50
input_size_bytes:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
flag_size_bytes:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
offset_size_bytes:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm80:
-
threads: 992
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 576
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 18
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 384
items: 4
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 320
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 384
items: 6
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 5
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 512
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 18
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 15
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 192
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 5
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 512
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 18
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 192
items: 12
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 5
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
sm90:
-
threads: 256
items: 22
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 22
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 384
items: 17
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 384
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 512
items: 5
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 448
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 448
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 384
items: 15
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 384
items: 11
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 512
items: 3
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 384
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 320
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 128
items: 12
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 192
items: 5
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 512
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 224
items: 6
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 160
items: 5
load_algorithm: BLOCK_LOAD_DIRECT
sm100:
-
threads: 384
nominal_4b_items: 22
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 448
nominal_4b_items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 256
nominal_4b_items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 384
nominal_4b_items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 384
nominal_4b_items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 512
nominal_4b_items: 19
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 384
nominal_4b_items: 23
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 512
nominal_4b_items: 5
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 512
nominal_4b_items: 5
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 896
nominal_4b_items: 20
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 1024
nominal_4b_items: 20
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
nominal_4b_items: 22
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 448
nominal_4b_items: 20
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 512
nominal_4b_items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
nominal_4b_items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 320
nominal_4b_items: 22
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_CA
-
threads: 384
nominal_4b_items: 21
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_CA
-
threads: 512
nominal_4b_items: 3
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 512
nominal_4b_items: 3
load_algorithm: BLOCK_LOAD_DIRECT
-
nominal_4b_items: 15
threads: 608
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 22
threads: 320
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 19
threads: 320
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 20
threads: 416
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 22
threads: 576
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 20
threads: 608
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 18
threads: 608
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 14
threads: 512
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 22
threads: 224
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 22
threads: 320
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 19
threads: 608
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 23
threads: 416
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 20
threads: 608
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 22
threads: 288
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 19
threads: 608
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 23
threads: 416
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 20
threads: 448
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 18
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 19
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 21
threads: 384
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 20
threads: 448
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_CA
-
nominal_4b_items: 14
threads: 320
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 14
threads: 640
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 19
threads: 384
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 24
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 18
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 11
threads: 448
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 20
threads: 384
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 12
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 12
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 14
threads: 352
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
nominal_4b_items: 11
threads: 512
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
nominal_4B_items_per_thread: 10
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,185 @@
# muh schema for three_way_partition
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_three_way_partition.cuh
# Generated by muh/extract.py
algorithm: three_way_partition
source: cub/cub/device/dispatch/tuning/tuning_three_way_partition.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
lookback_delay.kind:
type: enum
values:
- no_delay
- fixed_delay
- exponential_backoff
- exponential_backoff_jitter
- exponential_backoff_jitter_window
- exponential_backon_jitter_window
- exponential_backon_jitter
- exponential_backon
lookback_delay.delay:
type: int
range:
- 0
- 2000
step: 50
lookback_delay.l2_write_latency:
type: int
range:
- 0
- 2000
step: 50
input_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
offset_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm80:
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 224
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
sm90:
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 320
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 384
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 24
load_algorithm: BLOCK_LOAD_DIRECT
-
threads: 640
items: 24
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 23
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 18
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 256
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
threads: 128
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
sm100:
-
items: 12
threads: 256
load_algorithm: BLOCK_LOAD_DIRECT
-
items: 14
threads: 288
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
items: 11
threads: 512
load_algorithm: BLOCK_LOAD_DIRECT
-
items: 10
threads: 256
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
items: 20
threads: 768
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
items: 20
threads: 768
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
items: 15
threads: 768
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
-
items: 14
threads: 320
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

52
muh/schema/topk.yaml Normal file
View File

@@ -0,0 +1,52 @@
# muh schema for topk
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_topk.cuh
# Generated by muh/extract.py
algorithm: topk
source: cub/cub/device/dispatch/tuning/tuning_topk.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
bits_per_pass:
type: int
range:
- 4
- 11
step: 1
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

107
muh/schema/transform.yaml Normal file
View File

@@ -0,0 +1,107 @@
# muh schema for transform
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_transform.cuh
# Generated by muh/extract.py
algorithm: transform
source: cub/cub/device/dispatch/tuning/tuning_transform.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread_no_input:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
min_items_per_thread:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
max_items_per_thread:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
prefetch_byte_stride:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
unroll_factor:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
items_per_thread:
type: int
range:
- 1
- 32
step: 1
vec_size:
type: int
range:
- 1
- 8
step: 1
store_vec_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
min_bytes_in_flight:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
copy_alignment:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
smem_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
tile_padding:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
max_alignment:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,19 @@
# muh schema for transform_tile
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_transform_tile.cuh
# Generated by muh/extract.py
algorithm: transform_tile
source: cub/cub/device/dispatch/tuning/tuning_transform_tile.cuh
parameters:
items:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD

View File

@@ -0,0 +1,379 @@
# muh schema for unique_by_key
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_unique_by_key.cuh
# Generated by muh/extract.py
algorithm: unique_by_key
source: cub/cub/device/dispatch/tuning/tuning_unique_by_key.cuh
parameters:
threads_per_block:
type: int
range:
- 32
- 1024
step: 32
items_per_thread:
type: int
range:
- 1
- 32
step: 1
load_algorithm:
type: enum
values:
- BLOCK_LOAD_DIRECT
- BLOCK_LOAD_VECTORIZE
- BLOCK_LOAD_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
- BLOCK_LOAD_STRIPED
load_modifier:
type: enum
values:
- LOAD_DEFAULT
- LOAD_CA
- LOAD_CG
- LOAD_CS
- LOAD_CV
- LOAD_LDG
scan_algorithm:
type: enum
values:
- BLOCK_SCAN_RAKING
- BLOCK_SCAN_RAKING_MEMOIZE
- BLOCK_SCAN_WARP_SCANS
lookback_delay.kind:
type: enum
values:
- no_delay
- fixed_delay
- exponential_backoff
- exponential_backoff_jitter
- exponential_backoff_jitter_window
- exponential_backon_jitter_window
- exponential_backon_jitter
- exponential_backon
lookback_delay.delay:
type: int
range:
- 0
- 2000
step: 50
lookback_delay.l2_write_latency:
type: int
range:
- 0
- 2000
step: 50
key_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
value_size:
type: int
range:
- 1
- 1024
step: 1
note: unknown_range
reference_tunings:
sm80:
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 224
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 128
items: 15
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 320
items: 20
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 192
items: 22
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 10
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 11
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 192
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 128
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
sm90:
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 448
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 288
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 288
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 23
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 224
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 448
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 9
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 14
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 9
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 9
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 640
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 448
items: 11
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
sm100:
-
threads: 512
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 288
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 12
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 11
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_CA
-
threads: 384
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 224
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 11
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 512
items: 14
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 7
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 9
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 384
items: 10
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 576
items: 7
load_algorithm: BLOCK_LOAD_DIRECT
load_modifier: LOAD_DEFAULT
-
threads: 256
items: 9
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
load_modifier: LOAD_DEFAULT
bi_v100:
status: pending_benchmark
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
threads_per_block: TBD
items_per_thread: TBD