[MUH] Bootstrap muh toolchain — extract/parse/gen_yaml/gen_patch + baseline.muh
Pipeline: 1. extract.py: Parses all 26 CCCL tuning_*.cuh → 26 YAML schemas in muh/schema/ 2. parse.py: .muh file parser with extends-inheritance + schema validation 3. gen_yaml.py: .muh → computility-run.yaml (verified: matches competition reference) 4. gen_patch.py: .muh → vllm kernel unified diff patches (6 algorithm mappings) 5. baseline.muh: Competition reference config, all tuning values pending BI-V100 benchmarks Schemas extracted: 26 algorithms, 8-19 params each, SM75/80/90/100 reference tunings Priority mapping: reduce→attention, topk→sampling, scan→paged_attention, transform→activations, batch_memcpy→KV_cache, for→RoPE Tested: extract→parse→validate→gen_yaml→gen_patch full pipeline passes
This commit is contained in:
32
muh/schema/_index.yaml
Normal file
32
muh/schema/_index.yaml
Normal file
@@ -0,0 +1,32 @@
|
||||
# muh schema index — all extracted CCCL tuning algorithms
|
||||
|
||||
algorithms:
|
||||
- adjacent_difference
|
||||
- batch_memcpy
|
||||
- batched_topk
|
||||
- find
|
||||
- find_bound_sorted_values
|
||||
- for
|
||||
- histogram
|
||||
- merge
|
||||
- merge_sort
|
||||
- radix_sort
|
||||
- reduce
|
||||
- reduce_by_key
|
||||
- rle_encode
|
||||
- rle_non_trivial_runs
|
||||
- scan
|
||||
- scan_by_key
|
||||
- segmented_radix_sort
|
||||
- segmented_reduce
|
||||
- segmented_scan
|
||||
- segmented_sort
|
||||
- select_if
|
||||
- three_way_partition
|
||||
- topk
|
||||
- transform
|
||||
- transform_tile
|
||||
- unique_by_key
|
||||
|
||||
total: 26
|
||||
source: cccl_upstream/cub/cub/device/dispatch/tuning/
|
||||
56
muh/schema/adjacent_difference.yaml
Normal file
56
muh/schema/adjacent_difference.yaml
Normal file
@@ -0,0 +1,56 @@
|
||||
# muh schema for adjacent_difference
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_adjacent_difference.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: adjacent_difference
|
||||
source: cub/cub/device/dispatch/tuning/tuning_adjacent_difference.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
value_type_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
67
muh/schema/batch_memcpy.yaml
Normal file
67
muh/schema/batch_memcpy.yaml
Normal file
@@ -0,0 +1,67 @@
|
||||
# muh schema for batch_memcpy
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_batch_memcpy.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: batch_memcpy
|
||||
source: cub/cub/device/dispatch/tuning/tuning_batch_memcpy.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
buffers_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bytes_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
block_level_tile_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
warp_level_threshold:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
block_level_threshold:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
buffer_lookback_delay:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
block_lookback_delay:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
46
muh/schema/batched_topk.yaml
Normal file
46
muh/schema/batched_topk.yaml
Normal file
@@ -0,0 +1,46 @@
|
||||
# muh schema for batched_topk
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_batched_topk.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: batched_topk
|
||||
source: cub/cub/device/dispatch/tuning/tuning_batched_topk.cuh
|
||||
parameters:
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
46
muh/schema/find.yaml
Normal file
46
muh/schema/find.yaml
Normal file
@@ -0,0 +1,46 @@
|
||||
# muh schema for find
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_find.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: find
|
||||
source: cub/cub/device/dispatch/tuning/tuning_find.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
input_type_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
47
muh/schema/find_bound_sorted_values.yaml
Normal file
47
muh/schema/find_bound_sorted_values.yaml
Normal file
@@ -0,0 +1,47 @@
|
||||
# muh schema for find_bound_sorted_values
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_find_bound_sorted_values.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: find_bound_sorted_values
|
||||
source: cub/cub/device/dispatch/tuning/tuning_find_bound_sorted_values.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
range_type_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
values_type_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
24
muh/schema/for.yaml
Normal file
24
muh/schema/for.yaml
Normal file
@@ -0,0 +1,24 @@
|
||||
# muh schema for for
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_for.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: for
|
||||
source: cub/cub/device/dispatch/tuning/tuning_for.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
117
muh/schema/histogram.yaml
Normal file
117
muh/schema/histogram.yaml
Normal file
@@ -0,0 +1,117 @@
|
||||
# muh schema for histogram
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_histogram.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: histogram
|
||||
source: cub/cub/device/dispatch/tuning/tuning_histogram.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
pixels_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
init_kernel_pdl_trigger_max_bins:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
sample_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
counter_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
sample_size_bytes:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
num_channels:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
num_active_channels:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm90:
|
||||
-
|
||||
threads: 768
|
||||
items: 12
|
||||
load_modifier: LOAD_LDG
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 960
|
||||
items: 10
|
||||
load_modifier: LOAD_DEFAULT
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
sm100:
|
||||
-
|
||||
items: 12
|
||||
threads: 928
|
||||
load_modifier: LOAD_CA
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
vec_size: 1
|
||||
-
|
||||
items: 12
|
||||
threads: 448
|
||||
load_modifier: LOAD_LDG
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
vec_size: 1
|
||||
init_kernel_pdl_trigger_max_bins: 2048
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
89
muh/schema/merge.yaml
Normal file
89
muh/schema/merge.yaml
Normal file
@@ -0,0 +1,89 @@
|
||||
# muh schema for merge
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_merge.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: merge
|
||||
source: cub/cub/device/dispatch/tuning/tuning_merge.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_align:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_is_trivially_relocatable:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_align:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_is_trivially_relocatable:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
63
muh/schema/merge_sort.yaml
Normal file
63
muh/schema/merge_sort.yaml
Normal file
@@ -0,0 +1,63 @@
|
||||
# muh schema for merge_sort
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_merge_sort.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: merge_sort
|
||||
source: cub/cub/device/dispatch/tuning/tuning_merge_sort.cuh
|
||||
parameters:
|
||||
ItemsPerThread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
103
muh/schema/radix_sort.yaml
Normal file
103
muh/schema/radix_sort.yaml
Normal file
@@ -0,0 +1,103 @@
|
||||
# muh schema for radix_sort
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_radix_sort.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: radix_sort
|
||||
source: cub/cub/device/dispatch/tuning/tuning_radix_sort.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
private_partitions:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
radix_bits:
|
||||
type: int
|
||||
range:
|
||||
- 4
|
||||
- 8
|
||||
step: 1
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
rank_private_partitions:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
threads:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
items:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
102
muh/schema/reduce.yaml
Normal file
102
muh/schema/reduce.yaml
Normal file
@@ -0,0 +1,102 @@
|
||||
# muh schema for reduce
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_reduce.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: reduce
|
||||
source: cub/cub/device/dispatch/tuning/tuning_reduce.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
reduce_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_REDUCE_RAKING
|
||||
- BLOCK_REDUCE_RAKING_COMMUTATIVE_ONLY
|
||||
- BLOCK_REDUCE_WARP_REDUCTIONS
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
items:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
threads:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
items_per_vec_load:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm100:
|
||||
-
|
||||
items: 15
|
||||
threads: 512
|
||||
items_per_vec_load: 2
|
||||
-
|
||||
items: 15
|
||||
threads: 512
|
||||
items_per_vec_load: 1
|
||||
-
|
||||
items: 16
|
||||
threads: 512
|
||||
items_per_vec_load: 2
|
||||
-
|
||||
items: 16
|
||||
threads: 640
|
||||
items_per_vec_load: 1
|
||||
-
|
||||
threads_per_block: 256
|
||||
items_per_thread: 16
|
||||
items_per_vec_load: 4
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
390
muh/schema/reduce_by_key.yaml
Normal file
390
muh/schema/reduce_by_key.yaml
Normal file
@@ -0,0 +1,390 @@
|
||||
# muh schema for reduce_by_key
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_reduce_by_key.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: reduce_by_key
|
||||
source: cub/cub/device/dispatch/tuning/tuning_reduce_by_key.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 160
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 160
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 320
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
nominal_4B_items_per_thread: 6
|
||||
sm100:
|
||||
-
|
||||
items: 13
|
||||
threads: 576
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 10
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 128
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 19
|
||||
threads: 128
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 14
|
||||
threads: 128
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 11
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 10
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 10
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 11
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 10
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 9
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 11
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 9
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 9
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
155
muh/schema/rle_encode.yaml
Normal file
155
muh/schema/rle_encode.yaml
Normal file
@@ -0,0 +1,155 @@
|
||||
# muh schema for rle_encode
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: rle_encode
|
||||
source: cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
length_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
nominal_4B_items_per_thread: 6
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
155
muh/schema/rle_non_trivial_runs.yaml
Normal file
155
muh/schema/rle_non_trivial_runs.yaml
Normal file
@@ -0,0 +1,155 @@
|
||||
# muh schema for rle_non_trivial_runs
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_rle_non_trivial_runs.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: rle_non_trivial_runs
|
||||
source: cub/cub/device/dispatch/tuning/tuning_rle_non_trivial_runs.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
length_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 192
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
threads: 224
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 224
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 224
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 288
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
nominal_4B_items_per_thread: 15
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
255
muh/schema/scan.yaml
Normal file
255
muh/schema/scan.yaml
Normal file
@@ -0,0 +1,255 @@
|
||||
# muh schema for scan
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_scan.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: scan
|
||||
source: cub/cub/device/dispatch/tuning/tuning_scan.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
reduce_and_scan_warps:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
lookahead_items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 16
|
||||
step: 1
|
||||
lookahead_stages:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
block_idx_stages:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
delay_constructor:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
input_value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
input_value_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
output_value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
output_value_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm75:
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
sm80:
|
||||
-
|
||||
threads: 320
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 352
|
||||
items: 16
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 320
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 8
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 384
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 640
|
||||
items: 24
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
sm90:
|
||||
|
||||
sm100:
|
||||
-
|
||||
items: 18
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 13
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 13
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 22
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 19
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 23
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 22
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
450
muh/schema/scan_by_key.yaml
Normal file
450
muh/schema/scan_by_key.yaml
Normal file
@@ -0,0 +1,450 @@
|
||||
# muh schema for scan_by_key
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_scan_by_key.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: scan_by_key
|
||||
source: cub/cub/device/dispatch/tuning/tuning_scan_by_key.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 288
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 8
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 320
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 160
|
||||
items: 17
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 160
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 17
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 320
|
||||
items: 8
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 16
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 288
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
store_algorithm: BLOCK_STORE_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
items: 13
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 13
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 19
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 18
|
||||
threads: 192
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 12
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 14
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 13
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 20
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 13
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 20
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 14
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 12
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 15
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
items: 22
|
||||
threads: 160
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
items: 23
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
store_algorithm: BLOCK_STORE_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
nominal_4b_items_per_thread: 9
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
26
muh/schema/segmented_radix_sort.yaml
Normal file
26
muh/schema/segmented_radix_sort.yaml
Normal file
@@ -0,0 +1,26 @@
|
||||
# muh schema for segmented_radix_sort
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_radix_sort.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: segmented_radix_sort
|
||||
source: cub/cub/device/dispatch/tuning/tuning_segmented_radix_sort.cuh
|
||||
parameters:
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
60
muh/schema/segmented_reduce.yaml
Normal file
60
muh/schema/segmented_reduce.yaml
Normal file
@@ -0,0 +1,60 @@
|
||||
# muh schema for segmented_reduce
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_reduce.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: segmented_reduce
|
||||
source: cub/cub/device/dispatch/tuning/tuning_segmented_reduce.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
threads_per_warp:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
76
muh/schema/segmented_scan.yaml
Normal file
76
muh/schema/segmented_scan.yaml
Normal file
@@ -0,0 +1,76 @@
|
||||
# muh schema for segmented_scan
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_scan.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: segmented_scan
|
||||
source: cub/cub/device/dispatch/tuning/tuning_segmented_scan.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
store_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_STORE_DIRECT
|
||||
- BLOCK_STORE_WARP_TRANSPOSE
|
||||
- BLOCK_STORE_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_STORE_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
max_segments:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
accum_align:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
82
muh/schema/segmented_sort.yaml
Normal file
82
muh/schema/segmented_sort.yaml
Normal file
@@ -0,0 +1,82 @@
|
||||
# muh schema for segmented_sort
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_segmented_sort.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: segmented_sort
|
||||
source: cub/cub/device/dispatch/tuning/tuning_segmented_sort.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
radix_bits:
|
||||
type: int
|
||||
range:
|
||||
- 4
|
||||
- 8
|
||||
step: 1
|
||||
threads_per_warp:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
partitioning_threshold:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
508
muh/schema/select_if.yaml
Normal file
508
muh/schema/select_if.yaml
Normal file
@@ -0,0 +1,508 @@
|
||||
# muh schema for select_if
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_select_if.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: select_if
|
||||
source: cub/cub/device/dispatch/tuning/tuning_select_if.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
input_size_bytes:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
flag_size_bytes:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size_bytes:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 992
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 576
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 384
|
||||
items: 4
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 320
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 6
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 512
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 384
|
||||
items: 17
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 384
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 512
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 448
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 448
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
items: 3
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 320
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 192
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 6
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 160
|
||||
items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
sm100:
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 5
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 896
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 1024
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
nominal_4b_items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
nominal_4b_items: 20
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 320
|
||||
nominal_4b_items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 384
|
||||
nominal_4b_items: 21
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 3
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 512
|
||||
nominal_4b_items: 3
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
nominal_4b_items: 15
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 576
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 18
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 14
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 224
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 23
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 22
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 608
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 23
|
||||
threads: 416
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 448
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 18
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 21
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 448
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
nominal_4b_items: 14
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 14
|
||||
threads: 640
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 19
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 24
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 18
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 11
|
||||
threads: 448
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 20
|
||||
threads: 384
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 12
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 12
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 14
|
||||
threads: 352
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
nominal_4b_items: 11
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
nominal_4B_items_per_thread: 10
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
185
muh/schema/three_way_partition.yaml
Normal file
185
muh/schema/three_way_partition.yaml
Normal file
@@ -0,0 +1,185 @@
|
||||
# muh schema for three_way_partition
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_three_way_partition.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: three_way_partition
|
||||
source: cub/cub/device/dispatch/tuning/tuning_three_way_partition.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
input_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
offset_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 224
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 320
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 384
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 24
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 640
|
||||
items: 24
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 18
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
items: 12
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
items: 14
|
||||
threads: 288
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 11
|
||||
threads: 512
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
items: 10
|
||||
threads: 256
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 20
|
||||
threads: 768
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 20
|
||||
threads: 768
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 15
|
||||
threads: 768
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
items: 14
|
||||
threads: 320
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
52
muh/schema/topk.yaml
Normal file
52
muh/schema/topk.yaml
Normal file
@@ -0,0 +1,52 @@
|
||||
# muh schema for topk
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_topk.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: topk
|
||||
source: cub/cub/device/dispatch/tuning/tuning_topk.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
bits_per_pass:
|
||||
type: int
|
||||
range:
|
||||
- 4
|
||||
- 11
|
||||
step: 1
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
107
muh/schema/transform.yaml
Normal file
107
muh/schema/transform.yaml
Normal file
@@ -0,0 +1,107 @@
|
||||
# muh schema for transform
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_transform.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: transform
|
||||
source: cub/cub/device/dispatch/tuning/tuning_transform.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread_no_input:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
min_items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
max_items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
prefetch_byte_stride:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
unroll_factor:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 8
|
||||
step: 1
|
||||
store_vec_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
min_bytes_in_flight:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
copy_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
smem_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
tile_padding:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
max_alignment:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
19
muh/schema/transform_tile.yaml
Normal file
19
muh/schema/transform_tile.yaml
Normal file
@@ -0,0 +1,19 @@
|
||||
# muh schema for transform_tile
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_transform_tile.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: transform_tile
|
||||
source: cub/cub/device/dispatch/tuning/tuning_transform_tile.cuh
|
||||
parameters:
|
||||
items:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
379
muh/schema/unique_by_key.yaml
Normal file
379
muh/schema/unique_by_key.yaml
Normal file
@@ -0,0 +1,379 @@
|
||||
# muh schema for unique_by_key
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_unique_by_key.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: unique_by_key
|
||||
source: cub/cub/device/dispatch/tuning/tuning_unique_by_key.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
value_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 224
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 15
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 320
|
||||
items: 20
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 192
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 192
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 288
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 288
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 23
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 640
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 448
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
sm100:
|
||||
-
|
||||
threads: 512
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 288
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 12
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 512
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 384
|
||||
items: 10
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 576
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
Reference in New Issue
Block a user