[MUH] Bootstrap muh toolchain — extract/parse/gen_yaml/gen_patch + baseline.muh
Pipeline: 1. extract.py: Parses all 26 CCCL tuning_*.cuh → 26 YAML schemas in muh/schema/ 2. parse.py: .muh file parser with extends-inheritance + schema validation 3. gen_yaml.py: .muh → computility-run.yaml (verified: matches competition reference) 4. gen_patch.py: .muh → vllm kernel unified diff patches (6 algorithm mappings) 5. baseline.muh: Competition reference config, all tuning values pending BI-V100 benchmarks Schemas extracted: 26 algorithms, 8-19 params each, SM75/80/90/100 reference tunings Priority mapping: reduce→attention, topk→sampling, scan→paged_attention, transform→activations, batch_memcpy→KV_cache, for→RoPE Tested: extract→parse→validate→gen_yaml→gen_patch full pipeline passes
This commit is contained in:
155
muh/schema/rle_encode.yaml
Normal file
155
muh/schema/rle_encode.yaml
Normal file
@@ -0,0 +1,155 @@
|
||||
# muh schema for rle_encode
|
||||
# Auto-extracted from cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh
|
||||
# Generated by muh/extract.py
|
||||
|
||||
algorithm: rle_encode
|
||||
source: cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh
|
||||
parameters:
|
||||
threads_per_block:
|
||||
type: int
|
||||
range:
|
||||
- 32
|
||||
- 1024
|
||||
step: 32
|
||||
items_per_thread:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 32
|
||||
step: 1
|
||||
load_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_LOAD_DIRECT
|
||||
- BLOCK_LOAD_VECTORIZE
|
||||
- BLOCK_LOAD_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE
|
||||
- BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED
|
||||
- BLOCK_LOAD_STRIPED
|
||||
load_modifier:
|
||||
type: enum
|
||||
values:
|
||||
- LOAD_DEFAULT
|
||||
- LOAD_CA
|
||||
- LOAD_CG
|
||||
- LOAD_CS
|
||||
- LOAD_CV
|
||||
- LOAD_LDG
|
||||
scan_algorithm:
|
||||
type: enum
|
||||
values:
|
||||
- BLOCK_SCAN_RAKING
|
||||
- BLOCK_SCAN_RAKING_MEMOIZE
|
||||
- BLOCK_SCAN_WARP_SCANS
|
||||
lookback_delay.kind:
|
||||
type: enum
|
||||
values:
|
||||
- no_delay
|
||||
- fixed_delay
|
||||
- exponential_backoff
|
||||
- exponential_backoff_jitter
|
||||
- exponential_backoff_jitter_window
|
||||
- exponential_backon_jitter_window
|
||||
- exponential_backon_jitter
|
||||
- exponential_backon
|
||||
lookback_delay.delay:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
lookback_delay.l2_write_latency:
|
||||
type: int
|
||||
range:
|
||||
- 0
|
||||
- 2000
|
||||
step: 50
|
||||
length_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
key_size:
|
||||
type: int
|
||||
range:
|
||||
- 1
|
||||
- 1024
|
||||
step: 1
|
||||
note: unknown_range
|
||||
reference_tunings:
|
||||
sm80:
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 7
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm90:
|
||||
-
|
||||
threads: 256
|
||||
items: 13
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 128
|
||||
items: 22
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
-
|
||||
threads: 192
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 19
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
sm100:
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 224
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 256
|
||||
items: 14
|
||||
load_algorithm: BLOCK_LOAD_DIRECT
|
||||
load_modifier: LOAD_CA
|
||||
-
|
||||
threads: 224
|
||||
items: 9
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
load_modifier: LOAD_DEFAULT
|
||||
-
|
||||
threads: 128
|
||||
items: 11
|
||||
load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE
|
||||
-
|
||||
nominal_4B_items_per_thread: 6
|
||||
bi_v100:
|
||||
status: pending_benchmark
|
||||
note: Run muh benchmark on Iluvatar BI-V100 to fill these values
|
||||
threads_per_block: TBD
|
||||
items_per_thread: TBD
|
||||
Reference in New Issue
Block a user