# muh schema for rle_encode # Auto-extracted from cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh # Generated by muh/extract.py algorithm: rle_encode source: cub/cub/device/dispatch/tuning/tuning_rle_encode.cuh parameters: threads_per_block: type: int range: - 32 - 1024 step: 32 items_per_thread: type: int range: - 1 - 32 step: 1 load_algorithm: type: enum values: - BLOCK_LOAD_DIRECT - BLOCK_LOAD_VECTORIZE - BLOCK_LOAD_TRANSPOSE - BLOCK_LOAD_WARP_TRANSPOSE - BLOCK_LOAD_WARP_TRANSPOSE_TIMESLICED - BLOCK_LOAD_STRIPED load_modifier: type: enum values: - LOAD_DEFAULT - LOAD_CA - LOAD_CG - LOAD_CS - LOAD_CV - LOAD_LDG scan_algorithm: type: enum values: - BLOCK_SCAN_RAKING - BLOCK_SCAN_RAKING_MEMOIZE - BLOCK_SCAN_WARP_SCANS lookback_delay.kind: type: enum values: - no_delay - fixed_delay - exponential_backoff - exponential_backoff_jitter - exponential_backoff_jitter_window - exponential_backon_jitter_window - exponential_backon_jitter - exponential_backon lookback_delay.delay: type: int range: - 0 - 2000 step: 50 lookback_delay.l2_write_latency: type: int range: - 0 - 2000 step: 50 length_size: type: int range: - 1 - 1024 step: 1 note: unknown_range key_size: type: int range: - 1 - 1024 step: 1 note: unknown_range reference_tunings: sm80: - threads: 256 items: 14 load_algorithm: BLOCK_LOAD_DIRECT - threads: 256 items: 13 load_algorithm: BLOCK_LOAD_DIRECT - threads: 256 items: 13 load_algorithm: BLOCK_LOAD_DIRECT - threads: 224 items: 9 load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE - threads: 128 items: 7 load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE sm90: - threads: 256 items: 13 load_algorithm: BLOCK_LOAD_DIRECT - threads: 128 items: 22 load_algorithm: BLOCK_LOAD_DIRECT - threads: 192 items: 14 load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE - threads: 128 items: 19 load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE - threads: 128 items: 11 load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE sm100: - threads: 256 items: 14 load_algorithm: BLOCK_LOAD_DIRECT load_modifier: LOAD_CA - threads: 224 items: 14 load_algorithm: BLOCK_LOAD_DIRECT load_modifier: LOAD_DEFAULT - threads: 256 items: 14 load_algorithm: BLOCK_LOAD_DIRECT load_modifier: LOAD_CA - threads: 224 items: 9 load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE load_modifier: LOAD_DEFAULT - threads: 128 items: 11 load_algorithm: BLOCK_LOAD_WARP_TRANSPOSE - nominal_4B_items_per_thread: 6 bi_v100: status: pending_benchmark note: Run muh benchmark on Iluvatar BI-V100 to fill these values threads_per_block: TBD items_per_thread: TBD