diff --git a/qwen3_6_scripts/patch_xformers_sdpa_seq.py b/qwen3_6_scripts/patch_xformers_sdpa_seq.py index 8de1565a..af7037c2 100644 --- a/qwen3_6_scripts/patch_xformers_sdpa_seq.py +++ b/qwen3_6_scripts/patch_xformers_sdpa_seq.py @@ -200,6 +200,10 @@ FALLBACK_METHOD = ''' max_seqlen = max(seq_lens_list) try: + # Skip flash_attn during profiling — OOMs on large dummy batch + import os + if os.environ.get("BI100_IN_STARTUP_PROFILE") == "1": + raise RuntimeError("skip flash_attn during profiling") out = _ixf.flash_attn_varlen_func( q_flat.to(torch.float16), k_flat.to(torch.float16),