[Lint]Style: Convert vllm-ascend/ to ruff format(Batch #6) (#6001)

### What this PR does / why we need it? | File Path | | :--- | | ` vllm_ascend/eplb/adaptor/abstract_adaptor.py` | | ` vllm_ascend/eplb/adaptor/vllm_adaptor.py` | | ` vllm_ascend/eplb/core/eplb_device_transfer_loader.py` | | ` vllm_ascend/eplb/core/eplb_utils.py` | | ` vllm_ascend/eplb/core/eplb_worker.py` | | ` vllm_ascend/eplb/core/policy/policy_abstract.py` | | ` vllm_ascend/eplb/core/policy/policy_default_eplb.py` | | ` vllm_ascend/eplb/core/policy/policy_factory.py` | | ` vllm_ascend/eplb/core/policy/policy_flashlb.py` | | ` vllm_ascend/eplb/core/policy/policy_random.py` | | ` vllm_ascend/eplb/core/policy/policy_swift_balancer.py` | | ` vllm_ascend/eplb/eplb_updator.py` | | ` vllm_ascend/eplb/utils.py` | | ` vllm_ascend/model_loader/netloader/executor/elastic_load.py` | | ` vllm_ascend/model_loader/netloader/executor/netloader_pg.py` | | ` vllm_ascend/model_loader/netloader/interaction/elastic.py` | | ` vllm_ascend/model_loader/netloader/load.py` | | ` vllm_ascend/model_loader/netloader/netloader.py` | | ` vllm_ascend/model_loader/netloader/utils.py` | | ` vllm_ascend/patch/platform/__init__.py` | | ` vllm_ascend/patch/platform/patch_balance_schedule.py` | | ` vllm_ascend/patch/platform/patch_ec_connector.py` | | ` vllm_ascend/patch/platform/patch_mamba_config.py` | | ` vllm_ascend/patch/platform/patch_multiproc_executor.py` | | ` vllm_ascend/patch/platform/patch_sched_yield.py` | - vLLM version: v0.13.0 - vLLM main: 2c24bc6996 --------- Signed-off-by: MrZ20 <2609716663@qq.com>
2026-01-24 22:08:33 +08:00
parent 153da1a669
commit 4e53c1d900
26 changed files with 894 additions and 1148 deletions
--- a/vllm_ascend/patch/platform/patch_mamba_config.py
+++ b/vllm_ascend/patch/platform/patch_mamba_config.py
@@ -38,7 +38,8 @@ def verify_and_update_config(cls, vllm_config) -> None:
        block_size=1,
        num_kv_heads=model_config.get_num_kv_heads(parallel_config),
        head_size=model_config.get_head_size(),
-        dtype=kv_cache_dtype).page_size_bytes
+        dtype=kv_cache_dtype,
+    ).page_size_bytes

    model_cls, _ = ModelRegistry.resolve_model_cls(
        model_config.architecture,
@@ -58,23 +59,20 @@ def verify_and_update_config(cls, vllm_config) -> None:
    # block size to multiple of 16, so let's suggest a value
    # that would work (note: FA is currently not compatible
    # with mamba layers, use FlashInfer instead).
-    attn_block_size = block_alignment_bytes * cdiv(
-        mamba_page_size, block_alignment_bytes * attn_page_size_1_token)
+    attn_block_size = block_alignment_bytes * cdiv(mamba_page_size, block_alignment_bytes * attn_page_size_1_token)

    # override attention block size if either (a) the
    # user has not set it or (b) the user has set it
    # too small.
-    if (cache_config.block_size is None
-            or cache_config.block_size < attn_block_size):
+    if cache_config.block_size is None or cache_config.block_size < attn_block_size:
        cache_config.block_size = attn_block_size
        logger.info(
-            "Setting attention block size to %d tokens "
-            "to ensure that attention page size is >= mamba page size.",
-            attn_block_size)
+            "Setting attention block size to %d tokens to ensure that attention page size is >= mamba page size.",
+            attn_block_size,
+        )

    # compute new attention page size
-    attn_page_size = \
-        cache_config.block_size * attn_page_size_1_token
+    attn_page_size = cache_config.block_size * attn_page_size_1_token

    assert attn_page_size >= mamba_page_size

@@ -83,15 +81,15 @@ def verify_and_update_config(cls, vllm_config) -> None:
        return

    # pad mamba page size to exactly match attention
-    if (cache_config.mamba_page_size_padded is None
-            or cache_config.mamba_page_size_padded != attn_page_size):
-        cache_config.mamba_page_size_padded = (attn_page_size)
-        mamba_padding_pct = 100 * (attn_page_size -
-                                   mamba_page_size) / mamba_page_size
+    if cache_config.mamba_page_size_padded is None or cache_config.mamba_page_size_padded != attn_page_size:
+        cache_config.mamba_page_size_padded = attn_page_size
+        mamba_padding_pct = 100 * (attn_page_size - mamba_page_size) / mamba_page_size
        logger.info(
            "Padding mamba page size by %.2f%% to ensure "
            "that mamba page size and attention page size are "
-            "exactly equal.", mamba_padding_pct)
+            "exactly equal.",
+            mamba_padding_pct,
+        )


 vllm.model_executor.models.config.HybridAttentionMambaModelConfig.verify_and_update_config = verify_and_update_config