init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -0,0 +1,232 @@
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import contextlib
import gc
import math
import multiprocessing
import os
from typing import Any
from unittest.mock import patch
import pytest
import torch
from vllm.utils.network_utils import get_open_port
from tests.e2e.conftest import wait_until_npu_memory_free
from vllm_ascend.utils import AscendDeviceType, get_ascend_device_type
MODELS = [
# Offline data parallel mode will be not supported/useful for dense models
# "Qwen/Qwen3-0.6B",
"vllm-ascend/DeepSeek-V2-Lite-W8A8",
]
def _install_spies(counters: dict[str, Any]) -> contextlib.ExitStack:
"""Installs thread-safe spies on NPU methods to track invocation counts."""
from vllm_ascend.worker.model_runner_v1 import NPUModelRunner
def make_spy(cls, method_name, counter):
original = getattr(cls, method_name)
def spy(self, *args, **kwargs):
with counter.get_lock():
counter.value += 1
return original(self, *args, **kwargs)
return spy
stack = contextlib.ExitStack()
hooks = [
(torch.npu.NPUGraph, "replay", counters["replay"]),
(torch.npu.NPUGraph, "__init__", counters["capture"]),
(NPUModelRunner, "execute_model", counters["exec_model"]),
(NPUModelRunner, "_dummy_run", counters["dummy_run"]),
]
for cls, method, counter in hooks:
stack.enter_context(patch.object(cls, method, make_spy(cls, method, counter)))
return stack
def _run_worker_process(
rank: int,
local_rank: int,
world_size: int,
master_ip: str,
master_port: int,
counters: dict[str, Any],
model_path: str,
max_tokens: int,
):
"""Main entry point for the worker process."""
os.environ.update(
{
"VLLM_DP_RANK": str(rank),
"VLLM_DP_RANK_LOCAL": str(local_rank),
"VLLM_DP_SIZE": str(world_size),
"VLLM_DP_MASTER_IP": master_ip,
"VLLM_DP_MASTER_PORT": str(master_port),
}
)
# Import vLLM only after environment setup
from vllm import LLM, SamplingParams
from vllm.distributed.parallel_state import destroy_distributed_environment, destroy_model_parallel
# Apply hooks and run inference
with _install_spies(counters):
prompts = [
"Hello, my name is",
"The president of the United States is",
"The capital of France is",
"The future of AI is",
]
# Simple data sharding
chunk_size = len(prompts) // world_size
start_idx = rank * chunk_size
end_idx = start_idx + chunk_size if rank < world_size - 1 else len(prompts)
local_prompts = prompts[start_idx:end_idx]
llm = LLM(
model=model_path,
quantization="ascend" if "W8A8" in model_path else None,
enable_expert_parallel="DeepSeek" in model_path,
trust_remote_code=True,
)
# Expose model config to the main test process
counters["hidden_layers"].value = llm.llm_engine.model_config.hf_text_config.num_hidden_layers
llm.generate(local_prompts, SamplingParams(max_tokens=max_tokens, temperature=0.0))
# Explicit cleanup is mandatory in multi-process vLLM tests
del llm
destroy_model_parallel()
destroy_distributed_environment()
with contextlib.suppress(AssertionError):
torch.distributed.destroy_process_group()
gc.collect()
torch.npu.empty_cache()
torch.npu.reset_peak_memory_stats()
@pytest.mark.skip(reason="fix me")
@pytest.mark.parametrize("model", MODELS)
@pytest.mark.parametrize("max_tokens", [4, 36])
@patch.dict(os.environ, {"ASCEND_RT_VISIBLE_DEVICES": "0,1"})
@wait_until_npu_memory_free(target_free_percentage=0.6)
def test_models_aclgraph_capture_replay_metrics_dp2(
model: str,
max_tokens: int,
monkeypatch: pytest.MonkeyPatch,
) -> None:
# Counter doesn't work in default "spawn" mode
monkeypatch.delenv("VLLM_WORKER_MULTIPROC_METHOD", raising=False)
# Shared counters for cross-process assertion
counters = {
"replay": multiprocessing.Value("i", 0),
"capture": multiprocessing.Value("i", 0),
"exec_model": multiprocessing.Value("i", 0),
"dummy_run": multiprocessing.Value("i", 0),
"hidden_layers": multiprocessing.Value("i", -1),
}
dp_size = 2
port = get_open_port()
# Launch workers
workers = []
for rank in range(dp_size):
p = multiprocessing.Process(
target=_run_worker_process,
args=(rank, rank, dp_size, "127.0.0.1", port, counters, model, max_tokens),
)
p.start()
workers.append(p)
# Supervision loop
for p in workers:
p.join(timeout=900)
if p.exitcode != 0:
for k in workers:
if k.is_alive():
k.kill()
raise RuntimeError(f"Worker {p.pid} failed with exit code {p.exitcode}")
actual_capture = counters["capture"].value
actual_replay = counters["replay"].value
num_execute_model = counters["exec_model"].value
num_dummy_run = counters["dummy_run"].value
num_layers = counters["hidden_layers"].value
num_acl_graphs = num_layers + 1
num_comm_groups = sum(1 for s in [dp_size, 1] if s > 1) # dp_size=2, tp_size=1
# Metric 1: Graph Capture (ACL Graph Construction)
# Ref: vllm_ascend.utils.update_aclgraph_sizes
max_batch_sizes = math.floor((1800 - num_comm_groups * 40) / num_acl_graphs / (1 + num_comm_groups * 2))
expected_capture = max_batch_sizes * num_acl_graphs * dp_size
assert actual_capture == expected_capture, (
f"Capture count mismatch. Expected: {expected_capture}, Got: {actual_capture}"
)
# Metric 2: Model Execution (NPUModelRunner.execute_model)
# vLLM Step Breakdown:
# 1. First step (prefill, 1 prompt)
# 2. Generation steps (max_tokens)
# 3. Final step (likely EOS/idle step), no replay here
total_steps = max_tokens + 1 # this includes the 1 and 2 above
# vllm default enables Async scheduler, this will take 1 more steps
expected_exec_model = (total_steps + 1 + 1) * dp_size
assert num_execute_model == expected_exec_model, (
f"Model execution count mismatch. Expected: {expected_exec_model}, Got: {num_execute_model}"
)
# Metric 3: Dummy Runs (Warmup & Alignment)
# vLLM synchronizes globally every 32 steps.
# Ref: vllm.v1.engine.core.DPEngineCoreProc._has_global_unfinished_reqs
aligned_steps = (total_steps + 31) // 32 * 32
# Part A: Warmup runs (Profile run + 2 runs per captured graph)
warmup_runs = 1 + (2 * max_batch_sizes)
soc_version = get_ascend_device_type()
if soc_version in {AscendDeviceType.A3} and "DeepSeek" in model:
# An extra warmup run is needed for MC2 warmup here
warmup_runs += 1
# Part B: Alignment padding (Empty runs to hit the 32-step boundary)
padding_runs = aligned_steps - total_steps
expected_dummy_run = (warmup_runs + padding_runs) * dp_size
assert num_dummy_run == expected_dummy_run, (
f"Dummy run count mismatch. Expected: {expected_dummy_run}, Got: {num_dummy_run}"
)
# Metric 4: Graph Replay (Inference Execution)
# Replays happen for every aligned step across all graphs.
expected_replay = num_acl_graphs * aligned_steps * dp_size
assert actual_replay == expected_replay, f"Replay count mismatch. Expected: {expected_replay}, Got: {actual_replay}"

View File

@@ -0,0 +1,24 @@
import pytest
from tests.e2e.conftest import VllmRunner
from tests.e2e.pull_request.one_card.lora.test_ilama_lora import EXPECTED_LORA_OUTPUT, MODEL_PATH, do_sample
@pytest.mark.parametrize("distributed_executor_backend", ["mp"])
def test_ilama_lora_tp2(distributed_executor_backend, ilama_lora_files):
with VllmRunner(
MODEL_PATH,
enable_lora=True,
max_loras=4,
dtype="half",
max_model_len=1024,
max_num_seqs=16,
tensor_parallel_size=2,
cudagraph_capture_sizes=[1, 2, 4, 8],
distributed_executor_backend=distributed_executor_backend,
enforce_eager=True,
) as vllm_model:
output = do_sample(vllm_model.model, ilama_lora_files, lora_id=2)
for i in range(len(EXPECTED_LORA_OUTPUT)):
assert output[i] == EXPECTED_LORA_OUTPUT[i]

View File

@@ -0,0 +1,30 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
import pytest
from tests.e2e.conftest import VllmRunner, wait_until_npu_memory_free
from tests.e2e.pull_request.one_card.lora.test_llama32_lora import generate_and_test
from vllm_ascend.utils import enable_custom_op
enable_custom_op()
# For hk region, we need to use the model from hf to avoid the network issue
MODEL_PATH = "vllm-ascend/Llama-3.2-3B-Instruct"
@pytest.mark.parametrize("fully_sharded_loras", [False, True])
@wait_until_npu_memory_free()
def test_llama_lora_tp2(llama32_lora_files, fully_sharded_loras):
with VllmRunner(
MODEL_PATH,
enable_lora=True,
# also test odd max_num_seqs
max_num_seqs=7,
max_model_len=1024,
max_loras=4,
tensor_parallel_size=2,
fully_sharded_loras=fully_sharded_loras,
compilation_config={"cudagraph_mode": "PIECEWISE"},
) as vllm_model:
llm = vllm_model.model
generate_and_test(llm, llama32_lora_files)

View File

@@ -0,0 +1,68 @@
import vllm
from vllm.lora.request import LoRARequest
MODEL_PATH = "Qwen/Qwen3-30B-A3B"
PROMPT_TEMPLATE = """<|im_start|>user
I want you to act as a SQL terminal in front of an example database, you need only to return the sql command to me.Below is an instruction that describes a task, Write a response that appropriately completes the request.
"
##Instruction:
candidate_poll contains tables such as candidate, people. Table candidate has columns such as Candidate_ID, People_ID, Poll_Source, Date, Support_rate, Consider_rate, Oppose_rate, Unsure_rate. Candidate_ID is the primary key.
Table people has columns such as People_ID, Sex, Name, Date_of_Birth, Height, Weight. People_ID is the primary key.
The People_ID of candidate is the foreign key of People_ID of people.
###Input:
{context}
###Response:<|im_end|>
<|im_start|>assistant""" # noqa: E501
EXPECTED_LORA_OUTPUT = [
"<think>\n\n</think>\n\nSELECT count(*) FROM candidate",
"<think>\n\n</think>\n\nSELECT count(*) FROM candidate",
"<think>\n\n</think>\n\nSELECT poll_source FROM candidate GROUP BY poll_source ORDER BY count(*) DESC LIMIT 1", # noqa: E501
"<think>\n\n</think>\n\nSELECT poll_source FROM candidate GROUP BY poll_source ORDER BY count(*) DESC LIMIT 1", # noqa: E501
]
def generate_and_test(llm: vllm.LLM, lora_path: str, lora_id: int) -> None:
prompts = [
PROMPT_TEMPLATE.format(context="How many candidates are there?"),
PROMPT_TEMPLATE.format(context="Count the number of candidates."),
PROMPT_TEMPLATE.format(
context="Which poll resource provided the most number of candidate information?" # noqa: E501
),
PROMPT_TEMPLATE.format(context="Return the poll resource associated with the most candidates."),
]
sampling_params = vllm.SamplingParams(temperature=0, max_tokens=64)
outputs = llm.generate(
prompts,
sampling_params,
lora_request=LoRARequest(str(lora_id), lora_id, lora_path) if lora_id else None,
)
# Print the outputs.
generated_texts: list[str] = []
for output in outputs:
prompt = output.prompt
generated_text = output.outputs[0].text.strip()
generated_texts.append(generated_text)
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
for i in range(len(EXPECTED_LORA_OUTPUT)):
assert generated_texts[i].startswith(EXPECTED_LORA_OUTPUT[i])
def test_qwen3moe_lora(qwen3moe_lora_files):
llm = vllm.LLM(
MODEL_PATH,
max_model_len=1024,
enable_lora=True,
max_loras=4,
enforce_eager=True,
trust_remote_code=True,
enable_chunked_prefill=True,
tensor_parallel_size=2,
)
generate_and_test(llm, qwen3moe_lora_files, lora_id=1)

View File

@@ -0,0 +1,429 @@
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
# Run `pytest tests/e2e/pull_request/two_card/spec_decode/test_spec_decode.py`.
from __future__ import annotations
import os
from unittest.mock import patch
import pytest
from transformers import AutoTokenizer
from vllm import SamplingParams
from vllm.config import CompilationConfig
from vllm.tokenizers.registry import resolve_tokenizer_args
from vllm.v1.metrics.reader import Counter, Vector
from tests.e2e.conftest import VllmRunner
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
MODELS = {
"eagle3": {
"main": "Qwen/Qwen3-8B",
"spec": "RedHatAI/Qwen3-8B-speculator.eagle3",
},
}
P_EAGLE_MODELS = {
"p-eagle": {
"main": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
"spec": "amazon/Qwen3-Coder-30B-A3B-Instruct-P-EAGLE",
},
}
VWN_EAGLE3_MODELS = {
"vwn_eagle3": {
"main": "Qwen/Qwen3-30B-A3B",
"spec": "vllm-ascend/Qwen3-30B-A3B-vwn-eagle-model",
},
}
# NOTE: golden may change (eagle_proposer only runs in eager mode currently),
# thus please update it if ci fails but you have better acceptance
BASELINES_SP = {
"eagle3": [0.68, 0.40, 0.18],
"p-eagle": [0.5625, 0.25, 0.0625, 0.0, 0.0, 0.0, 0.0, 0.0],
"vwn_eagle3": [0.75, 0.5, 0.3],
}
@pytest.mark.skip(reason="skip test_eagle3_sp_acceptance")
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
@pytest.mark.parametrize("method", ["eagle3"])
@pytest.mark.parametrize("num_speculative_tokens", [3])
@pytest.mark.parametrize("disable_padded_drafter_batch", [True, False])
@pytest.mark.parametrize("async_scheduling", [True, False])
def test_eagle3_sp_acceptance(
method: str,
num_speculative_tokens: int,
disable_padded_drafter_batch: bool,
async_scheduling: bool,
):
if disable_padded_drafter_batch and async_scheduling:
pytest.skip(
"skip disable_padded_drafter_batch=True and async_scheduling=True",
)
main_model_name = MODELS[method]["main"]
spec_model_name = MODELS[method]["spec"]
tokenizer = AutoTokenizer.from_pretrained(
main_model_name,
trust_remote_code=True,
)
sampling_params = SamplingParams(
temperature=0,
ignore_eos=False,
max_tokens=256,
)
# sp will only be enabled when query_lens > 1000
prompts = [
{
"role": "user",
"content": " " * 1000 + "Hello, my name is",
},
{
"role": "user",
"content": " " * 1000 + "The president of the United States is",
},
{
"role": "user",
"content": " " * 1000 + "The capital of France is",
},
{
"role": "user",
"content": " " * 1000 + "The future of AI is",
},
]
prompts = [
tokenizer.apply_chat_template(
[prompt],
tokenize=False,
add_generation_prompt=True,
)
for prompt in prompts
]
speculative_config = {
"enforce_eager": True,
"method": method,
"num_speculative_tokens": num_speculative_tokens,
"disable_padded_drafter_batch": disable_padded_drafter_batch,
"model": spec_model_name,
}
compilation_config = CompilationConfig(cudagraph_mode="FULL_DECODE_ONLY", cudagraph_capture_sizes=[12])
with VllmRunner(
main_model_name,
enforce_eager=True,
max_model_len=8192,
disable_log_stats=False,
tensor_parallel_size=2,
max_num_seqs=256,
distributed_executor_backend="mp",
gpu_memory_utilization=0.7,
speculative_config=speculative_config,
compilation_config=compilation_config,
async_scheduling=async_scheduling,
) as llm:
_ = llm.generate(prompts, sampling_params)
metrics = llm.model.get_metrics()
num_drafts = 0
num_accepted_tokens_per_pos = [0] * num_speculative_tokens
for metric in metrics:
if metric.name == "vllm:spec_decode_num_drafts":
assert isinstance(metric, Counter)
num_drafts += metric.value
elif metric.name == "vllm:spec_decode_num_accepted_tokens_per_pos":
assert isinstance(metric, Vector)
for pos in range(len(metric.values)):
num_accepted_tokens_per_pos[pos] += metric.values[pos]
acceptance_per_pos = [num_accepted_tokens / num_drafts for num_accepted_tokens in num_accepted_tokens_per_pos]
golden = BASELINES_SP[method]
match = all(abs(a - b) < 0.06 for a, b in zip(acceptance_per_pos, golden))
if not match:
print(f"acceptance_per_pos: {acceptance_per_pos}")
print(f"golden: {golden}")
assert match
def test_qwen3_eagle3_pcp2_tp1():
"""
Test Qwen3-8B with Eagle3 speculative decoding under PCP + TP1 configuration.
This test verifies that eagle3 spec decode works correctly with:
- PCP enabled (prefill_context_parallel_size=2)
- Tensor Parallel size = 1
- num_speculative_tokens = 3
- enforce_eager = True
"""
method = "eagle3"
num_speculative_tokens = 3
main_model_name = MODELS[method]["main"]
spec_model_name = MODELS[method]["spec"]
tokenizer = AutoTokenizer.from_pretrained(
main_model_name,
trust_remote_code=True,
)
sampling_params = SamplingParams(
temperature=0,
ignore_eos=False,
max_tokens=256,
)
prompts = [
{
"role": "user",
"content": "Hello, my name is",
},
{
"role": "user",
"content": "The president of the United States is",
},
{
"role": "user",
"content": "The capital of France is",
},
{
"role": "user",
"content": "The future of AI is",
},
]
prompts = [
tokenizer.apply_chat_template(
[prompt],
tokenize=False,
add_generation_prompt=True,
)
for prompt in prompts
]
speculative_config = {
"method": method,
"num_speculative_tokens": num_speculative_tokens,
"model": spec_model_name,
}
with VllmRunner(
main_model_name,
enforce_eager=True,
max_model_len=2048,
disable_log_stats=False,
tensor_parallel_size=1,
prefill_context_parallel_size=2,
max_num_seqs=256,
distributed_executor_backend="mp",
gpu_memory_utilization=0.7,
speculative_config=speculative_config,
) as llm:
llm.generate(prompts, sampling_params)
@pytest.mark.parametrize("method", P_EAGLE_MODELS.keys())
@pytest.mark.parametrize("num_speculative_tokens", [8])
@pytest.mark.parametrize("draft_tensor_parallel_size", [None, 2])
def test_p_eagle_acceptance(
method: str,
num_speculative_tokens: int,
draft_tensor_parallel_size: None | int,
):
"""
Test acceptance rate for parallel drafting speculative decoding
using a smaller draft model with parallel_drafting enabled.
"""
main_model_name = P_EAGLE_MODELS[method]["main"]
spec_model_name = P_EAGLE_MODELS[method]["spec"]
tokenizer_path = resolve_tokenizer_args(main_model_name)[1]
tokenizer = AutoTokenizer.from_pretrained(
tokenizer_path,
trust_remote_code=True,
)
sampling_params = SamplingParams(
temperature=0,
ignore_eos=False,
max_tokens=256,
)
prompts = [
{
"role": "user",
"content": "Hello, your name is",
},
]
prompts = [
tokenizer.apply_chat_template(
[prompt],
tokenize=False,
add_generation_prompt=True,
)
for prompt in prompts
]
speculative_config = {
"method": "eagle3",
"model": spec_model_name,
"num_speculative_tokens": num_speculative_tokens,
"draft_tensor_parallel_size": draft_tensor_parallel_size,
"parallel_drafting": True,
}
compilation_config = CompilationConfig(cudagraph_capture_sizes=[12])
with VllmRunner(
main_model_name,
max_model_len=4096,
disable_log_stats=False,
tensor_parallel_size=2,
max_num_seqs=256,
distributed_executor_backend="mp",
gpu_memory_utilization=0.8,
speculative_config=speculative_config,
compilation_config=compilation_config,
enable_prefix_caching=False,
) as llm:
outputs = llm.model.generate(prompts, sampling_params)
metrics = llm.model.get_metrics()
for output in outputs:
prompt = output.prompt
generated_text = output.outputs[0].text
output_tokens = output.outputs[0].token_ids
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
print(f"Output tokens: {output_tokens}")
num_drafts = 0
num_accepted_tokens_per_pos = [0] * num_speculative_tokens
for metric in metrics:
if metric.name == "vllm:spec_decode_num_drafts":
assert isinstance(metric, Counter)
num_drafts += metric.value
elif metric.name == "vllm:spec_decode_num_accepted_tokens_per_pos":
assert isinstance(metric, Vector)
for pos in range(len(metric.values)):
num_accepted_tokens_per_pos[pos] += metric.values[pos]
acceptance_per_pos = [num_accepted_tokens / num_drafts for num_accepted_tokens in num_accepted_tokens_per_pos]
golden = BASELINES_SP[method]
match = all(abs(a - b) < 0.1 for a, b in zip(acceptance_per_pos, golden))
if not match:
print(f"acceptance_per_pos: {acceptance_per_pos}")
print(f"golden: {golden}")
assert match
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
def test_qwen3_vwn_eagle3_tp2():
"""
Test Qwen3-30B-A3B with VWN-Eagle3 speculative decoding acceptance rate.
This test verifies that VWN-Eagle3 spec decode works correctly with:
- Tensor Parallel size = 4
- Expert Parallel enabled (for MoE)
- num_speculative_tokens = 3
- enforce_eager = True
- Acceptance rate matches baseline (tolerance 0.06)
"""
num_speculative_tokens = 3
main_model_name = VWN_EAGLE3_MODELS["vwn_eagle3"]["main"]
spec_model_name = VWN_EAGLE3_MODELS["vwn_eagle3"]["spec"]
tokenizer = AutoTokenizer.from_pretrained(
main_model_name,
trust_remote_code=True,
)
sampling_params = SamplingParams(
temperature=0,
ignore_eos=False,
max_tokens=256,
)
prompts = [
{
"role": "user",
"content": "Hello, my name is",
},
{
"role": "user",
"content": "The capital of France is",
},
{
"role": "user",
"content": "The future of AI is",
},
]
prompts = [
tokenizer.apply_chat_template(
[prompt],
tokenize=False,
add_generation_prompt=True,
)
for prompt in prompts
]
speculative_config = {
"method": "eagle3",
"num_speculative_tokens": num_speculative_tokens,
"model": spec_model_name,
}
with VllmRunner(
main_model_name,
enforce_eager=True,
max_model_len=2048,
disable_log_stats=False,
tensor_parallel_size=2,
max_num_seqs=16,
distributed_executor_backend="mp",
gpu_memory_utilization=0.92,
speculative_config=speculative_config,
enable_expert_parallel=True,
) as llm:
_ = llm.generate(prompts, sampling_params)
metrics = llm.model.get_metrics()
# Check acceptance rate
num_drafts = 0
num_accepted_tokens_per_pos = [0] * num_speculative_tokens
for metric in metrics:
if metric.name == "vllm:spec_decode_num_drafts":
assert isinstance(metric, Counter)
num_drafts += metric.value
elif metric.name == "vllm:spec_decode_num_accepted_tokens_per_pos":
assert isinstance(metric, Vector)
for pos in range(len(metric.values)):
num_accepted_tokens_per_pos[pos] += metric.values[pos]
acceptance_per_pos = [n / num_drafts for n in num_accepted_tokens_per_pos]
golden = BASELINES_SP["vwn_eagle3"]
match = all(abs(a - b) < 0.06 for a, b in zip(acceptance_per_pos, golden))
if not match:
print(f"acceptance_per_pos: {acceptance_per_pos}")
print(f"golden: {golden}")
assert match

View File

@@ -0,0 +1,79 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
"""
Compare the outputs of vLLM with and without aclgraph.
Run `pytest tests/e2e/pull_request/two_card/test_data_parallel.py`.
"""
import os
import subprocess
import sys
from pathlib import Path
from unittest.mock import patch
import pytest
from tests.e2e.conftest import wait_until_npu_memory_free
MODELS = ["Qwen/Qwen3-30B-A3B", "vllm-ascend/Qwen3-30B-A3B-W8A8"]
REPO_ROOT = Path(__file__).resolve().parents[4]
DATA_PARALLEL_SCRIPT = REPO_ROOT / "examples" / "offline_data_parallel.py"
@pytest.mark.parametrize("model", MODELS)
@pytest.mark.parametrize("max_tokens", [32])
@patch.dict(os.environ, {"ASCEND_RT_VISIBLE_DEVICES": "0,1"})
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "1024"})
@wait_until_npu_memory_free(target_free_percentage=0.7)
def test_qwen3_inference_dp2(model, max_tokens):
moe_models = ["Qwen/Qwen3-30B-A3B", "vllm-ascend/Qwen3-30B-A3B-W8A8"]
quantization_models = ["vllm-ascend/Qwen3-30B-A3B-W8A8"]
env = os.environ.copy()
cmd = [
sys.executable,
str(DATA_PARALLEL_SCRIPT),
"--model",
model,
"--dp-size",
"2",
"--tp-size",
"1",
"--node-size",
"1",
"--node-rank",
"0",
"--trust-remote-code",
]
if model in moe_models:
cmd.append("--enable-expert-parallel")
if model in quantization_models:
cmd.append("--quantization")
cmd.append("ascend")
print(f"Running subprocess: {' '.join(cmd)}")
proc = subprocess.run(cmd, env=env, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, timeout=600)
output = proc.stdout.decode(errors="ignore")
print(output)
assert "DP rank 0 needs to process" in output
assert "DP rank 1 needs to process" in output
assert "Generated text:" in output
assert proc.returncode == 0

View File

@@ -0,0 +1,39 @@
#
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
from tests.e2e.conftest import VllmRunner
def test_deepseek_multistream_moe_tp2():
example_prompts = [
"Hello, my name is",
]
dtype = "half"
max_tokens = 5
with VllmRunner(
"vllm-ascend/DeepSeek-V3-Pruning",
dtype=dtype,
tensor_parallel_size=2,
cudagraph_capture_sizes=[1, 2, 4, 8],
distributed_executor_backend="mp",
additional_config={
"enable_multistream_moe": True,
"refresh": True,
},
) as vllm_model:
vllm_model.generate_greedy(example_prompts, max_tokens)

View File

@@ -0,0 +1,97 @@
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
import pytest
from vllm.utils.network_utils import get_open_port
from tests.e2e.conftest import DisaggEpdProxy, RemoteEPDServer
from tools.send_mm_request import send_image_request
MODELS = [
"Qwen/Qwen2.5-VL-7B-Instruct",
]
SHARED_STORAGE_PATH = "/dev/shm/epd/storage"
TENSOR_PARALLELS = [1]
@pytest.mark.asyncio
@pytest.mark.parametrize("model", MODELS)
@pytest.mark.parametrize("tp_size", TENSOR_PARALLELS)
async def test_models(model: str, tp_size: int) -> None:
encode_port = get_open_port()
pd_port = get_open_port()
vllm_server_args = [
[
"--port",
str(encode_port),
"--model",
model,
"--gpu-memory-utilization",
"0.01",
"--tensor-parallel-size",
str(tp_size),
"--enforce-eager",
"--no-enable-prefix-caching",
"--max-model-len",
"10000",
"--max-num-batched-tokens",
"10000",
"--max-num-seqs",
"1",
"--ec-transfer-config",
'{"ec_connector_extra_config":{"shared_storage_path":"'
+ SHARED_STORAGE_PATH
+ '"},"ec_connector":"ECExampleConnector","ec_role": "ec_producer"}',
],
[
"--port",
str(pd_port),
"--model",
model,
"--gpu-memory-utilization",
"0.95",
"--tensor-parallel-size",
str(tp_size),
"--enforce-eager",
"--max-model-len",
"10000",
"--max-num-batched-tokens",
"10000",
"--max-num-seqs",
"128",
"--ec-transfer-config",
'{"ec_connector_extra_config":{"shared_storage_path":"'
+ SHARED_STORAGE_PATH
+ '"},"ec_connector":"ECExampleConnector","ec_role": "ec_consumer"}',
],
]
proxy_port = get_open_port()
proxy_args = [
"--host",
"127.0.0.1",
"--port",
str(proxy_port),
"--encode-servers-urls",
f"http://localhost:{encode_port}",
"--decode-servers-urls",
f"http://localhost:{pd_port}",
"--prefill-servers-urls",
"disable",
]
with RemoteEPDServer(vllm_serve_args=vllm_server_args) as _, DisaggEpdProxy(proxy_args=proxy_args) as proxy:
send_image_request(model, proxy)

View File

@@ -0,0 +1,229 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
"""
Compare the outputs of vLLM with and without aclgraph.
Run `pytest tests/e2e/pull_request/two_card/test_external_launcher.py`.
"""
import os
import subprocess
import sys
from pathlib import Path
from unittest.mock import patch
import huggingface_hub
import pytest
import torch_npu
from modelscope import snapshot_download # type: ignore
from tests.e2e.conftest import wait_until_npu_memory_free
MODELS = ["Qwen/Qwen3-0.6B"]
MOE_MODELS = ["Qwen/Qwen3-30B-A3B"]
DEVICE_NAME = torch_npu.npu.get_device_name(0)[:10]
REPO_ROOT = Path(__file__).resolve().parents[4]
EXTERNAL_LAUNCHER_SCRIPT = REPO_ROOT / "examples" / "offline_external_launcher.py"
EXTERNAL_LAUNCHER_TIMEOUT_S = 720
def _decode_output(output):
if output is None:
return ""
if isinstance(output, bytes):
return output.decode(errors="ignore")
return output
def _run_external_launcher(cmd, env):
env = env.copy()
env["PYTHONUNBUFFERED"] = "1"
print(f"Running subprocess: {' '.join(cmd)}")
try:
proc = subprocess.run(
cmd,
env=env,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=EXTERNAL_LAUNCHER_TIMEOUT_S,
)
except subprocess.TimeoutExpired as exc:
print(f"Subprocess timed out after {EXTERNAL_LAUNCHER_TIMEOUT_S} seconds.")
output = _decode_output(exc.output)
if output:
print(output)
else:
print("No subprocess output captured before timeout.")
raise
output = _decode_output(proc.stdout)
print(output)
return proc, output
@pytest.mark.parametrize("model", MODELS)
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "500"})
def test_qwen3_external_launcher(model):
env = os.environ.copy()
# TODO: Change to 2 when ci machine has 4 cards
cmd = [
sys.executable,
str(EXTERNAL_LAUNCHER_SCRIPT),
"--model",
model,
"--tp-size",
"1",
"--node-size",
"1",
"--node-rank",
"0",
"--proc-per-node",
"2",
"--trust-remote-code",
]
proc, output = _run_external_launcher(cmd, env)
assert "TP RANKS: [0]" in output
assert "TP RANKS: [1]" in output
assert "Generated text:" in output
assert proc.returncode == 0
@pytest.mark.parametrize("model", MOE_MODELS)
@wait_until_npu_memory_free(target_free_percentage=0.7)
def test_qwen3_moe_external_launcher_ep_tp2(model):
env = os.environ.copy()
# TODO: Change to 2 when ci machine has 4 cards
cmd = [
sys.executable,
str(EXTERNAL_LAUNCHER_SCRIPT),
"--model",
model,
"--tp-size",
"2",
"--node-size",
"1",
"--node-rank",
"0",
"--proc-per-node",
"2",
"--trust-remote-code",
"--enable-expert-parallel",
]
proc, output = _run_external_launcher(cmd, env)
assert "TP RANKS: [0, 1]" in output
assert "Generated text:" in output
assert proc.returncode == 0
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_NZ": "0"})
@wait_until_npu_memory_free(target_free_percentage=0.7)
def test_qwen3_external_launcher_with_sleepmode():
env = os.environ.copy()
# TODO: Change to 2 when ci machine has 4 cards
cmd = [
sys.executable,
str(EXTERNAL_LAUNCHER_SCRIPT),
"--model",
"Qwen/Qwen3-8B",
"--tp-size",
"1",
"--node-size",
"1",
"--node-rank",
"0",
"--proc-per-node",
"2",
"--trust-remote-code",
"--enable-sleep-mode",
"--temperature",
"0",
"--model-weight-gib",
"16",
]
proc, output = _run_external_launcher(cmd, env)
assert "Generated text:" in output
assert "Sleep and wake up successfully!!" in output
assert proc.returncode == 0
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_NZ": "0"})
@wait_until_npu_memory_free(target_free_percentage=0.7)
def test_qwen3_external_launcher_with_sleepmode_level2():
env = os.environ.copy()
model_path = snapshot_download(
"Qwen/Qwen3-8B",
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
)
# TODO: Add moe model test
cmd = [
sys.executable,
str(EXTERNAL_LAUNCHER_SCRIPT),
"--model",
model_path,
"--tp-size",
"1",
"--node-size",
"1",
"--node-rank",
"0",
"--proc-per-node",
"2",
"--trust-remote-code",
"--enable-sleep-mode",
"--temperature",
"0",
"--model-weight-gib",
"16",
"--sleep-mode-level",
"2",
]
proc, output = _run_external_launcher(cmd, env)
assert "Generated text:" in output
assert "Sleep and wake up successfully!!" in output
assert proc.returncode == 0
@pytest.mark.skipif(
DEVICE_NAME != "Ascend910B",
reason="This test is only for Ascend910B devices.",
)
@pytest.mark.parametrize("model", MODELS)
@wait_until_npu_memory_free(target_free_percentage=0.7)
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "1", "HCCL_BUFFSIZE": "500"})
def test_qwen3_external_launcher_with_matmul_allreduce(model):
env = os.environ.copy()
cmd = [
sys.executable,
str(EXTERNAL_LAUNCHER_SCRIPT),
"--model",
model,
"--trust-remote-code",
]
proc, output = _run_external_launcher(cmd, env)
assert "Generated text:" in output
assert proc.returncode == 0

View File

@@ -0,0 +1,117 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
#
"""Compare the short outputs of HF and vLLM when using greedy sampling.
Run `pytest tests/e2e/pull_request/two_card/test_flashcomm_distributed.py`.
"""
import os
from unittest.mock import patch
import pytest
from vllm import SamplingParams
from vllm.config import KVTransferConfig
from tests.e2e.conftest import VllmRunner
QWEN_DENSE_MODELS = [
"vllm-ascend/Qwen3-0.6B-W8A8",
]
@pytest.mark.skip(reason="test is broken, fix me")
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
@patch.dict(os.environ, {"VLLM_ASCEND_FLASHCOMM2_PARALLEL_SIZE": "1"})
def test_qwen3_moe_fc2_oshard_tp2() -> None:
example_prompts = [
"Hello, my name is",
]
sampling_params = SamplingParams(max_tokens=5, temperature=0.0, top_k=50, top_p=0.9)
with VllmRunner(
"Qwen/Qwen3-30B-A3B",
dtype="auto",
tensor_parallel_size=2,
distributed_executor_backend="mp",
enable_expert_parallel=True,
enforce_eager=True,
additional_config={"layer_sharding": ["o_proj"]},
kv_transfer_config=KVTransferConfig(kv_role="kv_producer"),
) as vllm_model:
vllm_model.generate(example_prompts, sampling_params)
@pytest.mark.skip(reason="test is broken, fix me")
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
def test_deepseek_v2_lite_fc1_tp2() -> None:
example_prompts = [
"test" * 1001,
]
sampling_params = SamplingParams(max_tokens=5, temperature=0.0, top_k=50, top_p=0.9)
with VllmRunner(
"vllm-ascend/DeepSeek-V2-Lite-W8A8",
dtype="auto",
tensor_parallel_size=2,
distributed_executor_backend="mp",
enable_expert_parallel=True,
enforce_eager=True,
quantization="ascend",
) as vllm_model:
vllm_model.generate(example_prompts, sampling_params)
@pytest.mark.parametrize("model", QWEN_DENSE_MODELS)
@pytest.mark.skip(reason="test is broken, fix me")
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
def test_qwen3_dense_fc1_tp2(model):
example_prompts = [
"Hello, my name is",
]
max_tokens = 5
with VllmRunner(
model,
max_model_len=8192,
dtype="auto",
tensor_parallel_size=2,
cudagraph_capture_sizes=[1, 2, 4, 8],
quantization="ascend",
) as vllm_model:
vllm_model.generate_greedy(example_prompts, max_tokens)
@pytest.mark.parametrize("model", QWEN_DENSE_MODELS)
@pytest.mark.skip(reason="test is broken, fix me")
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
def test_qwen3_dense_prefetch_mlp_weight_tp2(model):
example_prompts = [
"Hello, my name is",
]
max_tokens = 5
with VllmRunner(
model,
max_model_len=8192,
dtype="auto",
tensor_parallel_size=2,
cudagraph_capture_sizes=[1, 2, 4, 8],
quantization="ascend",
additional_config={"weight_prefetch_config": {"enabled": True}},
) as vllm_model:
vllm_model.generate_greedy(example_prompts, max_tokens)

View File

@@ -0,0 +1,44 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
#
"""Compare the short outputs of HF and vLLM when using greedy sampling.
Run `pytest tests/e2e/pull_request/two_card/test_gpt_oss_distributed.py`.
"""
import pytest
from tests.e2e.conftest import VllmRunner
GPT_OSS_MODELS = [
"unsloth/gpt-oss-20b-BF16",
]
@pytest.mark.parametrize("model", GPT_OSS_MODELS)
def test_gpt_oss_distributed_tp2(model):
example_prompts = [
"Hello, my name is",
]
max_tokens = 5
with VllmRunner(
model,
tensor_parallel_size=2,
enforce_eager=True,
) as vllm_model:
vllm_model.generate_greedy(example_prompts, max_tokens)

View File

@@ -0,0 +1,342 @@
#
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
"""End-to-end test for the HCCL weight transfer engine.
This test starts a vLLM server with dummy weights and the HCCL weight transfer
backend enabled, then runs the trainer side of an RLHF-style weight sync from a
separate NPU. It exercises the full control plane (HTTP) + data plane (HCCL
packed broadcast + layerwise reload) and asserts the server's weights actually
change after the broadcast.
To keep the test self-contained and download-free, the trainer model is built
from the architecture config with random weights (only the tiny config/tokenizer
are needed, which the server already fetches). The parameter names/shapes/dtypes
match the real checkpoint, so the broadcast pipeline is fully exercised; we just
don't assert "coherent text" since the broadcast weights are random. Set
``WEIGHT_TRANSFER_TEST_MODEL=/path/to/checkpoint`` to instead broadcast real
weights from a local checkpoint.
Topology (requires 2 NPUs):
- NPU 0: vLLM inference worker (rank 1 in the HCCL group)
- NPU 1: trainer / weight source (rank 0 in the HCCL group)
Refer to ``examples/rl/rlhf_http_hccl.py`` for the end-user workflow.
Run with::
pytest tests/e2e/multicard/2-cards/test_weight_transfer_hccl.py
"""
import os
import threading
import pytest
import requests
import torch
import torch_npu # noqa: F401 # registers the NPU backend
from transformers import AutoConfig, AutoModelForCausalLM
from vllm.utils.network_utils import get_ip, get_open_port
from tests.e2e.conftest import RemoteOpenAIServer
MODEL_NAME = "Qwen/Qwen3-0.6B"
# Device 0 hosts the inference worker, device 1 hosts the trainer.
INFERENCE_WORLD_SIZE = 1
TRAINER_DEVICE_INDEX = INFERENCE_WORLD_SIZE
PROMPTS = [
"Hello, my name is",
"The capital of France is",
]
# HTTP timeouts (seconds). Weight broadcast can take a while for large models.
INIT_TIMEOUT = 120
UPDATE_TIMEOUT = 300
CONTROL_TIMEOUT = 60
def _log(message: str) -> None:
"""Flushed log so step markers show up immediately even when stdout is piped."""
print(f"[trainer] {message}", flush=True)
def _build_trainer_model(device_index: int):
"""Build the trainer-side model without downloading the checkpoint weights.
By default the model is instantiated from the architecture config with random
weights (no ``model.safetensors`` download required); only the tiny config is
read, which the server already fetches. Its ``named_parameters`` carry the
same names/shapes/dtypes as the real checkpoint, so the HCCL broadcast +
layerwise reload path is exercised exactly as with real weights.
Set ``WEIGHT_TRANSFER_TEST_MODEL=/path/to/checkpoint`` to broadcast real
weights from a local directory instead.
"""
device = f"npu:{device_index}"
override_path = os.getenv("WEIGHT_TRANSFER_TEST_MODEL")
if override_path:
_log(f"loading real trainer weights from {override_path}")
model = AutoModelForCausalLM.from_pretrained(override_path, dtype=torch.bfloat16)
else:
_log("building trainer model from config with random weights (download-free)")
config = AutoConfig.from_pretrained(MODEL_NAME, trust_remote_code=True)
model = AutoModelForCausalLM.from_config(config)
model = model.to(device=device, dtype=torch.bfloat16)
return model
def _post(server: RemoteOpenAIServer, route: str, *, json=None, timeout=CONTROL_TIMEOUT):
response = requests.post(server.url_for(route), json=json, timeout=timeout)
response.raise_for_status()
return response
class _BackgroundPost(threading.Thread):
"""Run an HTTP POST in a thread while keeping its exception visible.
The trainer side blocks on collective HCCL ops, so the matching server-side
RPC must run concurrently. If that RPC fails, swallowing the exception would
deadlock the trainer forever; instead we record it and surface it on join().
"""
def __init__(self, server: RemoteOpenAIServer, route: str, *, json=None, timeout=CONTROL_TIMEOUT):
super().__init__(daemon=True)
self._server = server
self._route = route
self._json = json
self._timeout = timeout
self.error: BaseException | None = None
def run(self) -> None:
try:
_post(self._server, self._route, json=self._json, timeout=self._timeout)
_log(f"background POST /{self._route} done")
except BaseException as exc: # noqa: BLE001 - re-raised on join via raise_if_failed
self.error = exc
_log(f"background POST /{self._route} FAILED: {exc!r}")
def raise_if_failed(self) -> None:
if self.error is not None:
raise RuntimeError(f"server-side /{self._route} failed") from self.error
def _generate(client, model, prompts):
completions = []
for prompt in prompts:
response = client.completions.create(
model=model,
prompt=prompt,
max_tokens=16,
temperature=0,
)
completions.append(response.choices[0].text)
return completions
def _collect_weight_metadata(train_model):
"""Collect parameter metadata and size the packed buffer for broadcasting."""
names: list[str] = []
dtype_names: list[str] = []
shapes: list[list[int]] = []
max_tensor_bytes = 0
for name, parameter in train_model.named_parameters():
names.append(name)
dtype_names.append(str(parameter.dtype).split(".")[-1])
shapes.append(list(parameter.shape))
tensor_bytes = parameter.numel() * parameter.element_size()
max_tensor_bytes = max(max_tensor_bytes, tensor_bytes)
# Keep the 1 GiB default unless a single tensor needs more (+128 MiB headroom).
packed_buffer_size_bytes = max(max_tensor_bytes + 128 * 2**20, 2**30)
return names, dtype_names, shapes, packed_buffer_size_bytes
def _has_lifecycle_endpoints(server: RemoteOpenAIServer) -> bool:
"""Detect whether the server exposes the vLLM-main start/finish endpoints.
On vLLM main, ``/start_weight_update`` and ``/finish_weight_update`` drive
the layerwise reload lifecycle. On v0.20.2 these endpoints do not exist and
``update_weights`` is self-contained, so a probe returns 404.
"""
try:
response = requests.post(
server.url_for("start_weight_update"),
json={"is_checkpoint_format": True},
timeout=CONTROL_TIMEOUT,
)
except requests.RequestException:
return False
if response.status_code == 404:
return False
response.raise_for_status()
return True
@pytest.mark.skipif(
torch.npu.device_count() < 2,
reason="HCCL weight transfer e2e test requires at least 2 NPUs.",
)
def test_hccl_weight_transfer_updates_server_weights():
port = get_open_port()
server_args = [
"--enforce-eager",
"--load-format",
"dummy",
"--weight-transfer-config",
'{"backend": "nccl"}',
"--tensor-parallel-size",
str(INFERENCE_WORLD_SIZE),
"--max-model-len",
"1024",
"--gpu-memory-utilization",
"0.6",
"--port",
str(port),
"--trust-remote-code",
]
# The dev-mode endpoints (/init_weight_transfer_engine, /update_weights,
# /pause, /resume, ...) are only registered when VLLM_SERVER_DEV_MODE=1.
# Pin the server to NPU 0 so the trainer can own NPU 1 exclusively.
env_dict = {
"VLLM_SERVER_DEV_MODE": "1",
"ASCEND_RT_VISIBLE_DEVICES": "0",
"VLLM_ASCEND_ENABLE_NZ": "0",
}
_log(f"starting server on port {port} (device 0, dummy weights) ...")
with RemoteOpenAIServer(
MODEL_NAME,
vllm_serve_args=server_args,
# Health check, OpenAI client and control-plane requests all target this
# host; use loopback explicitly so they reach the local server directly.
server_host="127.0.0.1",
server_port=port,
env_dict=env_dict,
auto_port=False,
) as server:
client = server.get_client()
# 1) Baseline generation with dummy weights (expected to be nonsense).
_log("generating baseline outputs (dummy weights) ...")
outputs_before = _generate(client, MODEL_NAME, PROMPTS)
_log(f"outputs BEFORE weight update: {outputs_before}")
# 2) Build the trainer model on the trainer NPU (download-free by default).
_log(f"preparing trainer model on npu:{TRAINER_DEVICE_INDEX} ...")
torch.npu.set_device(TRAINER_DEVICE_INDEX)
train_model = _build_trainer_model(TRAINER_DEVICE_INDEX)
_log("trainer model ready")
# Import after the server is up so the HCCL engine plugin is registered.
from vllm_ascend.distributed.weight_transfer.hccl_engine import (
HCCLTrainerSendWeightsArgs,
HCCLWeightTransferEngine,
)
master_address = get_ip()
master_port = get_open_port()
rank_offset = 1
world_size = INFERENCE_WORLD_SIZE + 1 # workers + trainer
# 3) Build the HCCL process group on both sides. The server side blocks
# until the trainer connects, so kick it off in a background thread.
init_info = dict(
master_address=master_address,
master_port=master_port,
rank_offset=rank_offset,
world_size=world_size,
)
_log(f"HCCL rendezvous at {master_address}:{master_port} (world_size={world_size}) ...")
init_thread = _BackgroundPost(
server,
"init_weight_transfer_engine",
json={"init_info": init_info},
timeout=INIT_TIMEOUT,
)
init_thread.start()
model_update_group = HCCLWeightTransferEngine.trainer_init(
dict(
master_address=master_address,
master_port=master_port,
world_size=world_size,
),
)
_log("trainer_init returned, waiting for server init RPC ...")
init_thread.join()
init_thread.raise_if_failed()
_log("HCCL process group established")
# 4) Pause generation and start the weight update lifecycle. On vLLM
# main this probe also performs the actual /start_weight_update call,
# so we must not call it again below.
_post(server, "pause")
use_lifecycle = _has_lifecycle_endpoints(server)
_log(f"paused; lifecycle endpoints available: {use_lifecycle}")
names, dtype_names, shapes, packed_buffer_size_bytes = _collect_weight_metadata(train_model)
update_info = dict(
names=names,
dtype_names=dtype_names,
shapes=shapes,
packed=True,
packed_buffer_size_bytes=packed_buffer_size_bytes,
)
if not use_lifecycle:
# v0.20.2 folds the layerwise reload lifecycle into update_weights.
update_info["is_checkpoint_format"] = True
# update_weights blocks on the server while it waits for HCCL broadcasts,
# so run it in a thread while the trainer produces the data.
_log(f"broadcasting {len(names)} tensors via HCCL (packed) ...")
update_thread = _BackgroundPost(
server,
"update_weights",
json={"update_info": update_info},
timeout=UPDATE_TIMEOUT,
)
update_thread.start()
trainer_args = HCCLTrainerSendWeightsArgs(
group=model_update_group,
packed=True,
packed_buffer_size_bytes=packed_buffer_size_bytes,
)
HCCLWeightTransferEngine.trainer_send_weights(
iterator=train_model.named_parameters(),
trainer_args=trainer_args,
)
_log("trainer finished sending weights, waiting for server update RPC ...")
update_thread.join()
update_thread.raise_if_failed()
_log("weight broadcast complete")
# 5) Finalize the lifecycle and resume generation.
if use_lifecycle:
_post(server, "finish_weight_update")
_post(server, "resume")
# 6) Generation after the broadcast weights are loaded.
outputs_after = _generate(client, MODEL_NAME, PROMPTS)
_log(f"outputs AFTER weight update: {outputs_after}")
# Reaching here means the full HCCL transfer pipeline succeeded: every
# control-plane RPC raised on a non-2xx response and each background POST
# re-raised on join(). The broadcast weights differ from the server's dummy
# init, so the served model must now produce different generations.
assert outputs_after != outputs_before, "server weights did not change after HCCL transfer"

View File

@@ -0,0 +1,38 @@
import os
from unittest.mock import patch
import pytest
from vllm import SamplingParams
from vllm.sampling_params import RequestOutputKind
from tests.e2e.conftest import VllmRunner
MODELS = [
"Qwen/Qwen3.5-35B-A3B",
"Qwen/Qwen3-30B-A3B",
]
@pytest.mark.parametrize("model", MODELS)
@patch.dict(os.environ, {"OMP_NUM_THREADS": "1"})
def test_qwen3_moe_routing_replay(model):
prompts = [
"Hello, please introduce yourself.",
]
with VllmRunner(
model,
tensor_parallel_size=2,
enable_expert_parallel=True,
cudagraph_capture_sizes=[1, 2, 4, 8],
distributed_executor_backend="mp",
enable_return_routed_experts=True,
async_scheduling=False,
) as vllm_model:
sampling_params = SamplingParams(
max_tokens=5, temperature=0.8, top_p=0.95, output_kind=RequestOutputKind.FINAL_ONLY
)
inputs = vllm_model.get_inputs(prompts=prompts)
outputs = vllm_model.model.generate(prompts=inputs, sampling_params=sampling_params)
assert outputs[0].finished
assert len(outputs[0].outputs[0].text) > 0
assert outputs[0].outputs[0].routed_experts.size > 0

View File

@@ -0,0 +1,77 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
"""
Run `pytest tests/e2e/pull_request/two_card/test_offline_weight_load.py`.
"""
import os
import subprocess
import sys
from pathlib import Path
from unittest.mock import patch
import pytest
from tests.e2e.conftest import wait_until_npu_memory_free
MODELS = ["Qwen/Qwen3-30B-A3B"]
REPO_ROOT = Path(__file__).resolve().parents[4]
EXTERNAL_LAUNCHER_SCRIPT = REPO_ROOT / "examples" / "offline_external_launcher.py"
@pytest.mark.skip("fix me, unstable, timeout")
@pytest.mark.parametrize("model", MODELS)
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_NZ": "0"})
@wait_until_npu_memory_free(0.7)
def test_qwen3_offline_load_and_sleepmode_tp2(model):
env = os.environ.copy()
cmd = [
sys.executable,
str(EXTERNAL_LAUNCHER_SCRIPT),
"--model",
model,
"--tp-size",
"2",
"--node-size",
"1",
"--node-rank",
"0",
"--proc-per-node",
"2",
"--trust-remote-code",
"--enable-sleep-mode",
"--temperature",
"0",
"--model-weight-gib",
"0.8",
]
print(f"Running subprocess: {' '.join(cmd)}")
proc = subprocess.run(
cmd,
env=env,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=600,
)
output = proc.stdout.decode(errors="ignore")
print(output)
assert "Generated text:" in output
assert "Sleep and wake up successfully!!" in output
assert proc.returncode == 0

View File

@@ -0,0 +1,90 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
"""Compare the with and without prefix caching."""
import pytest
from tests.e2e.conftest import VllmRunner
from tests.e2e.model_utils import check_outputs_equal
MODELS = [
# for MHA
"Qwen/Qwen3-8B",
# for MLA
"deepseek-ai/DeepSeek-V2-Lite-Chat",
]
# A prompt containing a large markdown table. The table is randomly generated by GPT-4.
# ruff: noqa: E501
LONG_PROMPT = (
"You are a helpful assistant in recognizes the content of tables in markdown format. Here is a table as follows.\n# Table\n"
+ """
| ID | Name | Age | Occupation | Country | Email | Phone Number | Address |
|-----|---------------|-----|---------------|---------------|------------------------|----------------|------------------------------|
| 1 | John Doe | 29 | Engineer | USA | john.doe@example.com | 555-1234 | 123 Elm St, Springfield, IL |
| 2 | Jane Smith | 34 | Doctor | Canada | jane.smith@example.com | 555-5678 | 456 Oak St, Toronto, ON |
| 3 | Alice Johnson | 27 | Teacher | UK | alice.j@example.com | 555-8765 | 789 Pine St, London, UK |
| 4 | Bob Brown | 45 | Artist | Australia | bob.b@example.com | 555-4321 | 321 Maple St, Sydney, NSW |
| 5 | Carol White | 31 | Scientist | New Zealand | carol.w@example.com | 555-6789 | 654 Birch St, Wellington, NZ |
| 6 | Dave Green | 28 | Lawyer | Ireland | dave.g@example.com | 555-3456 | 987 Cedar St, Dublin, IE |
| 7 | Emma Black | 40 | Musician | USA | emma.b@example.com | 555-1111 | 246 Ash St, New York, NY |
| 8 | Frank Blue | 37 | Chef | Canada | frank.b@example.com | 555-2222 | 135 Spruce St, Vancouver, BC |
| 9 | Grace Yellow | 50 | Engineer | UK | grace.y@example.com | 555-3333 | 864 Fir St, Manchester, UK |
| 10 | Henry Violet | 32 | Artist | Australia | henry.v@example.com | 555-4444 | 753 Willow St, Melbourne, VIC|
| 11 | Irene Orange | 26 | Scientist | New Zealand | irene.o@example.com | 555-5555 | 912 Poplar St, Auckland, NZ |
| 12 | Jack Indigo | 38 | Teacher | Ireland | jack.i@example.com | 555-6666 | 159 Elm St, Cork, IE |
| 13 | Karen Red | 41 | Lawyer | USA | karen.r@example.com | 555-7777 | 357 Cedar St, Boston, MA |
| 14 | Leo Brown | 30 | Chef | Canada | leo.b@example.com | 555-8888 | 246 Oak St, Calgary, AB |
| 15 | Mia Green | 33 | Musician | UK | mia.g@example.com | 555-9999 | 975 Pine St, Edinburgh, UK |
| 16 | Noah Yellow | 29 | Doctor | Australia | noah.y@example.com | 555-0000 | 864 Birch St, Brisbane, QLD |
| 17 | Olivia Blue | 35 | Engineer | New Zealand | olivia.b@example.com | 555-1212 | 753 Maple St, Hamilton, NZ |
| 18 | Peter Black | 42 | Artist | Ireland | peter.b@example.com | 555-3434 | 912 Fir St, Limerick, IE |
| 19 | Quinn White | 28 | Scientist | USA | quinn.w@example.com | 555-5656 | 159 Willow St, Seattle, WA |
| 20 | Rachel Red | 31 | Teacher | Canada | rachel.r@example.com | 555-7878 | 357 Poplar St, Ottawa, ON |
| 21 | Steve Green | 44 | Lawyer | UK | steve.g@example.com | 555-9090 | 753 Elm St, Birmingham, UK |
| 22 | Tina Blue | 36 | Musician | Australia | tina.b@example.com | 555-1213 | 864 Cedar St, Perth, WA |
| 23 | Umar Black | 39 | Chef | New Zealand | umar.b@example.com | 555-3435 | 975 Spruce St, Christchurch, NZ|
| 24 | Victor Yellow | 43 | Engineer | Ireland | victor.y@example.com | 555-5657 | 246 Willow St, Galway, IE |
| 25 | Wendy Orange | 27 | Artist | USA | wendy.o@example.com | 555-7879 | 135 Elm St, Denver, CO |
| 26 | Xavier Green | 34 | Scientist | Canada | xavier.g@example.com | 555-9091 | 357 Oak St, Montreal, QC |
| 27 | Yara Red | 41 | Teacher | UK | yara.r@example.com | 555-1214 | 975 Pine St, Leeds, UK |
| 28 | Zack Blue | 30 | Lawyer | Australia | zack.b@example.com | 555-3436 | 135 Birch St, Adelaide, SA |
| 29 | Amy White | 33 | Musician | New Zealand | amy.w@example.com | 555-5658 | 159 Maple St, Wellington, NZ |
| 30 | Ben Black | 38 | Chef | Ireland | ben.b@example.com | 555-7870 | 246 Fir St, Waterford, IE |
"""
)
INPUT_PROMPTS = [
LONG_PROMPT + "Question: what is the age of John Doe? Your answer: The age of John Doe is ",
LONG_PROMPT + "Question: what is the age of Zack Blue? Your answer: The age of Zack Blue is ",
]
@pytest.mark.parametrize("model", MODELS)
@pytest.mark.parametrize("max_tokens", [50])
def test_models_prefix_cache_tp2(model: str, max_tokens: int) -> None:
with VllmRunner(
model,
max_model_len=2048,
tensor_parallel_size=2,
cudagraph_capture_sizes=[1, 2, 4, 8],
gpu_memory_utilization=0.7,
) as vllm_model:
prefix_cache_output = vllm_model.generate_greedy(INPUT_PROMPTS, max_tokens)
with VllmRunner(
model,
enable_prefix_caching=False,
max_model_len=2048,
tensor_parallel_size=2,
cudagraph_capture_sizes=[1, 2, 4, 8],
gpu_memory_utilization=0.7,
) as vllm_model:
vllm_output = vllm_model.generate_greedy(INPUT_PROMPTS, max_tokens)
check_outputs_equal(
outputs_0_lst=vllm_output,
outputs_1_lst=prefix_cache_output,
name_0="vllm_output",
name_1="prefix_cache_output",
)

View File

@@ -0,0 +1,82 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
import json
import pytest
import requests
from vllm.utils.network_utils import get_open_port
from tests.e2e.conftest import RemoteOpenAIServer, wait_until_npu_memory_free
from vllm_ascend.utils import vllm_version_is
pytestmark = pytest.mark.skipif(
not vllm_version_is("0.23.0"),
reason="broken on main, fix me.",
)
@wait_until_npu_memory_free()
def test_moe_tp_ep_eplb_full_decode_only():
"""Verify MoE serving with TP, EP, EPLB, and full decode only."""
model = "Qwen/Qwen3-30B-A3B"
port = get_open_port()
env_dict = {
"DYNAMIC_EPLB": "true",
"HCCL_BUFFSIZE": "1024",
}
server_args = [
"--max_model_len",
"8192",
"--tensor_parallel_size",
"2",
"--enable_expert_parallel",
"--port",
str(port),
"--compilation-config",
json.dumps({"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [8]}),
"--additional-config",
json.dumps(
{
"eplb_config": {
"dynamic_eplb": True,
"expert_heat_collection_interval": 100,
"algorithm_execution_interval": 20,
"num_redundant_experts": 2,
}
}
),
]
with RemoteOpenAIServer(model, server_args, server_port=port, auto_port=False, env_dict=env_dict) as server:
response = requests.post(
server.url_for("v1", "completions"),
json={
"model": model,
"prompt": "What is deeplearning?",
"max_tokens": 400,
"temperature": 0.0,
"top_p": 1.0,
"n": 1,
},
timeout=600,
)
response.raise_for_status()
output = response.json()
assert output["choices"][0]["text"]

View File

@@ -0,0 +1,44 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
import os
from unittest.mock import patch
from tests.e2e.conftest import VllmRunner, wait_until_npu_memory_free
EXAMPLE_PROMPTS = [
"Hello, my name is",
]
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "1024"})
@wait_until_npu_memory_free()
def test_qwen3_5_35b_a3b_w8a8_tp2_without_ep():
with VllmRunner(
"Eco-Tech/Qwen3.5-35B-A3B-w8a8-mtp",
max_model_len=4096,
tensor_parallel_size=2,
enable_expert_parallel=False,
quantization="ascend",
gpu_memory_utilization=0.9,
distributed_executor_backend="mp",
cudagraph_capture_sizes=[1, 2, 4, 8],
) as vllm_model:
outputs = vllm_model.generate_greedy(EXAMPLE_PROMPTS, max_tokens=5)
assert outputs[0][1]

View File

@@ -0,0 +1,103 @@
#
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
from unittest.mock import patch
from vllm.assets.image import ImageAsset
from tests.e2e.conftest import VllmRunner, qwen_prompt, wait_until_npu_memory_free
MODEL = "Qwen/Qwen3.6-27B"
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "1024"})
@wait_until_npu_memory_free()
def test_qwen3_6_27b_multimodel_fia_eager():
"""Verify multimodal generation with FIA op and eager mode."""
image = ImageAsset("cherry_blossom").pil_image.convert("RGB")
questions = [
"What is the content of this image?",
"Describe the content of this image in detail.",
"What's in the image?",
"Where is this image taken?",
]
images = [image] * len(questions)
prompts = qwen_prompt(questions)
with VllmRunner(
MODEL,
max_model_len=4096,
tensor_parallel_size=2,
language_model_only=False,
gpu_memory_utilization=0.9,
limit_mm_per_prompt={"image": 1},
mm_processor_kwargs={
"min_pixels": 28 * 28,
"max_pixels": 1280 * 28 * 28,
"fps": 1,
},
enforce_eager=True,
) as vllm_model:
outputs = vllm_model.generate_greedy(
prompts=prompts,
images=images,
max_tokens=64,
)
assert outputs[0][1]
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "1024"})
@wait_until_npu_memory_free()
def test_qwen3_6_27b_multimodel_fia_acl_graph():
"""Verify multimodal generation with FIA op and FULL_AND_PIECEWISE graph mode."""
image = ImageAsset("cherry_blossom").pil_image.convert("RGB")
questions = [
"What is the content of this image?",
"Describe the content of this image in detail.",
"What's in the image?",
"Where is this image taken?",
]
images = [image] * len(questions)
prompts = qwen_prompt(questions)
with VllmRunner(
MODEL,
max_model_len=4096,
tensor_parallel_size=2,
language_model_only=False,
gpu_memory_utilization=0.9,
limit_mm_per_prompt={"image": 1},
mm_processor_kwargs={
"min_pixels": 28 * 28,
"max_pixels": 1280 * 28 * 28,
"fps": 1,
},
compilation_config={
"cudagraph_mm_encoder": True,
"cudagraph_capture_sizes": [1],
"encoder_cudagraph_token_budgets": [128, 256, 512, 1024, 1536, 2048, 2560, 3072, 3584, 4096],
},
) as vllm_model:
outputs = vllm_model.generate_greedy(
prompts=prompts,
images=images,
max_tokens=64,
)
assert outputs[0][1]

View File

@@ -0,0 +1,77 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
import json
import openai
import pytest
from vllm.utils.network_utils import get_open_port
from tests.e2e.conftest import RemoteOpenAIServer
from vllm_ascend.utils import vllm_version_is
pytestmark = pytest.mark.skipif(
not vllm_version_is("0.23.0"),
reason="broken on main, fix me.",
)
@pytest.mark.asyncio
async def test_qwen3_moe_w8a8_distributed_tp2_ep_dynamic_eplb():
model = "vllm-ascend/Qwen3-30B-A3B-W8A8"
port = get_open_port()
compilation_config = json.dumps({"cudagraph_capture_sizes": [8]})
server_args = [
"--max_model_len",
"8192",
"--tensor_parallel_size",
"2",
"--enable_expert_parallel",
"--quantization",
"ascend",
"--port",
str(port),
"--compilation-config",
compilation_config,
]
env_dict = {"HCCL_BUFFSIZE": "1024"}
with RemoteOpenAIServer(model, server_args, server_port=port, auto_port=False, env_dict=env_dict) as server:
client = server.get_async_client()
batch = await client.completions.create(
model=model, prompt="What is deeplearning?", max_tokens=400, temperature=0, top_p=1.0, n=1
)
gt_choices: list[openai.types.CompletionChoice] = batch.choices
env_dict.update({"DYNAMIC_EPLB": "true"})
additional_config = {
"eplb_config": {
"dynamic_eplb": True,
"expert_heat_collection_interval": 100,
"algorithm_execution_interval": 20,
"num_redundant_experts": 2,
"eplb_policy_type": 2,
}
}
server_args.extend(["--additional-config", json.dumps(additional_config)])
with RemoteOpenAIServer(model, server_args, server_port=port, auto_port=False, env_dict=env_dict) as server:
client = server.get_async_client()
batch = await client.completions.create(
model=model, prompt="What is deeplearning?", max_tokens=400, temperature=0, top_p=1.0, n=1
)
eplb_choices: list[openai.types.CompletionChoice] = batch.choices
assert gt_choices[0].text == eplb_choices[0].text, f"{gt_choices[0].text=} \n {eplb_choices[0].text=}"

View File

@@ -0,0 +1,97 @@
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
from typing import Any
import openai
import pytest
from vllm.utils.network_utils import get_open_port
from tests.e2e.conftest import RemoteOpenAIServer
from tools.vllm_bench import run_vllm_bench_case
MODELS = [
"Qwen/Qwen3-8B",
]
prompts = [
"San Francisco is a",
]
api_keyword_args = {
"max_tokens": 10,
}
vllm_bench_cases = {
"dataset-name": "random",
"num_prompts": 500,
"request_rate": 20,
"random_input_len": 128,
"max_concurrency": 40,
"random_output_len": 100,
"temperature": 0.0,
}
# NOTE: Any changes for the baseline throughput should be approved by team members.
# The origin baseline: 1600.0. For some uncertain reasons, the throughput is decreased to 1514.0
baseline_throughput = 1514.0 # baseline throughput for Qwen3-8B, measured with num_prompts=500
@pytest.mark.skip(reason="Temporarily skipped due to flaky failures, pending investigation.")
@pytest.mark.parametrize("model", MODELS)
@pytest.mark.asyncio
async def test_models(model: str) -> None:
port = get_open_port()
env_dict = {
"TASK_QUEUE_ENABLE": "1",
"HCCL_OP_EXPANSION_MODE": "AIV",
}
server_args = [
"--distributed-executor-backend",
"mp",
"--tensor-parallel-size",
"1",
"--port",
str(port),
"--max-model-len",
"5500",
"--max-num-batched-tokens",
"40960",
"--compilation-config",
'{"cudagraph_mode": "FULL_DECODE_ONLY"}',
"--additional-config",
'{"pa_shape_list":[48,64,72,80],"weight_prefetch_config":{"enabled":true}}',
"--block-size",
"128",
"--trust-remote-code",
"--gpu-memory-utilization",
"0.9",
]
request_keyword_args: dict[str, Any] = {
**api_keyword_args,
}
with RemoteOpenAIServer(model, server_args, server_port=port, env_dict=env_dict, auto_port=False) as server:
client = server.get_async_client()
batch = await client.completions.create(
model=model,
prompt=prompts,
**request_keyword_args,
)
choices: list[openai.types.CompletionChoice] = batch.choices
assert choices[0].text, "empty response"
# vllm bench test
run_vllm_bench_case(model, port, vllm_bench_cases, baseline_throughput)

View File

@@ -0,0 +1,66 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
from vllm.assets.image import ImageAsset
from tests.e2e.conftest import VllmRunner, qwen_prompt, wait_until_npu_memory_free
@wait_until_npu_memory_free()
def test_multimodal_reasoning_pp_full_decode_only():
"""Verify multimodal generation with PP and full decode only."""
image = ImageAsset("cherry_blossom").pil_image.convert("RGB")
img_questions = [
"What is the content of this image?",
"Describe the content of this image in detail.",
"What's in the image?",
"Where is this image taken?",
]
images = [image] * len(img_questions)
prompts = qwen_prompt(img_questions)
with VllmRunner(
"Qwen/Qwen3-VL-30B-A3B-Instruct",
pipeline_parallel_size=2,
max_model_len=4096,
max_num_batched_tokens=1024,
gpu_memory_utilization=0.9,
limit_mm_per_prompt={"image": 1},
mm_processor_kwargs={
"min_pixels": 28 * 28,
"max_pixels": 1280 * 28 * 28,
"fps": 1,
},
hf_overrides={"text_config": {"architectures": ["Qwen3MoeForCausalLM"]}},
compilation_config={
"cudagraph_mode": "FULL_DECODE_ONLY",
"cudagraph_capture_sizes": [1, 2, 4, 8],
},
) as vllm_model:
outputs = vllm_model.generate_greedy(
prompts=prompts,
images=images,
max_tokens=64,
)
assert len(outputs) == len(prompts)
for _, output_str in outputs:
assert output_str, "Generated output should not be empty."

View File

@@ -0,0 +1,473 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Two-card e2e tests for SequenceParallelismMoePass patterns:
# - MiddleLayerAllgatherAddRMSNormPattern (all_gather + slice + RMSNorm)
# - Qwen3VLMiddleLayerAllgatherAddRMSNormPattern (all_gather + slice + add + RMSNorm)
# - AllGatherChunkNoOpPattern (all_gather + sequence_parallel_chunk_impl -> identity)
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import queue
import traceback
from collections.abc import Callable, Generator
from dataclasses import dataclass
from typing import Any
import pytest
import torch
import torch.nn as nn
import vllm.config
from vllm.compilation.passes.fx_utils import OpOverload
from vllm.config import ModelConfig, VllmConfig
from vllm.distributed import (
get_tensor_model_parallel_world_size,
get_tp_group,
init_distributed_environment,
tensor_model_parallel_all_gather,
)
from vllm.distributed.parallel_state import (
destroy_distributed_environment,
destroy_model_parallel,
initialize_model_parallel,
)
from vllm.utils.system_utils import update_environment_variables
import vllm_ascend.ops.register_custom_ops # noqa
from tests.e2e.pull_request.one_card.compile.backend import TestBackend as CompileTestBackend
from vllm_ascend.compilation.passes.sequence_parallelism_moe import (
SequenceParallelismMoePass,
)
from vllm_ascend.utils import enable_custom_op
MASTER_PORT = 29500
WORLD_SIZE = 2
WORKER_READY = "__ready__"
WORKER_STOP = "__stop__"
WORKER_RESULT_TIMEOUT_S = 180
WORKER_JOIN_TIMEOUT_S = 30
class BaseAllGatherRMSNormModel(nn.Module):
def __init__(
self,
hidden_size: int,
dtype: torch.dtype,
eps: float = 1e-6,
device: str = "npu",
):
super().__init__()
self.eps = eps
self.norm_w = torch.randn(hidden_size, dtype=dtype, device=device)
def _all_gather_sliced(self, x: torch.Tensor, num_tokens_helper: torch.Tensor) -> torch.Tensor:
num_tokens = num_tokens_helper.shape[0]
activated = torch.relu(x)
gathered = tensor_model_parallel_all_gather(activated, 0)
return gathered[:num_tokens]
@staticmethod
def ops_in_model_after() -> tuple[tuple[OpOverload, int], ...]:
return (
(torch.ops.vllm.all_gather.default, 1),
(torch.ops._C_ascend.npu_add_rms_norm_bias.default, 1),
(torch.ops.vllm.maybe_chunk_residual.default, 1),
)
class AllGatherRMSNormModel(BaseAllGatherRMSNormModel):
def forward(
self,
x: torch.Tensor,
residual: torch.Tensor,
num_tokens_helper: torch.Tensor,
) -> torch.Tensor:
sliced = self._all_gather_sliced(x, num_tokens_helper)
rms_out = torch.ops._C_ascend.npu_add_rms_norm_bias(sliced, residual, self.norm_w, None, self.eps)
return rms_out[0]
@staticmethod
def ops_in_model_before() -> tuple[tuple[OpOverload, int], ...]:
return (
(torch.ops.vllm.all_gather.default, 1),
(torch.ops._C_ascend.npu_add_rms_norm_bias.default, 1),
)
class Qwen3VLAllGatherRMSNormModel(BaseAllGatherRMSNormModel):
"""Exercises Qwen3VLMiddleLayerAllgatherAddRMSNormPattern (all_gather + slice + add + RMSNorm)."""
def forward(
self,
x: torch.Tensor,
residual: torch.Tensor,
num_tokens_helper: torch.Tensor,
deepstack_input_embeds: torch.Tensor,
) -> torch.Tensor:
sliced = self._all_gather_sliced(x, num_tokens_helper)
add_ = sliced + deepstack_input_embeds
result, _, residual = torch.ops._C_ascend.npu_add_rms_norm_bias(add_, residual, self.norm_w, None, self.eps)
# Keep the residual output live so the traced graph preserves the full pattern.
result = result - residual
return result
@staticmethod
def ops_in_model_before() -> tuple[tuple[OpOverload, int], ...]:
return (
(torch.ops.vllm.all_gather.default, 1),
(torch.ops.aten.add.Tensor, 1),
(torch.ops._C_ascend.npu_add_rms_norm_bias.default, 1),
)
class AllGatherChunkNoOpModel(nn.Module):
"""Exercises AllGatherChunkNoOpPattern (all_gather + sequence_parallel_chunk_impl -> identity)."""
def __init__(self) -> None:
super().__init__()
def forward(self, x: torch.Tensor) -> torch.Tensor:
z = torch.relu(x)
gathered = tensor_model_parallel_all_gather(z, 0)
return torch.ops.vllm.sequence_parallel_chunk_impl(gathered)
@staticmethod
def ops_in_model_before() -> tuple[tuple[OpOverload, int], ...]:
return (
(torch.ops.vllm.all_gather.default, 1),
(torch.ops.vllm.sequence_parallel_chunk_impl.default, 1),
)
@staticmethod
def ops_in_model_after() -> tuple[tuple[OpOverload, int], ...]:
return (
(torch.ops.vllm.all_gather.default, 0),
(torch.ops.vllm.sequence_parallel_chunk_impl.default, 0),
)
def _build_all_gather_rms_norm_inputs(
batch_size: int,
seq_len: int,
hidden_size: int,
dtype: torch.dtype,
tp_size: int,
) -> tuple[torch.Tensor, ...]:
local_tokens = batch_size * seq_len
num_tokens = local_tokens * tp_size
x = torch.randn(local_tokens, hidden_size, dtype=dtype)
residual = torch.zeros(num_tokens, hidden_size, dtype=dtype)
num_tokens_helper = torch.empty(num_tokens, device=x.device, dtype=dtype)
return (x, residual, num_tokens_helper)
def _build_qwen3vl_inputs(
batch_size: int,
seq_len: int,
hidden_size: int,
dtype: torch.dtype,
tp_size: int,
) -> tuple[torch.Tensor, ...]:
x, residual, num_tokens_helper = _build_all_gather_rms_norm_inputs(
batch_size=batch_size,
seq_len=seq_len,
hidden_size=hidden_size,
dtype=dtype,
tp_size=tp_size,
)
deepstack = torch.randn(num_tokens_helper.shape[0], hidden_size, dtype=dtype)
return (x, residual, num_tokens_helper, deepstack)
def _build_allgather_chunk_noop_inputs(
batch_size: int,
seq_len: int,
hidden_size: int,
dtype: torch.dtype,
tp_size: int,
) -> tuple[torch.Tensor, ...]:
del tp_size
local_tokens = batch_size * seq_len
x = torch.randn(local_tokens, hidden_size, dtype=dtype)
return (x,)
def _create_all_gather_rms_norm_model(
hidden_size: int,
dtype: torch.dtype,
eps: float,
device: str,
) -> nn.Module:
return AllGatherRMSNormModel(hidden_size=hidden_size, dtype=dtype, eps=eps, device=device)
def _create_qwen3vl_model(
hidden_size: int,
dtype: torch.dtype,
eps: float,
device: str,
) -> nn.Module:
return Qwen3VLAllGatherRMSNormModel(hidden_size=hidden_size, dtype=dtype, eps=eps, device=device)
def _create_allgather_chunk_noop_model(
hidden_size: int,
dtype: torch.dtype,
eps: float,
device: str,
) -> nn.Module:
del hidden_size, dtype, eps, device
return AllGatherChunkNoOpModel()
@dataclass(frozen=True)
class PatternTestCase:
model_factory: Any
input_builder: Any
dynamic_input_indices: tuple[int, ...]
pre_pass_expected_counts_factory: Any
post_pass_expected_counts_factory: Any
PATTERN_TEST_CASES = {
"middle_layer_allgather_add_rms_norm": PatternTestCase(
model_factory=_create_all_gather_rms_norm_model,
input_builder=_build_all_gather_rms_norm_inputs,
dynamic_input_indices=(0, 2),
pre_pass_expected_counts_factory=AllGatherRMSNormModel.ops_in_model_before,
post_pass_expected_counts_factory=AllGatherRMSNormModel.ops_in_model_after,
),
"qwen3vl_middle_layer_allgather_add_rms_norm": PatternTestCase(
model_factory=_create_qwen3vl_model,
input_builder=_build_qwen3vl_inputs,
dynamic_input_indices=(0, 2, 3),
pre_pass_expected_counts_factory=Qwen3VLAllGatherRMSNormModel.ops_in_model_before,
post_pass_expected_counts_factory=Qwen3VLAllGatherRMSNormModel.ops_in_model_after,
),
"allgather_chunk_noop": PatternTestCase(
model_factory=_create_allgather_chunk_noop_model,
input_builder=_build_allgather_chunk_noop_inputs,
dynamic_input_indices=(0,),
pre_pass_expected_counts_factory=AllGatherChunkNoOpModel.ops_in_model_before,
post_pass_expected_counts_factory=AllGatherChunkNoOpModel.ops_in_model_after,
),
}
def _assert_op_counts(
backend: CompileTestBackend,
expected_counts: tuple[tuple[OpOverload, int], ...],
before: bool = False,
) -> None:
for op, expected_count in expected_counts:
actual_count = backend.op_count(op, before=before)
stage = "before" if before else "after"
assert actual_count == expected_count, (
f"op {stage} pass: {op} expected {expected_count}, but got {actual_count}"
)
def _run_single_pattern_case(
local_rank: int,
case_name: str,
vllm_config: VllmConfig,
tp_size: int,
batch_size: int = 8,
seq_len: int = 16,
hidden_size: int = 16,
dtype: torch.dtype = torch.bfloat16,
eps: float = 1e-5,
) -> None:
case = PATTERN_TEST_CASES[case_name]
sp_moe_pass = SequenceParallelismMoePass(vllm_config)
backend = CompileTestBackend(custom_passes=[sp_moe_pass])
model = case.model_factory(
hidden_size=hidden_size,
dtype=dtype,
eps=eps,
device=f"npu:{local_rank}",
)
inputs = case.input_builder(
batch_size=batch_size,
seq_len=seq_len,
hidden_size=hidden_size,
dtype=dtype,
tp_size=tp_size,
)
for dynamic_input_index in case.dynamic_input_indices:
torch._dynamo.mark_dynamic(inputs[dynamic_input_index], 0)
unfused = model(*inputs)
compiled = torch.compile(model, backend=backend)
fused = compiled(*inputs)
assert unfused.shape == fused.shape
assert sp_moe_pass.matched_count == 1
_assert_op_counts(backend, case.pre_pass_expected_counts_factory(), before=True)
_assert_op_counts(backend, case.post_pass_expected_counts_factory())
def _run_sequence_parallelism_moe_test(
local_rank: int,
world_size: int,
master_port: int,
command_queue: Any,
result_queue: Any,
batch_size: int = 8,
seq_len: int = 16,
hidden_size: int = 16,
dtype: torch.dtype = torch.bfloat16,
eps: float = 1e-5,
) -> None:
torch.npu.set_device(local_rank)
torch.set_default_device(f"npu:{local_rank}")
torch.set_default_dtype(dtype)
torch.manual_seed(0)
update_environment_variables(
{
"RANK": str(local_rank),
"LOCAL_RANK": str(local_rank),
"WORLD_SIZE": str(world_size),
"MASTER_ADDR": "127.0.0.1",
"MASTER_PORT": str(master_port),
}
)
vllm_config = VllmConfig(model_config=ModelConfig(dtype=dtype))
try:
with vllm.config.set_current_vllm_config(vllm_config):
init_distributed_environment(
world_size=world_size,
rank=local_rank,
local_rank=local_rank,
backend="hccl",
)
initialize_model_parallel(tensor_model_parallel_size=world_size)
if not enable_custom_op():
raise RuntimeError("vllm_ascend custom ops are not available")
_ = get_tp_group().unique_name
tp_size = get_tensor_model_parallel_world_size()
result_queue.put((WORKER_READY, local_rank, "ok", ""))
while True:
case_name = command_queue.get()
if case_name == WORKER_STOP:
return
try:
_run_single_pattern_case(
local_rank=local_rank,
case_name=case_name,
vllm_config=vllm_config,
tp_size=tp_size,
batch_size=batch_size,
seq_len=seq_len,
hidden_size=hidden_size,
dtype=dtype,
eps=eps,
)
except Exception:
result_queue.put((case_name, local_rank, "error", traceback.format_exc()))
else:
result_queue.put((case_name, local_rank, "ok", ""))
finally:
destroy_model_parallel()
destroy_distributed_environment()
if torch.distributed.is_initialized():
torch.distributed.destroy_process_group()
def _worker_entrypoint(
local_rank: int,
world_size: int,
master_port: int,
command_queue: Any,
result_queue: Any,
) -> None:
try:
_run_sequence_parallelism_moe_test(
local_rank=local_rank,
world_size=world_size,
master_port=master_port,
command_queue=command_queue,
result_queue=result_queue,
)
except Exception:
result_queue.put((WORKER_READY, local_rank, "error", traceback.format_exc()))
def _wait_for_worker_reports(
result_queue: Any,
case_name: str,
expected_reports: int,
) -> None:
errors = []
for _ in range(expected_reports):
try:
reported_case_name, local_rank, status, payload = result_queue.get(timeout=WORKER_RESULT_TIMEOUT_S)
except queue.Empty as exc:
raise TimeoutError(f"Timed out waiting for worker reports for {case_name}") from exc
assert reported_case_name == case_name, f"Expected worker report for {case_name}, but got {reported_case_name}"
if status != "ok":
errors.append(f"rank {local_rank}:\n{payload}")
if errors:
raise AssertionError("\n\n".join(errors))
@pytest.fixture(scope="module")
def sequence_parallelism_moe_workers() -> Generator[Callable[[str], None], None, None]:
ctx = torch.multiprocessing.get_context("spawn")
command_queues = [ctx.Queue() for _ in range(WORLD_SIZE)]
result_queue = ctx.Queue()
workers = []
for local_rank in range(WORLD_SIZE):
worker = ctx.Process(
target=_worker_entrypoint,
args=(local_rank, WORLD_SIZE, MASTER_PORT, command_queues[local_rank], result_queue),
)
worker.start()
workers.append(worker)
try:
_wait_for_worker_reports(result_queue, WORKER_READY, WORLD_SIZE)
def _run_case(case_name: str) -> None:
for command_queue in command_queues:
command_queue.put(case_name)
_wait_for_worker_reports(result_queue, case_name, WORLD_SIZE)
yield _run_case
finally:
for command_queue in command_queues:
command_queue.put(WORKER_STOP)
for worker in workers:
worker.join(timeout=WORKER_JOIN_TIMEOUT_S)
if worker.is_alive():
worker.terminate()
worker.join()
@pytest.mark.parametrize("case_name", tuple(PATTERN_TEST_CASES), ids=tuple(PATTERN_TEST_CASES))
def test_sequence_parallelism_moe_patterns(
sequence_parallelism_moe_workers: Callable[[str], None], case_name: str
) -> None:
sequence_parallelism_moe_workers(case_name)

View File

@@ -0,0 +1,59 @@
import pytest
from tests.e2e.conftest import wait_until_npu_memory_free
from tests.e2e.pull_request.utils import compare_logprobs
MODELS = [
"deepseek-ai/DeepSeek-V2-Lite",
]
PROMPTS = [
"Hello, my name is",
"The capital of the United States is",
"The capital of France is",
"The future of AI is",
]
@wait_until_npu_memory_free(0.7)
@pytest.mark.parametrize("model", MODELS)
def test_deepseek_v2_lite_enable_shared_expert_dp_tp2(model: str, monkeypatch) -> None:
# FlashComm v1 / shared-expert-DP require HCCL_OP_EXPANSION_MODE to be unset.
monkeypatch.delenv("HCCL_OP_EXPANSION_MODE", raising=False)
# FlashComm1 + shared-expert-DP must stay numerically consistent with the
# plain eager baseline. `additional_config` is excluded from the baseline
# by compare_logprobs, so the baseline runs without either flag.
shared_expert_dp_config = {
"enable_flashcomm1": True,
"enable_shared_expert_dp": True,
}
# Eager mode: FlashComm1 + shared-expert-DP vs eager baseline.
compare_logprobs(
runner_kwargs={
"model_name": model,
"max_model_len": 1024,
"enforce_eager": True,
"tensor_parallel_size": 2,
"enable_expert_parallel": True,
"additional_config": shared_expert_dp_config,
},
prompts=PROMPTS,
)
# ACLGraph (FULL_DECODE_ONLY): FlashComm1 + shared-expert-DP vs eager baseline.
compare_logprobs(
runner_kwargs={
"model_name": model,
"max_model_len": 1024,
"tensor_parallel_size": 2,
"enable_expert_parallel": True,
"compilation_config": {
"cudagraph_capture_sizes": [1, 4, 8, 16],
"cudagraph_mode": "FULL_DECODE_ONLY",
},
"additional_config": shared_expert_dp_config,
},
prompts=PROMPTS,
)

View File

@@ -0,0 +1,61 @@
import pytest
from vllm import SamplingParams
from tests.e2e.conftest import VllmRunner
from tests.e2e.model_utils import check_outputs_equal
MODELS = [
"Qwen/Qwen3-VL-2B-Instruct",
]
@pytest.mark.parametrize("model", MODELS)
def test_qwen3_vl_sp_tp2(model: str) -> None:
prompts = [
"Hello, my name is",
"The capital of the United States is",
"The capital of France is",
"The future of AI is",
]
sampling_params = SamplingParams(max_tokens=10, temperature=0.0)
with VllmRunner(
model,
max_model_len=1024,
tensor_parallel_size=2,
compilation_config={
"cudagraph_capture_sizes": [2, 4],
"cudagraph_mode": "FULL_DECODE_ONLY",
"pass_config": {"enable_sp": False},
},
additional_config={"ascend_compilation_config": {"enable_npugraph_ex": False}},
) as runner:
no_sp_outputs = runner.model.generate(prompts, sampling_params)
with VllmRunner(
model,
max_model_len=1024,
tensor_parallel_size=2,
compilation_config={
"cudagraph_capture_sizes": [2, 4],
"cudagraph_mode": "FULL_DECODE_ONLY",
"pass_config": {"enable_sp": True, "sp_min_token_num": 10},
},
additional_config={"ascend_compilation_config": {"enable_npugraph_ex": False}},
) as runner:
sp_outputs = runner.model.generate(prompts, sampling_params)
no_sp_outputs_list = []
for output in no_sp_outputs:
no_sp_outputs_list.append((output.outputs[0].index, output.outputs[0].text))
sp_outputs_list = []
for output in sp_outputs:
sp_outputs_list.append((output.outputs[0].index, output.outputs[0].text))
check_outputs_equal(
outputs_0_lst=no_sp_outputs_list,
outputs_1_lst=sp_outputs_list,
name_0="no_sp_outputs",
name_1="sp_outputs",
)