0
tests/e2e/pull_request/two_card/__init__.py
Normal file
0
tests/e2e/pull_request/two_card/__init__.py
Normal file
@@ -0,0 +1,232 @@
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import contextlib
|
||||
import gc
|
||||
import math
|
||||
import multiprocessing
|
||||
import os
|
||||
from typing import Any
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
from vllm.utils.network_utils import get_open_port
|
||||
|
||||
from tests.e2e.conftest import wait_until_npu_memory_free
|
||||
from vllm_ascend.utils import AscendDeviceType, get_ascend_device_type
|
||||
|
||||
MODELS = [
|
||||
# Offline data parallel mode will be not supported/useful for dense models
|
||||
# "Qwen/Qwen3-0.6B",
|
||||
"vllm-ascend/DeepSeek-V2-Lite-W8A8",
|
||||
]
|
||||
|
||||
|
||||
def _install_spies(counters: dict[str, Any]) -> contextlib.ExitStack:
|
||||
"""Installs thread-safe spies on NPU methods to track invocation counts."""
|
||||
from vllm_ascend.worker.model_runner_v1 import NPUModelRunner
|
||||
|
||||
def make_spy(cls, method_name, counter):
|
||||
original = getattr(cls, method_name)
|
||||
|
||||
def spy(self, *args, **kwargs):
|
||||
with counter.get_lock():
|
||||
counter.value += 1
|
||||
return original(self, *args, **kwargs)
|
||||
|
||||
return spy
|
||||
|
||||
stack = contextlib.ExitStack()
|
||||
hooks = [
|
||||
(torch.npu.NPUGraph, "replay", counters["replay"]),
|
||||
(torch.npu.NPUGraph, "__init__", counters["capture"]),
|
||||
(NPUModelRunner, "execute_model", counters["exec_model"]),
|
||||
(NPUModelRunner, "_dummy_run", counters["dummy_run"]),
|
||||
]
|
||||
|
||||
for cls, method, counter in hooks:
|
||||
stack.enter_context(patch.object(cls, method, make_spy(cls, method, counter)))
|
||||
|
||||
return stack
|
||||
|
||||
|
||||
def _run_worker_process(
|
||||
rank: int,
|
||||
local_rank: int,
|
||||
world_size: int,
|
||||
master_ip: str,
|
||||
master_port: int,
|
||||
counters: dict[str, Any],
|
||||
model_path: str,
|
||||
max_tokens: int,
|
||||
):
|
||||
"""Main entry point for the worker process."""
|
||||
os.environ.update(
|
||||
{
|
||||
"VLLM_DP_RANK": str(rank),
|
||||
"VLLM_DP_RANK_LOCAL": str(local_rank),
|
||||
"VLLM_DP_SIZE": str(world_size),
|
||||
"VLLM_DP_MASTER_IP": master_ip,
|
||||
"VLLM_DP_MASTER_PORT": str(master_port),
|
||||
}
|
||||
)
|
||||
|
||||
# Import vLLM only after environment setup
|
||||
from vllm import LLM, SamplingParams
|
||||
from vllm.distributed.parallel_state import destroy_distributed_environment, destroy_model_parallel
|
||||
|
||||
# Apply hooks and run inference
|
||||
with _install_spies(counters):
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
# Simple data sharding
|
||||
chunk_size = len(prompts) // world_size
|
||||
start_idx = rank * chunk_size
|
||||
end_idx = start_idx + chunk_size if rank < world_size - 1 else len(prompts)
|
||||
local_prompts = prompts[start_idx:end_idx]
|
||||
|
||||
llm = LLM(
|
||||
model=model_path,
|
||||
quantization="ascend" if "W8A8" in model_path else None,
|
||||
enable_expert_parallel="DeepSeek" in model_path,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
|
||||
# Expose model config to the main test process
|
||||
counters["hidden_layers"].value = llm.llm_engine.model_config.hf_text_config.num_hidden_layers
|
||||
|
||||
llm.generate(local_prompts, SamplingParams(max_tokens=max_tokens, temperature=0.0))
|
||||
|
||||
# Explicit cleanup is mandatory in multi-process vLLM tests
|
||||
del llm
|
||||
|
||||
destroy_model_parallel()
|
||||
destroy_distributed_environment()
|
||||
|
||||
with contextlib.suppress(AssertionError):
|
||||
torch.distributed.destroy_process_group()
|
||||
|
||||
gc.collect()
|
||||
torch.npu.empty_cache()
|
||||
torch.npu.reset_peak_memory_stats()
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="fix me")
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [4, 36])
|
||||
@patch.dict(os.environ, {"ASCEND_RT_VISIBLE_DEVICES": "0,1"})
|
||||
@wait_until_npu_memory_free(target_free_percentage=0.6)
|
||||
def test_models_aclgraph_capture_replay_metrics_dp2(
|
||||
model: str,
|
||||
max_tokens: int,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
# Counter doesn't work in default "spawn" mode
|
||||
monkeypatch.delenv("VLLM_WORKER_MULTIPROC_METHOD", raising=False)
|
||||
|
||||
# Shared counters for cross-process assertion
|
||||
counters = {
|
||||
"replay": multiprocessing.Value("i", 0),
|
||||
"capture": multiprocessing.Value("i", 0),
|
||||
"exec_model": multiprocessing.Value("i", 0),
|
||||
"dummy_run": multiprocessing.Value("i", 0),
|
||||
"hidden_layers": multiprocessing.Value("i", -1),
|
||||
}
|
||||
|
||||
dp_size = 2
|
||||
port = get_open_port()
|
||||
|
||||
# Launch workers
|
||||
workers = []
|
||||
for rank in range(dp_size):
|
||||
p = multiprocessing.Process(
|
||||
target=_run_worker_process,
|
||||
args=(rank, rank, dp_size, "127.0.0.1", port, counters, model, max_tokens),
|
||||
)
|
||||
p.start()
|
||||
workers.append(p)
|
||||
|
||||
# Supervision loop
|
||||
for p in workers:
|
||||
p.join(timeout=900)
|
||||
if p.exitcode != 0:
|
||||
for k in workers:
|
||||
if k.is_alive():
|
||||
k.kill()
|
||||
raise RuntimeError(f"Worker {p.pid} failed with exit code {p.exitcode}")
|
||||
|
||||
actual_capture = counters["capture"].value
|
||||
actual_replay = counters["replay"].value
|
||||
num_execute_model = counters["exec_model"].value
|
||||
num_dummy_run = counters["dummy_run"].value
|
||||
num_layers = counters["hidden_layers"].value
|
||||
|
||||
num_acl_graphs = num_layers + 1
|
||||
num_comm_groups = sum(1 for s in [dp_size, 1] if s > 1) # dp_size=2, tp_size=1
|
||||
|
||||
# Metric 1: Graph Capture (ACL Graph Construction)
|
||||
# Ref: vllm_ascend.utils.update_aclgraph_sizes
|
||||
max_batch_sizes = math.floor((1800 - num_comm_groups * 40) / num_acl_graphs / (1 + num_comm_groups * 2))
|
||||
|
||||
expected_capture = max_batch_sizes * num_acl_graphs * dp_size
|
||||
assert actual_capture == expected_capture, (
|
||||
f"Capture count mismatch. Expected: {expected_capture}, Got: {actual_capture}"
|
||||
)
|
||||
|
||||
# Metric 2: Model Execution (NPUModelRunner.execute_model)
|
||||
# vLLM Step Breakdown:
|
||||
# 1. First step (prefill, 1 prompt)
|
||||
# 2. Generation steps (max_tokens)
|
||||
# 3. Final step (likely EOS/idle step), no replay here
|
||||
total_steps = max_tokens + 1 # this includes the 1 and 2 above
|
||||
# vllm default enables Async scheduler, this will take 1 more steps
|
||||
expected_exec_model = (total_steps + 1 + 1) * dp_size
|
||||
|
||||
assert num_execute_model == expected_exec_model, (
|
||||
f"Model execution count mismatch. Expected: {expected_exec_model}, Got: {num_execute_model}"
|
||||
)
|
||||
|
||||
# Metric 3: Dummy Runs (Warmup & Alignment)
|
||||
# vLLM synchronizes globally every 32 steps.
|
||||
# Ref: vllm.v1.engine.core.DPEngineCoreProc._has_global_unfinished_reqs
|
||||
aligned_steps = (total_steps + 31) // 32 * 32
|
||||
|
||||
# Part A: Warmup runs (Profile run + 2 runs per captured graph)
|
||||
warmup_runs = 1 + (2 * max_batch_sizes)
|
||||
soc_version = get_ascend_device_type()
|
||||
if soc_version in {AscendDeviceType.A3} and "DeepSeek" in model:
|
||||
# An extra warmup run is needed for MC2 warmup here
|
||||
warmup_runs += 1
|
||||
|
||||
# Part B: Alignment padding (Empty runs to hit the 32-step boundary)
|
||||
padding_runs = aligned_steps - total_steps
|
||||
|
||||
expected_dummy_run = (warmup_runs + padding_runs) * dp_size
|
||||
|
||||
assert num_dummy_run == expected_dummy_run, (
|
||||
f"Dummy run count mismatch. Expected: {expected_dummy_run}, Got: {num_dummy_run}"
|
||||
)
|
||||
|
||||
# Metric 4: Graph Replay (Inference Execution)
|
||||
# Replays happen for every aligned step across all graphs.
|
||||
expected_replay = num_acl_graphs * aligned_steps * dp_size
|
||||
|
||||
assert actual_replay == expected_replay, f"Replay count mismatch. Expected: {expected_replay}, Got: {actual_replay}"
|
||||
0
tests/e2e/pull_request/two_card/lora/__init__.py
Normal file
0
tests/e2e/pull_request/two_card/lora/__init__.py
Normal file
24
tests/e2e/pull_request/two_card/lora/test_ilama_lora_tp2.py
Normal file
24
tests/e2e/pull_request/two_card/lora/test_ilama_lora_tp2.py
Normal file
@@ -0,0 +1,24 @@
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.pull_request.one_card.lora.test_ilama_lora import EXPECTED_LORA_OUTPUT, MODEL_PATH, do_sample
|
||||
|
||||
|
||||
@pytest.mark.parametrize("distributed_executor_backend", ["mp"])
|
||||
def test_ilama_lora_tp2(distributed_executor_backend, ilama_lora_files):
|
||||
with VllmRunner(
|
||||
MODEL_PATH,
|
||||
enable_lora=True,
|
||||
max_loras=4,
|
||||
dtype="half",
|
||||
max_model_len=1024,
|
||||
max_num_seqs=16,
|
||||
tensor_parallel_size=2,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
distributed_executor_backend=distributed_executor_backend,
|
||||
enforce_eager=True,
|
||||
) as vllm_model:
|
||||
output = do_sample(vllm_model.model, ilama_lora_files, lora_id=2)
|
||||
|
||||
for i in range(len(EXPECTED_LORA_OUTPUT)):
|
||||
assert output[i] == EXPECTED_LORA_OUTPUT[i]
|
||||
30
tests/e2e/pull_request/two_card/lora/test_llama32_lora_tp2.py
Executable file
30
tests/e2e/pull_request/two_card/lora/test_llama32_lora_tp2.py
Executable file
@@ -0,0 +1,30 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner, wait_until_npu_memory_free
|
||||
from tests.e2e.pull_request.one_card.lora.test_llama32_lora import generate_and_test
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
|
||||
enable_custom_op()
|
||||
|
||||
# For hk region, we need to use the model from hf to avoid the network issue
|
||||
MODEL_PATH = "vllm-ascend/Llama-3.2-3B-Instruct"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("fully_sharded_loras", [False, True])
|
||||
@wait_until_npu_memory_free()
|
||||
def test_llama_lora_tp2(llama32_lora_files, fully_sharded_loras):
|
||||
with VllmRunner(
|
||||
MODEL_PATH,
|
||||
enable_lora=True,
|
||||
# also test odd max_num_seqs
|
||||
max_num_seqs=7,
|
||||
max_model_len=1024,
|
||||
max_loras=4,
|
||||
tensor_parallel_size=2,
|
||||
fully_sharded_loras=fully_sharded_loras,
|
||||
compilation_config={"cudagraph_mode": "PIECEWISE"},
|
||||
) as vllm_model:
|
||||
llm = vllm_model.model
|
||||
generate_and_test(llm, llama32_lora_files)
|
||||
@@ -0,0 +1,68 @@
|
||||
import vllm
|
||||
from vllm.lora.request import LoRARequest
|
||||
|
||||
MODEL_PATH = "Qwen/Qwen3-30B-A3B"
|
||||
|
||||
PROMPT_TEMPLATE = """<|im_start|>user
|
||||
I want you to act as a SQL terminal in front of an example database, you need only to return the sql command to me.Below is an instruction that describes a task, Write a response that appropriately completes the request.
|
||||
"
|
||||
##Instruction:
|
||||
candidate_poll contains tables such as candidate, people. Table candidate has columns such as Candidate_ID, People_ID, Poll_Source, Date, Support_rate, Consider_rate, Oppose_rate, Unsure_rate. Candidate_ID is the primary key.
|
||||
Table people has columns such as People_ID, Sex, Name, Date_of_Birth, Height, Weight. People_ID is the primary key.
|
||||
The People_ID of candidate is the foreign key of People_ID of people.
|
||||
|
||||
|
||||
###Input:
|
||||
{context}
|
||||
|
||||
###Response:<|im_end|>
|
||||
<|im_start|>assistant""" # noqa: E501
|
||||
|
||||
EXPECTED_LORA_OUTPUT = [
|
||||
"<think>\n\n</think>\n\nSELECT count(*) FROM candidate",
|
||||
"<think>\n\n</think>\n\nSELECT count(*) FROM candidate",
|
||||
"<think>\n\n</think>\n\nSELECT poll_source FROM candidate GROUP BY poll_source ORDER BY count(*) DESC LIMIT 1", # noqa: E501
|
||||
"<think>\n\n</think>\n\nSELECT poll_source FROM candidate GROUP BY poll_source ORDER BY count(*) DESC LIMIT 1", # noqa: E501
|
||||
]
|
||||
|
||||
|
||||
def generate_and_test(llm: vllm.LLM, lora_path: str, lora_id: int) -> None:
|
||||
prompts = [
|
||||
PROMPT_TEMPLATE.format(context="How many candidates are there?"),
|
||||
PROMPT_TEMPLATE.format(context="Count the number of candidates."),
|
||||
PROMPT_TEMPLATE.format(
|
||||
context="Which poll resource provided the most number of candidate information?" # noqa: E501
|
||||
),
|
||||
PROMPT_TEMPLATE.format(context="Return the poll resource associated with the most candidates."),
|
||||
]
|
||||
sampling_params = vllm.SamplingParams(temperature=0, max_tokens=64)
|
||||
outputs = llm.generate(
|
||||
prompts,
|
||||
sampling_params,
|
||||
lora_request=LoRARequest(str(lora_id), lora_id, lora_path) if lora_id else None,
|
||||
)
|
||||
# Print the outputs.
|
||||
generated_texts: list[str] = []
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text.strip()
|
||||
generated_texts.append(generated_text)
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
|
||||
for i in range(len(EXPECTED_LORA_OUTPUT)):
|
||||
assert generated_texts[i].startswith(EXPECTED_LORA_OUTPUT[i])
|
||||
|
||||
|
||||
def test_qwen3moe_lora(qwen3moe_lora_files):
|
||||
llm = vllm.LLM(
|
||||
MODEL_PATH,
|
||||
max_model_len=1024,
|
||||
enable_lora=True,
|
||||
max_loras=4,
|
||||
enforce_eager=True,
|
||||
trust_remote_code=True,
|
||||
enable_chunked_prefill=True,
|
||||
tensor_parallel_size=2,
|
||||
)
|
||||
|
||||
generate_and_test(llm, qwen3moe_lora_files, lora_id=1)
|
||||
429
tests/e2e/pull_request/two_card/spec_decode/test_spec_decode.py
Normal file
429
tests/e2e/pull_request/two_card/spec_decode/test_spec_decode.py
Normal file
@@ -0,0 +1,429 @@
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
# Run `pytest tests/e2e/pull_request/two_card/spec_decode/test_spec_decode.py`.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from transformers import AutoTokenizer
|
||||
from vllm import SamplingParams
|
||||
from vllm.config import CompilationConfig
|
||||
from vllm.tokenizers.registry import resolve_tokenizer_args
|
||||
from vllm.v1.metrics.reader import Counter, Vector
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
|
||||
MODELS = {
|
||||
"eagle3": {
|
||||
"main": "Qwen/Qwen3-8B",
|
||||
"spec": "RedHatAI/Qwen3-8B-speculator.eagle3",
|
||||
},
|
||||
}
|
||||
|
||||
P_EAGLE_MODELS = {
|
||||
"p-eagle": {
|
||||
"main": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
|
||||
"spec": "amazon/Qwen3-Coder-30B-A3B-Instruct-P-EAGLE",
|
||||
},
|
||||
}
|
||||
|
||||
VWN_EAGLE3_MODELS = {
|
||||
"vwn_eagle3": {
|
||||
"main": "Qwen/Qwen3-30B-A3B",
|
||||
"spec": "vllm-ascend/Qwen3-30B-A3B-vwn-eagle-model",
|
||||
},
|
||||
}
|
||||
|
||||
# NOTE: golden may change (eagle_proposer only runs in eager mode currently),
|
||||
# thus please update it if ci fails but you have better acceptance
|
||||
BASELINES_SP = {
|
||||
"eagle3": [0.68, 0.40, 0.18],
|
||||
"p-eagle": [0.5625, 0.25, 0.0625, 0.0, 0.0, 0.0, 0.0, 0.0],
|
||||
"vwn_eagle3": [0.75, 0.5, 0.3],
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="skip test_eagle3_sp_acceptance")
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
|
||||
@pytest.mark.parametrize("method", ["eagle3"])
|
||||
@pytest.mark.parametrize("num_speculative_tokens", [3])
|
||||
@pytest.mark.parametrize("disable_padded_drafter_batch", [True, False])
|
||||
@pytest.mark.parametrize("async_scheduling", [True, False])
|
||||
def test_eagle3_sp_acceptance(
|
||||
method: str,
|
||||
num_speculative_tokens: int,
|
||||
disable_padded_drafter_batch: bool,
|
||||
async_scheduling: bool,
|
||||
):
|
||||
if disable_padded_drafter_batch and async_scheduling:
|
||||
pytest.skip(
|
||||
"skip disable_padded_drafter_batch=True and async_scheduling=True",
|
||||
)
|
||||
|
||||
main_model_name = MODELS[method]["main"]
|
||||
spec_model_name = MODELS[method]["spec"]
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
main_model_name,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0,
|
||||
ignore_eos=False,
|
||||
max_tokens=256,
|
||||
)
|
||||
|
||||
# sp will only be enabled when query_lens > 1000
|
||||
prompts = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": " " * 1000 + "Hello, my name is",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": " " * 1000 + "The president of the United States is",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": " " * 1000 + "The capital of France is",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": " " * 1000 + "The future of AI is",
|
||||
},
|
||||
]
|
||||
prompts = [
|
||||
tokenizer.apply_chat_template(
|
||||
[prompt],
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
)
|
||||
for prompt in prompts
|
||||
]
|
||||
|
||||
speculative_config = {
|
||||
"enforce_eager": True,
|
||||
"method": method,
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
"disable_padded_drafter_batch": disable_padded_drafter_batch,
|
||||
"model": spec_model_name,
|
||||
}
|
||||
|
||||
compilation_config = CompilationConfig(cudagraph_mode="FULL_DECODE_ONLY", cudagraph_capture_sizes=[12])
|
||||
|
||||
with VllmRunner(
|
||||
main_model_name,
|
||||
enforce_eager=True,
|
||||
max_model_len=8192,
|
||||
disable_log_stats=False,
|
||||
tensor_parallel_size=2,
|
||||
max_num_seqs=256,
|
||||
distributed_executor_backend="mp",
|
||||
gpu_memory_utilization=0.7,
|
||||
speculative_config=speculative_config,
|
||||
compilation_config=compilation_config,
|
||||
async_scheduling=async_scheduling,
|
||||
) as llm:
|
||||
_ = llm.generate(prompts, sampling_params)
|
||||
metrics = llm.model.get_metrics()
|
||||
|
||||
num_drafts = 0
|
||||
num_accepted_tokens_per_pos = [0] * num_speculative_tokens
|
||||
for metric in metrics:
|
||||
if metric.name == "vllm:spec_decode_num_drafts":
|
||||
assert isinstance(metric, Counter)
|
||||
num_drafts += metric.value
|
||||
elif metric.name == "vllm:spec_decode_num_accepted_tokens_per_pos":
|
||||
assert isinstance(metric, Vector)
|
||||
for pos in range(len(metric.values)):
|
||||
num_accepted_tokens_per_pos[pos] += metric.values[pos]
|
||||
|
||||
acceptance_per_pos = [num_accepted_tokens / num_drafts for num_accepted_tokens in num_accepted_tokens_per_pos]
|
||||
golden = BASELINES_SP[method]
|
||||
|
||||
match = all(abs(a - b) < 0.06 for a, b in zip(acceptance_per_pos, golden))
|
||||
if not match:
|
||||
print(f"acceptance_per_pos: {acceptance_per_pos}")
|
||||
print(f"golden: {golden}")
|
||||
|
||||
assert match
|
||||
|
||||
|
||||
def test_qwen3_eagle3_pcp2_tp1():
|
||||
"""
|
||||
Test Qwen3-8B with Eagle3 speculative decoding under PCP + TP1 configuration.
|
||||
This test verifies that eagle3 spec decode works correctly with:
|
||||
- PCP enabled (prefill_context_parallel_size=2)
|
||||
- Tensor Parallel size = 1
|
||||
- num_speculative_tokens = 3
|
||||
- enforce_eager = True
|
||||
"""
|
||||
method = "eagle3"
|
||||
num_speculative_tokens = 3
|
||||
|
||||
main_model_name = MODELS[method]["main"]
|
||||
spec_model_name = MODELS[method]["spec"]
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
main_model_name,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0,
|
||||
ignore_eos=False,
|
||||
max_tokens=256,
|
||||
)
|
||||
|
||||
prompts = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hello, my name is",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "The president of the United States is",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "The capital of France is",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "The future of AI is",
|
||||
},
|
||||
]
|
||||
prompts = [
|
||||
tokenizer.apply_chat_template(
|
||||
[prompt],
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
)
|
||||
for prompt in prompts
|
||||
]
|
||||
|
||||
speculative_config = {
|
||||
"method": method,
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
"model": spec_model_name,
|
||||
}
|
||||
|
||||
with VllmRunner(
|
||||
main_model_name,
|
||||
enforce_eager=True,
|
||||
max_model_len=2048,
|
||||
disable_log_stats=False,
|
||||
tensor_parallel_size=1,
|
||||
prefill_context_parallel_size=2,
|
||||
max_num_seqs=256,
|
||||
distributed_executor_backend="mp",
|
||||
gpu_memory_utilization=0.7,
|
||||
speculative_config=speculative_config,
|
||||
) as llm:
|
||||
llm.generate(prompts, sampling_params)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("method", P_EAGLE_MODELS.keys())
|
||||
@pytest.mark.parametrize("num_speculative_tokens", [8])
|
||||
@pytest.mark.parametrize("draft_tensor_parallel_size", [None, 2])
|
||||
def test_p_eagle_acceptance(
|
||||
method: str,
|
||||
num_speculative_tokens: int,
|
||||
draft_tensor_parallel_size: None | int,
|
||||
):
|
||||
"""
|
||||
Test acceptance rate for parallel drafting speculative decoding
|
||||
using a smaller draft model with parallel_drafting enabled.
|
||||
"""
|
||||
main_model_name = P_EAGLE_MODELS[method]["main"]
|
||||
spec_model_name = P_EAGLE_MODELS[method]["spec"]
|
||||
|
||||
tokenizer_path = resolve_tokenizer_args(main_model_name)[1]
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
tokenizer_path,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0,
|
||||
ignore_eos=False,
|
||||
max_tokens=256,
|
||||
)
|
||||
|
||||
prompts = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hello, your name is",
|
||||
},
|
||||
]
|
||||
prompts = [
|
||||
tokenizer.apply_chat_template(
|
||||
[prompt],
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
)
|
||||
for prompt in prompts
|
||||
]
|
||||
|
||||
speculative_config = {
|
||||
"method": "eagle3",
|
||||
"model": spec_model_name,
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
"draft_tensor_parallel_size": draft_tensor_parallel_size,
|
||||
"parallel_drafting": True,
|
||||
}
|
||||
|
||||
compilation_config = CompilationConfig(cudagraph_capture_sizes=[12])
|
||||
|
||||
with VllmRunner(
|
||||
main_model_name,
|
||||
max_model_len=4096,
|
||||
disable_log_stats=False,
|
||||
tensor_parallel_size=2,
|
||||
max_num_seqs=256,
|
||||
distributed_executor_backend="mp",
|
||||
gpu_memory_utilization=0.8,
|
||||
speculative_config=speculative_config,
|
||||
compilation_config=compilation_config,
|
||||
enable_prefix_caching=False,
|
||||
) as llm:
|
||||
outputs = llm.model.generate(prompts, sampling_params)
|
||||
metrics = llm.model.get_metrics()
|
||||
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
output_tokens = output.outputs[0].token_ids
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
print(f"Output tokens: {output_tokens}")
|
||||
|
||||
num_drafts = 0
|
||||
num_accepted_tokens_per_pos = [0] * num_speculative_tokens
|
||||
for metric in metrics:
|
||||
if metric.name == "vllm:spec_decode_num_drafts":
|
||||
assert isinstance(metric, Counter)
|
||||
num_drafts += metric.value
|
||||
elif metric.name == "vllm:spec_decode_num_accepted_tokens_per_pos":
|
||||
assert isinstance(metric, Vector)
|
||||
for pos in range(len(metric.values)):
|
||||
num_accepted_tokens_per_pos[pos] += metric.values[pos]
|
||||
|
||||
acceptance_per_pos = [num_accepted_tokens / num_drafts for num_accepted_tokens in num_accepted_tokens_per_pos]
|
||||
|
||||
golden = BASELINES_SP[method]
|
||||
|
||||
match = all(abs(a - b) < 0.1 for a, b in zip(acceptance_per_pos, golden))
|
||||
if not match:
|
||||
print(f"acceptance_per_pos: {acceptance_per_pos}")
|
||||
print(f"golden: {golden}")
|
||||
|
||||
assert match
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
|
||||
def test_qwen3_vwn_eagle3_tp2():
|
||||
"""
|
||||
Test Qwen3-30B-A3B with VWN-Eagle3 speculative decoding acceptance rate.
|
||||
This test verifies that VWN-Eagle3 spec decode works correctly with:
|
||||
- Tensor Parallel size = 4
|
||||
- Expert Parallel enabled (for MoE)
|
||||
- num_speculative_tokens = 3
|
||||
- enforce_eager = True
|
||||
- Acceptance rate matches baseline (tolerance 0.06)
|
||||
"""
|
||||
num_speculative_tokens = 3
|
||||
main_model_name = VWN_EAGLE3_MODELS["vwn_eagle3"]["main"]
|
||||
spec_model_name = VWN_EAGLE3_MODELS["vwn_eagle3"]["spec"]
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
main_model_name,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0,
|
||||
ignore_eos=False,
|
||||
max_tokens=256,
|
||||
)
|
||||
|
||||
prompts = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hello, my name is",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "The capital of France is",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "The future of AI is",
|
||||
},
|
||||
]
|
||||
prompts = [
|
||||
tokenizer.apply_chat_template(
|
||||
[prompt],
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
)
|
||||
for prompt in prompts
|
||||
]
|
||||
|
||||
speculative_config = {
|
||||
"method": "eagle3",
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
"model": spec_model_name,
|
||||
}
|
||||
|
||||
with VllmRunner(
|
||||
main_model_name,
|
||||
enforce_eager=True,
|
||||
max_model_len=2048,
|
||||
disable_log_stats=False,
|
||||
tensor_parallel_size=2,
|
||||
max_num_seqs=16,
|
||||
distributed_executor_backend="mp",
|
||||
gpu_memory_utilization=0.92,
|
||||
speculative_config=speculative_config,
|
||||
enable_expert_parallel=True,
|
||||
) as llm:
|
||||
_ = llm.generate(prompts, sampling_params)
|
||||
metrics = llm.model.get_metrics()
|
||||
|
||||
# Check acceptance rate
|
||||
num_drafts = 0
|
||||
num_accepted_tokens_per_pos = [0] * num_speculative_tokens
|
||||
for metric in metrics:
|
||||
if metric.name == "vllm:spec_decode_num_drafts":
|
||||
assert isinstance(metric, Counter)
|
||||
num_drafts += metric.value
|
||||
elif metric.name == "vllm:spec_decode_num_accepted_tokens_per_pos":
|
||||
assert isinstance(metric, Vector)
|
||||
for pos in range(len(metric.values)):
|
||||
num_accepted_tokens_per_pos[pos] += metric.values[pos]
|
||||
|
||||
acceptance_per_pos = [n / num_drafts for n in num_accepted_tokens_per_pos]
|
||||
golden = BASELINES_SP["vwn_eagle3"]
|
||||
|
||||
match = all(abs(a - b) < 0.06 for a, b in zip(acceptance_per_pos, golden))
|
||||
if not match:
|
||||
print(f"acceptance_per_pos: {acceptance_per_pos}")
|
||||
print(f"golden: {golden}")
|
||||
|
||||
assert match
|
||||
79
tests/e2e/pull_request/two_card/test_data_parallel.py
Normal file
79
tests/e2e/pull_request/two_card/test_data_parallel.py
Normal file
@@ -0,0 +1,79 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
"""
|
||||
Compare the outputs of vLLM with and without aclgraph.
|
||||
|
||||
Run `pytest tests/e2e/pull_request/two_card/test_data_parallel.py`.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import wait_until_npu_memory_free
|
||||
|
||||
MODELS = ["Qwen/Qwen3-30B-A3B", "vllm-ascend/Qwen3-30B-A3B-W8A8"]
|
||||
REPO_ROOT = Path(__file__).resolve().parents[4]
|
||||
DATA_PARALLEL_SCRIPT = REPO_ROOT / "examples" / "offline_data_parallel.py"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [32])
|
||||
@patch.dict(os.environ, {"ASCEND_RT_VISIBLE_DEVICES": "0,1"})
|
||||
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "1024"})
|
||||
@wait_until_npu_memory_free(target_free_percentage=0.7)
|
||||
def test_qwen3_inference_dp2(model, max_tokens):
|
||||
moe_models = ["Qwen/Qwen3-30B-A3B", "vllm-ascend/Qwen3-30B-A3B-W8A8"]
|
||||
quantization_models = ["vllm-ascend/Qwen3-30B-A3B-W8A8"]
|
||||
env = os.environ.copy()
|
||||
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(DATA_PARALLEL_SCRIPT),
|
||||
"--model",
|
||||
model,
|
||||
"--dp-size",
|
||||
"2",
|
||||
"--tp-size",
|
||||
"1",
|
||||
"--node-size",
|
||||
"1",
|
||||
"--node-rank",
|
||||
"0",
|
||||
"--trust-remote-code",
|
||||
]
|
||||
|
||||
if model in moe_models:
|
||||
cmd.append("--enable-expert-parallel")
|
||||
if model in quantization_models:
|
||||
cmd.append("--quantization")
|
||||
cmd.append("ascend")
|
||||
|
||||
print(f"Running subprocess: {' '.join(cmd)}")
|
||||
proc = subprocess.run(cmd, env=env, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, timeout=600)
|
||||
output = proc.stdout.decode(errors="ignore")
|
||||
|
||||
print(output)
|
||||
|
||||
assert "DP rank 0 needs to process" in output
|
||||
assert "DP rank 1 needs to process" in output
|
||||
assert "Generated text:" in output
|
||||
assert proc.returncode == 0
|
||||
@@ -0,0 +1,39 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
|
||||
def test_deepseek_multistream_moe_tp2():
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
dtype = "half"
|
||||
max_tokens = 5
|
||||
with VllmRunner(
|
||||
"vllm-ascend/DeepSeek-V3-Pruning",
|
||||
dtype=dtype,
|
||||
tensor_parallel_size=2,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
distributed_executor_backend="mp",
|
||||
additional_config={
|
||||
"enable_multistream_moe": True,
|
||||
"refresh": True,
|
||||
},
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
@@ -0,0 +1,97 @@
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
import pytest
|
||||
from vllm.utils.network_utils import get_open_port
|
||||
|
||||
from tests.e2e.conftest import DisaggEpdProxy, RemoteEPDServer
|
||||
from tools.send_mm_request import send_image_request
|
||||
|
||||
MODELS = [
|
||||
"Qwen/Qwen2.5-VL-7B-Instruct",
|
||||
]
|
||||
SHARED_STORAGE_PATH = "/dev/shm/epd/storage"
|
||||
TENSOR_PARALLELS = [1]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("tp_size", TENSOR_PARALLELS)
|
||||
async def test_models(model: str, tp_size: int) -> None:
|
||||
encode_port = get_open_port()
|
||||
pd_port = get_open_port()
|
||||
vllm_server_args = [
|
||||
[
|
||||
"--port",
|
||||
str(encode_port),
|
||||
"--model",
|
||||
model,
|
||||
"--gpu-memory-utilization",
|
||||
"0.01",
|
||||
"--tensor-parallel-size",
|
||||
str(tp_size),
|
||||
"--enforce-eager",
|
||||
"--no-enable-prefix-caching",
|
||||
"--max-model-len",
|
||||
"10000",
|
||||
"--max-num-batched-tokens",
|
||||
"10000",
|
||||
"--max-num-seqs",
|
||||
"1",
|
||||
"--ec-transfer-config",
|
||||
'{"ec_connector_extra_config":{"shared_storage_path":"'
|
||||
+ SHARED_STORAGE_PATH
|
||||
+ '"},"ec_connector":"ECExampleConnector","ec_role": "ec_producer"}',
|
||||
],
|
||||
[
|
||||
"--port",
|
||||
str(pd_port),
|
||||
"--model",
|
||||
model,
|
||||
"--gpu-memory-utilization",
|
||||
"0.95",
|
||||
"--tensor-parallel-size",
|
||||
str(tp_size),
|
||||
"--enforce-eager",
|
||||
"--max-model-len",
|
||||
"10000",
|
||||
"--max-num-batched-tokens",
|
||||
"10000",
|
||||
"--max-num-seqs",
|
||||
"128",
|
||||
"--ec-transfer-config",
|
||||
'{"ec_connector_extra_config":{"shared_storage_path":"'
|
||||
+ SHARED_STORAGE_PATH
|
||||
+ '"},"ec_connector":"ECExampleConnector","ec_role": "ec_consumer"}',
|
||||
],
|
||||
]
|
||||
proxy_port = get_open_port()
|
||||
proxy_args = [
|
||||
"--host",
|
||||
"127.0.0.1",
|
||||
"--port",
|
||||
str(proxy_port),
|
||||
"--encode-servers-urls",
|
||||
f"http://localhost:{encode_port}",
|
||||
"--decode-servers-urls",
|
||||
f"http://localhost:{pd_port}",
|
||||
"--prefill-servers-urls",
|
||||
"disable",
|
||||
]
|
||||
|
||||
with RemoteEPDServer(vllm_serve_args=vllm_server_args) as _, DisaggEpdProxy(proxy_args=proxy_args) as proxy:
|
||||
send_image_request(model, proxy)
|
||||
229
tests/e2e/pull_request/two_card/test_external_launcher.py
Normal file
229
tests/e2e/pull_request/two_card/test_external_launcher.py
Normal file
@@ -0,0 +1,229 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
"""
|
||||
Compare the outputs of vLLM with and without aclgraph.
|
||||
|
||||
Run `pytest tests/e2e/pull_request/two_card/test_external_launcher.py`.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import huggingface_hub
|
||||
import pytest
|
||||
import torch_npu
|
||||
from modelscope import snapshot_download # type: ignore
|
||||
|
||||
from tests.e2e.conftest import wait_until_npu_memory_free
|
||||
|
||||
MODELS = ["Qwen/Qwen3-0.6B"]
|
||||
MOE_MODELS = ["Qwen/Qwen3-30B-A3B"]
|
||||
DEVICE_NAME = torch_npu.npu.get_device_name(0)[:10]
|
||||
REPO_ROOT = Path(__file__).resolve().parents[4]
|
||||
EXTERNAL_LAUNCHER_SCRIPT = REPO_ROOT / "examples" / "offline_external_launcher.py"
|
||||
EXTERNAL_LAUNCHER_TIMEOUT_S = 720
|
||||
|
||||
|
||||
def _decode_output(output):
|
||||
if output is None:
|
||||
return ""
|
||||
if isinstance(output, bytes):
|
||||
return output.decode(errors="ignore")
|
||||
return output
|
||||
|
||||
|
||||
def _run_external_launcher(cmd, env):
|
||||
env = env.copy()
|
||||
env["PYTHONUNBUFFERED"] = "1"
|
||||
|
||||
print(f"Running subprocess: {' '.join(cmd)}")
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
cmd,
|
||||
env=env,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
timeout=EXTERNAL_LAUNCHER_TIMEOUT_S,
|
||||
)
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
print(f"Subprocess timed out after {EXTERNAL_LAUNCHER_TIMEOUT_S} seconds.")
|
||||
output = _decode_output(exc.output)
|
||||
if output:
|
||||
print(output)
|
||||
else:
|
||||
print("No subprocess output captured before timeout.")
|
||||
raise
|
||||
|
||||
output = _decode_output(proc.stdout)
|
||||
print(output)
|
||||
return proc, output
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "500"})
|
||||
def test_qwen3_external_launcher(model):
|
||||
env = os.environ.copy()
|
||||
# TODO: Change to 2 when ci machine has 4 cards
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(EXTERNAL_LAUNCHER_SCRIPT),
|
||||
"--model",
|
||||
model,
|
||||
"--tp-size",
|
||||
"1",
|
||||
"--node-size",
|
||||
"1",
|
||||
"--node-rank",
|
||||
"0",
|
||||
"--proc-per-node",
|
||||
"2",
|
||||
"--trust-remote-code",
|
||||
]
|
||||
|
||||
proc, output = _run_external_launcher(cmd, env)
|
||||
|
||||
assert "TP RANKS: [0]" in output
|
||||
assert "TP RANKS: [1]" in output
|
||||
assert "Generated text:" in output
|
||||
assert proc.returncode == 0
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MOE_MODELS)
|
||||
@wait_until_npu_memory_free(target_free_percentage=0.7)
|
||||
def test_qwen3_moe_external_launcher_ep_tp2(model):
|
||||
env = os.environ.copy()
|
||||
# TODO: Change to 2 when ci machine has 4 cards
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(EXTERNAL_LAUNCHER_SCRIPT),
|
||||
"--model",
|
||||
model,
|
||||
"--tp-size",
|
||||
"2",
|
||||
"--node-size",
|
||||
"1",
|
||||
"--node-rank",
|
||||
"0",
|
||||
"--proc-per-node",
|
||||
"2",
|
||||
"--trust-remote-code",
|
||||
"--enable-expert-parallel",
|
||||
]
|
||||
|
||||
proc, output = _run_external_launcher(cmd, env)
|
||||
|
||||
assert "TP RANKS: [0, 1]" in output
|
||||
assert "Generated text:" in output
|
||||
assert proc.returncode == 0
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_NZ": "0"})
|
||||
@wait_until_npu_memory_free(target_free_percentage=0.7)
|
||||
def test_qwen3_external_launcher_with_sleepmode():
|
||||
env = os.environ.copy()
|
||||
# TODO: Change to 2 when ci machine has 4 cards
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(EXTERNAL_LAUNCHER_SCRIPT),
|
||||
"--model",
|
||||
"Qwen/Qwen3-8B",
|
||||
"--tp-size",
|
||||
"1",
|
||||
"--node-size",
|
||||
"1",
|
||||
"--node-rank",
|
||||
"0",
|
||||
"--proc-per-node",
|
||||
"2",
|
||||
"--trust-remote-code",
|
||||
"--enable-sleep-mode",
|
||||
"--temperature",
|
||||
"0",
|
||||
"--model-weight-gib",
|
||||
"16",
|
||||
]
|
||||
|
||||
proc, output = _run_external_launcher(cmd, env)
|
||||
|
||||
assert "Generated text:" in output
|
||||
assert "Sleep and wake up successfully!!" in output
|
||||
assert proc.returncode == 0
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_NZ": "0"})
|
||||
@wait_until_npu_memory_free(target_free_percentage=0.7)
|
||||
def test_qwen3_external_launcher_with_sleepmode_level2():
|
||||
env = os.environ.copy()
|
||||
model_path = snapshot_download(
|
||||
"Qwen/Qwen3-8B",
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
# TODO: Add moe model test
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(EXTERNAL_LAUNCHER_SCRIPT),
|
||||
"--model",
|
||||
model_path,
|
||||
"--tp-size",
|
||||
"1",
|
||||
"--node-size",
|
||||
"1",
|
||||
"--node-rank",
|
||||
"0",
|
||||
"--proc-per-node",
|
||||
"2",
|
||||
"--trust-remote-code",
|
||||
"--enable-sleep-mode",
|
||||
"--temperature",
|
||||
"0",
|
||||
"--model-weight-gib",
|
||||
"16",
|
||||
"--sleep-mode-level",
|
||||
"2",
|
||||
]
|
||||
|
||||
proc, output = _run_external_launcher(cmd, env)
|
||||
|
||||
assert "Generated text:" in output
|
||||
assert "Sleep and wake up successfully!!" in output
|
||||
assert proc.returncode == 0
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
DEVICE_NAME != "Ascend910B",
|
||||
reason="This test is only for Ascend910B devices.",
|
||||
)
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@wait_until_npu_memory_free(target_free_percentage=0.7)
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "1", "HCCL_BUFFSIZE": "500"})
|
||||
def test_qwen3_external_launcher_with_matmul_allreduce(model):
|
||||
env = os.environ.copy()
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(EXTERNAL_LAUNCHER_SCRIPT),
|
||||
"--model",
|
||||
model,
|
||||
"--trust-remote-code",
|
||||
]
|
||||
|
||||
proc, output = _run_external_launcher(cmd, env)
|
||||
|
||||
assert "Generated text:" in output
|
||||
assert proc.returncode == 0
|
||||
117
tests/e2e/pull_request/two_card/test_flashcomm_distributed.py
Normal file
117
tests/e2e/pull_request/two_card/test_flashcomm_distributed.py
Normal file
@@ -0,0 +1,117 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
|
||||
#
|
||||
"""Compare the short outputs of HF and vLLM when using greedy sampling.
|
||||
|
||||
Run `pytest tests/e2e/pull_request/two_card/test_flashcomm_distributed.py`.
|
||||
"""
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
from vllm.config import KVTransferConfig
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
QWEN_DENSE_MODELS = [
|
||||
"vllm-ascend/Qwen3-0.6B-W8A8",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="test is broken, fix me")
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_FLASHCOMM2_PARALLEL_SIZE": "1"})
|
||||
def test_qwen3_moe_fc2_oshard_tp2() -> None:
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
sampling_params = SamplingParams(max_tokens=5, temperature=0.0, top_k=50, top_p=0.9)
|
||||
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3-30B-A3B",
|
||||
dtype="auto",
|
||||
tensor_parallel_size=2,
|
||||
distributed_executor_backend="mp",
|
||||
enable_expert_parallel=True,
|
||||
enforce_eager=True,
|
||||
additional_config={"layer_sharding": ["o_proj"]},
|
||||
kv_transfer_config=KVTransferConfig(kv_role="kv_producer"),
|
||||
) as vllm_model:
|
||||
vllm_model.generate(example_prompts, sampling_params)
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="test is broken, fix me")
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
|
||||
def test_deepseek_v2_lite_fc1_tp2() -> None:
|
||||
example_prompts = [
|
||||
"test" * 1001,
|
||||
]
|
||||
sampling_params = SamplingParams(max_tokens=5, temperature=0.0, top_k=50, top_p=0.9)
|
||||
with VllmRunner(
|
||||
"vllm-ascend/DeepSeek-V2-Lite-W8A8",
|
||||
dtype="auto",
|
||||
tensor_parallel_size=2,
|
||||
distributed_executor_backend="mp",
|
||||
enable_expert_parallel=True,
|
||||
enforce_eager=True,
|
||||
quantization="ascend",
|
||||
) as vllm_model:
|
||||
vllm_model.generate(example_prompts, sampling_params)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", QWEN_DENSE_MODELS)
|
||||
@pytest.mark.skip(reason="test is broken, fix me")
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
|
||||
def test_qwen3_dense_fc1_tp2(model):
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
max_tokens = 5
|
||||
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=8192,
|
||||
dtype="auto",
|
||||
tensor_parallel_size=2,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
quantization="ascend",
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", QWEN_DENSE_MODELS)
|
||||
@pytest.mark.skip(reason="test is broken, fix me")
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
|
||||
def test_qwen3_dense_prefetch_mlp_weight_tp2(model):
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
max_tokens = 5
|
||||
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=8192,
|
||||
dtype="auto",
|
||||
tensor_parallel_size=2,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
quantization="ascend",
|
||||
additional_config={"weight_prefetch_config": {"enabled": True}},
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
44
tests/e2e/pull_request/two_card/test_gpt_oss_distributed.py
Normal file
44
tests/e2e/pull_request/two_card/test_gpt_oss_distributed.py
Normal file
@@ -0,0 +1,44 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
|
||||
#
|
||||
"""Compare the short outputs of HF and vLLM when using greedy sampling.
|
||||
|
||||
Run `pytest tests/e2e/pull_request/two_card/test_gpt_oss_distributed.py`.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
GPT_OSS_MODELS = [
|
||||
"unsloth/gpt-oss-20b-BF16",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", GPT_OSS_MODELS)
|
||||
def test_gpt_oss_distributed_tp2(model):
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
max_tokens = 5
|
||||
with VllmRunner(
|
||||
model,
|
||||
tensor_parallel_size=2,
|
||||
enforce_eager=True,
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
342
tests/e2e/pull_request/two_card/test_hccl_weight_transfer.py
Normal file
342
tests/e2e/pull_request/two_card/test_hccl_weight_transfer.py
Normal file
@@ -0,0 +1,342 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
"""End-to-end test for the HCCL weight transfer engine.
|
||||
|
||||
This test starts a vLLM server with dummy weights and the HCCL weight transfer
|
||||
backend enabled, then runs the trainer side of an RLHF-style weight sync from a
|
||||
separate NPU. It exercises the full control plane (HTTP) + data plane (HCCL
|
||||
packed broadcast + layerwise reload) and asserts the server's weights actually
|
||||
change after the broadcast.
|
||||
|
||||
To keep the test self-contained and download-free, the trainer model is built
|
||||
from the architecture config with random weights (only the tiny config/tokenizer
|
||||
are needed, which the server already fetches). The parameter names/shapes/dtypes
|
||||
match the real checkpoint, so the broadcast pipeline is fully exercised; we just
|
||||
don't assert "coherent text" since the broadcast weights are random. Set
|
||||
``WEIGHT_TRANSFER_TEST_MODEL=/path/to/checkpoint`` to instead broadcast real
|
||||
weights from a local checkpoint.
|
||||
|
||||
Topology (requires 2 NPUs):
|
||||
- NPU 0: vLLM inference worker (rank 1 in the HCCL group)
|
||||
- NPU 1: trainer / weight source (rank 0 in the HCCL group)
|
||||
|
||||
Refer to ``examples/rl/rlhf_http_hccl.py`` for the end-user workflow.
|
||||
|
||||
Run with::
|
||||
|
||||
pytest tests/e2e/multicard/2-cards/test_weight_transfer_hccl.py
|
||||
"""
|
||||
|
||||
import os
|
||||
import threading
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
import torch
|
||||
import torch_npu # noqa: F401 # registers the NPU backend
|
||||
from transformers import AutoConfig, AutoModelForCausalLM
|
||||
from vllm.utils.network_utils import get_ip, get_open_port
|
||||
|
||||
from tests.e2e.conftest import RemoteOpenAIServer
|
||||
|
||||
MODEL_NAME = "Qwen/Qwen3-0.6B"
|
||||
|
||||
# Device 0 hosts the inference worker, device 1 hosts the trainer.
|
||||
INFERENCE_WORLD_SIZE = 1
|
||||
TRAINER_DEVICE_INDEX = INFERENCE_WORLD_SIZE
|
||||
|
||||
PROMPTS = [
|
||||
"Hello, my name is",
|
||||
"The capital of France is",
|
||||
]
|
||||
|
||||
# HTTP timeouts (seconds). Weight broadcast can take a while for large models.
|
||||
INIT_TIMEOUT = 120
|
||||
UPDATE_TIMEOUT = 300
|
||||
CONTROL_TIMEOUT = 60
|
||||
|
||||
|
||||
def _log(message: str) -> None:
|
||||
"""Flushed log so step markers show up immediately even when stdout is piped."""
|
||||
print(f"[trainer] {message}", flush=True)
|
||||
|
||||
|
||||
def _build_trainer_model(device_index: int):
|
||||
"""Build the trainer-side model without downloading the checkpoint weights.
|
||||
|
||||
By default the model is instantiated from the architecture config with random
|
||||
weights (no ``model.safetensors`` download required); only the tiny config is
|
||||
read, which the server already fetches. Its ``named_parameters`` carry the
|
||||
same names/shapes/dtypes as the real checkpoint, so the HCCL broadcast +
|
||||
layerwise reload path is exercised exactly as with real weights.
|
||||
|
||||
Set ``WEIGHT_TRANSFER_TEST_MODEL=/path/to/checkpoint`` to broadcast real
|
||||
weights from a local directory instead.
|
||||
"""
|
||||
device = f"npu:{device_index}"
|
||||
override_path = os.getenv("WEIGHT_TRANSFER_TEST_MODEL")
|
||||
if override_path:
|
||||
_log(f"loading real trainer weights from {override_path}")
|
||||
model = AutoModelForCausalLM.from_pretrained(override_path, dtype=torch.bfloat16)
|
||||
else:
|
||||
_log("building trainer model from config with random weights (download-free)")
|
||||
config = AutoConfig.from_pretrained(MODEL_NAME, trust_remote_code=True)
|
||||
model = AutoModelForCausalLM.from_config(config)
|
||||
model = model.to(device=device, dtype=torch.bfloat16)
|
||||
return model
|
||||
|
||||
|
||||
def _post(server: RemoteOpenAIServer, route: str, *, json=None, timeout=CONTROL_TIMEOUT):
|
||||
response = requests.post(server.url_for(route), json=json, timeout=timeout)
|
||||
response.raise_for_status()
|
||||
return response
|
||||
|
||||
|
||||
class _BackgroundPost(threading.Thread):
|
||||
"""Run an HTTP POST in a thread while keeping its exception visible.
|
||||
|
||||
The trainer side blocks on collective HCCL ops, so the matching server-side
|
||||
RPC must run concurrently. If that RPC fails, swallowing the exception would
|
||||
deadlock the trainer forever; instead we record it and surface it on join().
|
||||
"""
|
||||
|
||||
def __init__(self, server: RemoteOpenAIServer, route: str, *, json=None, timeout=CONTROL_TIMEOUT):
|
||||
super().__init__(daemon=True)
|
||||
self._server = server
|
||||
self._route = route
|
||||
self._json = json
|
||||
self._timeout = timeout
|
||||
self.error: BaseException | None = None
|
||||
|
||||
def run(self) -> None:
|
||||
try:
|
||||
_post(self._server, self._route, json=self._json, timeout=self._timeout)
|
||||
_log(f"background POST /{self._route} done")
|
||||
except BaseException as exc: # noqa: BLE001 - re-raised on join via raise_if_failed
|
||||
self.error = exc
|
||||
_log(f"background POST /{self._route} FAILED: {exc!r}")
|
||||
|
||||
def raise_if_failed(self) -> None:
|
||||
if self.error is not None:
|
||||
raise RuntimeError(f"server-side /{self._route} failed") from self.error
|
||||
|
||||
|
||||
def _generate(client, model, prompts):
|
||||
completions = []
|
||||
for prompt in prompts:
|
||||
response = client.completions.create(
|
||||
model=model,
|
||||
prompt=prompt,
|
||||
max_tokens=16,
|
||||
temperature=0,
|
||||
)
|
||||
completions.append(response.choices[0].text)
|
||||
return completions
|
||||
|
||||
|
||||
def _collect_weight_metadata(train_model):
|
||||
"""Collect parameter metadata and size the packed buffer for broadcasting."""
|
||||
names: list[str] = []
|
||||
dtype_names: list[str] = []
|
||||
shapes: list[list[int]] = []
|
||||
max_tensor_bytes = 0
|
||||
for name, parameter in train_model.named_parameters():
|
||||
names.append(name)
|
||||
dtype_names.append(str(parameter.dtype).split(".")[-1])
|
||||
shapes.append(list(parameter.shape))
|
||||
tensor_bytes = parameter.numel() * parameter.element_size()
|
||||
max_tensor_bytes = max(max_tensor_bytes, tensor_bytes)
|
||||
|
||||
# Keep the 1 GiB default unless a single tensor needs more (+128 MiB headroom).
|
||||
packed_buffer_size_bytes = max(max_tensor_bytes + 128 * 2**20, 2**30)
|
||||
return names, dtype_names, shapes, packed_buffer_size_bytes
|
||||
|
||||
|
||||
def _has_lifecycle_endpoints(server: RemoteOpenAIServer) -> bool:
|
||||
"""Detect whether the server exposes the vLLM-main start/finish endpoints.
|
||||
|
||||
On vLLM main, ``/start_weight_update`` and ``/finish_weight_update`` drive
|
||||
the layerwise reload lifecycle. On v0.20.2 these endpoints do not exist and
|
||||
``update_weights`` is self-contained, so a probe returns 404.
|
||||
"""
|
||||
try:
|
||||
response = requests.post(
|
||||
server.url_for("start_weight_update"),
|
||||
json={"is_checkpoint_format": True},
|
||||
timeout=CONTROL_TIMEOUT,
|
||||
)
|
||||
except requests.RequestException:
|
||||
return False
|
||||
if response.status_code == 404:
|
||||
return False
|
||||
response.raise_for_status()
|
||||
return True
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
torch.npu.device_count() < 2,
|
||||
reason="HCCL weight transfer e2e test requires at least 2 NPUs.",
|
||||
)
|
||||
def test_hccl_weight_transfer_updates_server_weights():
|
||||
port = get_open_port()
|
||||
server_args = [
|
||||
"--enforce-eager",
|
||||
"--load-format",
|
||||
"dummy",
|
||||
"--weight-transfer-config",
|
||||
'{"backend": "nccl"}',
|
||||
"--tensor-parallel-size",
|
||||
str(INFERENCE_WORLD_SIZE),
|
||||
"--max-model-len",
|
||||
"1024",
|
||||
"--gpu-memory-utilization",
|
||||
"0.6",
|
||||
"--port",
|
||||
str(port),
|
||||
"--trust-remote-code",
|
||||
]
|
||||
# The dev-mode endpoints (/init_weight_transfer_engine, /update_weights,
|
||||
# /pause, /resume, ...) are only registered when VLLM_SERVER_DEV_MODE=1.
|
||||
# Pin the server to NPU 0 so the trainer can own NPU 1 exclusively.
|
||||
env_dict = {
|
||||
"VLLM_SERVER_DEV_MODE": "1",
|
||||
"ASCEND_RT_VISIBLE_DEVICES": "0",
|
||||
"VLLM_ASCEND_ENABLE_NZ": "0",
|
||||
}
|
||||
|
||||
_log(f"starting server on port {port} (device 0, dummy weights) ...")
|
||||
with RemoteOpenAIServer(
|
||||
MODEL_NAME,
|
||||
vllm_serve_args=server_args,
|
||||
# Health check, OpenAI client and control-plane requests all target this
|
||||
# host; use loopback explicitly so they reach the local server directly.
|
||||
server_host="127.0.0.1",
|
||||
server_port=port,
|
||||
env_dict=env_dict,
|
||||
auto_port=False,
|
||||
) as server:
|
||||
client = server.get_client()
|
||||
|
||||
# 1) Baseline generation with dummy weights (expected to be nonsense).
|
||||
_log("generating baseline outputs (dummy weights) ...")
|
||||
outputs_before = _generate(client, MODEL_NAME, PROMPTS)
|
||||
_log(f"outputs BEFORE weight update: {outputs_before}")
|
||||
|
||||
# 2) Build the trainer model on the trainer NPU (download-free by default).
|
||||
_log(f"preparing trainer model on npu:{TRAINER_DEVICE_INDEX} ...")
|
||||
torch.npu.set_device(TRAINER_DEVICE_INDEX)
|
||||
train_model = _build_trainer_model(TRAINER_DEVICE_INDEX)
|
||||
_log("trainer model ready")
|
||||
|
||||
# Import after the server is up so the HCCL engine plugin is registered.
|
||||
from vllm_ascend.distributed.weight_transfer.hccl_engine import (
|
||||
HCCLTrainerSendWeightsArgs,
|
||||
HCCLWeightTransferEngine,
|
||||
)
|
||||
|
||||
master_address = get_ip()
|
||||
master_port = get_open_port()
|
||||
rank_offset = 1
|
||||
world_size = INFERENCE_WORLD_SIZE + 1 # workers + trainer
|
||||
|
||||
# 3) Build the HCCL process group on both sides. The server side blocks
|
||||
# until the trainer connects, so kick it off in a background thread.
|
||||
init_info = dict(
|
||||
master_address=master_address,
|
||||
master_port=master_port,
|
||||
rank_offset=rank_offset,
|
||||
world_size=world_size,
|
||||
)
|
||||
_log(f"HCCL rendezvous at {master_address}:{master_port} (world_size={world_size}) ...")
|
||||
init_thread = _BackgroundPost(
|
||||
server,
|
||||
"init_weight_transfer_engine",
|
||||
json={"init_info": init_info},
|
||||
timeout=INIT_TIMEOUT,
|
||||
)
|
||||
init_thread.start()
|
||||
model_update_group = HCCLWeightTransferEngine.trainer_init(
|
||||
dict(
|
||||
master_address=master_address,
|
||||
master_port=master_port,
|
||||
world_size=world_size,
|
||||
),
|
||||
)
|
||||
_log("trainer_init returned, waiting for server init RPC ...")
|
||||
init_thread.join()
|
||||
init_thread.raise_if_failed()
|
||||
_log("HCCL process group established")
|
||||
|
||||
# 4) Pause generation and start the weight update lifecycle. On vLLM
|
||||
# main this probe also performs the actual /start_weight_update call,
|
||||
# so we must not call it again below.
|
||||
_post(server, "pause")
|
||||
use_lifecycle = _has_lifecycle_endpoints(server)
|
||||
_log(f"paused; lifecycle endpoints available: {use_lifecycle}")
|
||||
|
||||
names, dtype_names, shapes, packed_buffer_size_bytes = _collect_weight_metadata(train_model)
|
||||
update_info = dict(
|
||||
names=names,
|
||||
dtype_names=dtype_names,
|
||||
shapes=shapes,
|
||||
packed=True,
|
||||
packed_buffer_size_bytes=packed_buffer_size_bytes,
|
||||
)
|
||||
if not use_lifecycle:
|
||||
# v0.20.2 folds the layerwise reload lifecycle into update_weights.
|
||||
update_info["is_checkpoint_format"] = True
|
||||
|
||||
# update_weights blocks on the server while it waits for HCCL broadcasts,
|
||||
# so run it in a thread while the trainer produces the data.
|
||||
_log(f"broadcasting {len(names)} tensors via HCCL (packed) ...")
|
||||
update_thread = _BackgroundPost(
|
||||
server,
|
||||
"update_weights",
|
||||
json={"update_info": update_info},
|
||||
timeout=UPDATE_TIMEOUT,
|
||||
)
|
||||
update_thread.start()
|
||||
|
||||
trainer_args = HCCLTrainerSendWeightsArgs(
|
||||
group=model_update_group,
|
||||
packed=True,
|
||||
packed_buffer_size_bytes=packed_buffer_size_bytes,
|
||||
)
|
||||
HCCLWeightTransferEngine.trainer_send_weights(
|
||||
iterator=train_model.named_parameters(),
|
||||
trainer_args=trainer_args,
|
||||
)
|
||||
_log("trainer finished sending weights, waiting for server update RPC ...")
|
||||
update_thread.join()
|
||||
update_thread.raise_if_failed()
|
||||
_log("weight broadcast complete")
|
||||
|
||||
# 5) Finalize the lifecycle and resume generation.
|
||||
if use_lifecycle:
|
||||
_post(server, "finish_weight_update")
|
||||
_post(server, "resume")
|
||||
|
||||
# 6) Generation after the broadcast weights are loaded.
|
||||
outputs_after = _generate(client, MODEL_NAME, PROMPTS)
|
||||
_log(f"outputs AFTER weight update: {outputs_after}")
|
||||
|
||||
# Reaching here means the full HCCL transfer pipeline succeeded: every
|
||||
# control-plane RPC raised on a non-2xx response and each background POST
|
||||
# re-raised on join(). The broadcast weights differ from the server's dummy
|
||||
# init, so the served model must now produce different generations.
|
||||
assert outputs_after != outputs_before, "server weights did not change after HCCL transfer"
|
||||
38
tests/e2e/pull_request/two_card/test_moe_routing_replay.py
Normal file
38
tests/e2e/pull_request/two_card/test_moe_routing_replay.py
Normal file
@@ -0,0 +1,38 @@
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
from vllm.sampling_params import RequestOutputKind
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
MODELS = [
|
||||
"Qwen/Qwen3.5-35B-A3B",
|
||||
"Qwen/Qwen3-30B-A3B",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@patch.dict(os.environ, {"OMP_NUM_THREADS": "1"})
|
||||
def test_qwen3_moe_routing_replay(model):
|
||||
prompts = [
|
||||
"Hello, please introduce yourself.",
|
||||
]
|
||||
with VllmRunner(
|
||||
model,
|
||||
tensor_parallel_size=2,
|
||||
enable_expert_parallel=True,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
distributed_executor_backend="mp",
|
||||
enable_return_routed_experts=True,
|
||||
async_scheduling=False,
|
||||
) as vllm_model:
|
||||
sampling_params = SamplingParams(
|
||||
max_tokens=5, temperature=0.8, top_p=0.95, output_kind=RequestOutputKind.FINAL_ONLY
|
||||
)
|
||||
inputs = vllm_model.get_inputs(prompts=prompts)
|
||||
outputs = vllm_model.model.generate(prompts=inputs, sampling_params=sampling_params)
|
||||
assert outputs[0].finished
|
||||
assert len(outputs[0].outputs[0].text) > 0
|
||||
assert outputs[0].outputs[0].routed_experts.size > 0
|
||||
77
tests/e2e/pull_request/two_card/test_offline_weight_load.py
Normal file
77
tests/e2e/pull_request/two_card/test_offline_weight_load.py
Normal file
@@ -0,0 +1,77 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
"""
|
||||
Run `pytest tests/e2e/pull_request/two_card/test_offline_weight_load.py`.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import wait_until_npu_memory_free
|
||||
|
||||
MODELS = ["Qwen/Qwen3-30B-A3B"]
|
||||
REPO_ROOT = Path(__file__).resolve().parents[4]
|
||||
EXTERNAL_LAUNCHER_SCRIPT = REPO_ROOT / "examples" / "offline_external_launcher.py"
|
||||
|
||||
|
||||
@pytest.mark.skip("fix me, unstable, timeout")
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_NZ": "0"})
|
||||
@wait_until_npu_memory_free(0.7)
|
||||
def test_qwen3_offline_load_and_sleepmode_tp2(model):
|
||||
env = os.environ.copy()
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(EXTERNAL_LAUNCHER_SCRIPT),
|
||||
"--model",
|
||||
model,
|
||||
"--tp-size",
|
||||
"2",
|
||||
"--node-size",
|
||||
"1",
|
||||
"--node-rank",
|
||||
"0",
|
||||
"--proc-per-node",
|
||||
"2",
|
||||
"--trust-remote-code",
|
||||
"--enable-sleep-mode",
|
||||
"--temperature",
|
||||
"0",
|
||||
"--model-weight-gib",
|
||||
"0.8",
|
||||
]
|
||||
|
||||
print(f"Running subprocess: {' '.join(cmd)}")
|
||||
proc = subprocess.run(
|
||||
cmd,
|
||||
env=env,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
timeout=600,
|
||||
)
|
||||
output = proc.stdout.decode(errors="ignore")
|
||||
|
||||
print(output)
|
||||
|
||||
assert "Generated text:" in output
|
||||
assert "Sleep and wake up successfully!!" in output
|
||||
assert proc.returncode == 0
|
||||
90
tests/e2e/pull_request/two_card/test_prefix_caching.py
Normal file
90
tests/e2e/pull_request/two_card/test_prefix_caching.py
Normal file
@@ -0,0 +1,90 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
"""Compare the with and without prefix caching."""
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.model_utils import check_outputs_equal
|
||||
|
||||
MODELS = [
|
||||
# for MHA
|
||||
"Qwen/Qwen3-8B",
|
||||
# for MLA
|
||||
"deepseek-ai/DeepSeek-V2-Lite-Chat",
|
||||
]
|
||||
|
||||
# A prompt containing a large markdown table. The table is randomly generated by GPT-4.
|
||||
# ruff: noqa: E501
|
||||
LONG_PROMPT = (
|
||||
"You are a helpful assistant in recognizes the content of tables in markdown format. Here is a table as follows.\n# Table\n"
|
||||
+ """
|
||||
| ID | Name | Age | Occupation | Country | Email | Phone Number | Address |
|
||||
|-----|---------------|-----|---------------|---------------|------------------------|----------------|------------------------------|
|
||||
| 1 | John Doe | 29 | Engineer | USA | john.doe@example.com | 555-1234 | 123 Elm St, Springfield, IL |
|
||||
| 2 | Jane Smith | 34 | Doctor | Canada | jane.smith@example.com | 555-5678 | 456 Oak St, Toronto, ON |
|
||||
| 3 | Alice Johnson | 27 | Teacher | UK | alice.j@example.com | 555-8765 | 789 Pine St, London, UK |
|
||||
| 4 | Bob Brown | 45 | Artist | Australia | bob.b@example.com | 555-4321 | 321 Maple St, Sydney, NSW |
|
||||
| 5 | Carol White | 31 | Scientist | New Zealand | carol.w@example.com | 555-6789 | 654 Birch St, Wellington, NZ |
|
||||
| 6 | Dave Green | 28 | Lawyer | Ireland | dave.g@example.com | 555-3456 | 987 Cedar St, Dublin, IE |
|
||||
| 7 | Emma Black | 40 | Musician | USA | emma.b@example.com | 555-1111 | 246 Ash St, New York, NY |
|
||||
| 8 | Frank Blue | 37 | Chef | Canada | frank.b@example.com | 555-2222 | 135 Spruce St, Vancouver, BC |
|
||||
| 9 | Grace Yellow | 50 | Engineer | UK | grace.y@example.com | 555-3333 | 864 Fir St, Manchester, UK |
|
||||
| 10 | Henry Violet | 32 | Artist | Australia | henry.v@example.com | 555-4444 | 753 Willow St, Melbourne, VIC|
|
||||
| 11 | Irene Orange | 26 | Scientist | New Zealand | irene.o@example.com | 555-5555 | 912 Poplar St, Auckland, NZ |
|
||||
| 12 | Jack Indigo | 38 | Teacher | Ireland | jack.i@example.com | 555-6666 | 159 Elm St, Cork, IE |
|
||||
| 13 | Karen Red | 41 | Lawyer | USA | karen.r@example.com | 555-7777 | 357 Cedar St, Boston, MA |
|
||||
| 14 | Leo Brown | 30 | Chef | Canada | leo.b@example.com | 555-8888 | 246 Oak St, Calgary, AB |
|
||||
| 15 | Mia Green | 33 | Musician | UK | mia.g@example.com | 555-9999 | 975 Pine St, Edinburgh, UK |
|
||||
| 16 | Noah Yellow | 29 | Doctor | Australia | noah.y@example.com | 555-0000 | 864 Birch St, Brisbane, QLD |
|
||||
| 17 | Olivia Blue | 35 | Engineer | New Zealand | olivia.b@example.com | 555-1212 | 753 Maple St, Hamilton, NZ |
|
||||
| 18 | Peter Black | 42 | Artist | Ireland | peter.b@example.com | 555-3434 | 912 Fir St, Limerick, IE |
|
||||
| 19 | Quinn White | 28 | Scientist | USA | quinn.w@example.com | 555-5656 | 159 Willow St, Seattle, WA |
|
||||
| 20 | Rachel Red | 31 | Teacher | Canada | rachel.r@example.com | 555-7878 | 357 Poplar St, Ottawa, ON |
|
||||
| 21 | Steve Green | 44 | Lawyer | UK | steve.g@example.com | 555-9090 | 753 Elm St, Birmingham, UK |
|
||||
| 22 | Tina Blue | 36 | Musician | Australia | tina.b@example.com | 555-1213 | 864 Cedar St, Perth, WA |
|
||||
| 23 | Umar Black | 39 | Chef | New Zealand | umar.b@example.com | 555-3435 | 975 Spruce St, Christchurch, NZ|
|
||||
| 24 | Victor Yellow | 43 | Engineer | Ireland | victor.y@example.com | 555-5657 | 246 Willow St, Galway, IE |
|
||||
| 25 | Wendy Orange | 27 | Artist | USA | wendy.o@example.com | 555-7879 | 135 Elm St, Denver, CO |
|
||||
| 26 | Xavier Green | 34 | Scientist | Canada | xavier.g@example.com | 555-9091 | 357 Oak St, Montreal, QC |
|
||||
| 27 | Yara Red | 41 | Teacher | UK | yara.r@example.com | 555-1214 | 975 Pine St, Leeds, UK |
|
||||
| 28 | Zack Blue | 30 | Lawyer | Australia | zack.b@example.com | 555-3436 | 135 Birch St, Adelaide, SA |
|
||||
| 29 | Amy White | 33 | Musician | New Zealand | amy.w@example.com | 555-5658 | 159 Maple St, Wellington, NZ |
|
||||
| 30 | Ben Black | 38 | Chef | Ireland | ben.b@example.com | 555-7870 | 246 Fir St, Waterford, IE |
|
||||
"""
|
||||
)
|
||||
|
||||
INPUT_PROMPTS = [
|
||||
LONG_PROMPT + "Question: what is the age of John Doe? Your answer: The age of John Doe is ",
|
||||
LONG_PROMPT + "Question: what is the age of Zack Blue? Your answer: The age of Zack Blue is ",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [50])
|
||||
def test_models_prefix_cache_tp2(model: str, max_tokens: int) -> None:
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=2048,
|
||||
tensor_parallel_size=2,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
gpu_memory_utilization=0.7,
|
||||
) as vllm_model:
|
||||
prefix_cache_output = vllm_model.generate_greedy(INPUT_PROMPTS, max_tokens)
|
||||
|
||||
with VllmRunner(
|
||||
model,
|
||||
enable_prefix_caching=False,
|
||||
max_model_len=2048,
|
||||
tensor_parallel_size=2,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
gpu_memory_utilization=0.7,
|
||||
) as vllm_model:
|
||||
vllm_output = vllm_model.generate_greedy(INPUT_PROMPTS, max_tokens)
|
||||
|
||||
check_outputs_equal(
|
||||
outputs_0_lst=vllm_output,
|
||||
outputs_1_lst=prefix_cache_output,
|
||||
name_0="vllm_output",
|
||||
name_1="prefix_cache_output",
|
||||
)
|
||||
82
tests/e2e/pull_request/two_card/test_qwen3_30b_a3b.py
Normal file
82
tests/e2e/pull_request/two_card/test_qwen3_30b_a3b.py
Normal file
@@ -0,0 +1,82 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
from vllm.utils.network_utils import get_open_port
|
||||
|
||||
from tests.e2e.conftest import RemoteOpenAIServer, wait_until_npu_memory_free
|
||||
from vllm_ascend.utils import vllm_version_is
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
not vllm_version_is("0.23.0"),
|
||||
reason="broken on main, fix me.",
|
||||
)
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_moe_tp_ep_eplb_full_decode_only():
|
||||
"""Verify MoE serving with TP, EP, EPLB, and full decode only."""
|
||||
model = "Qwen/Qwen3-30B-A3B"
|
||||
port = get_open_port()
|
||||
env_dict = {
|
||||
"DYNAMIC_EPLB": "true",
|
||||
"HCCL_BUFFSIZE": "1024",
|
||||
}
|
||||
server_args = [
|
||||
"--max_model_len",
|
||||
"8192",
|
||||
"--tensor_parallel_size",
|
||||
"2",
|
||||
"--enable_expert_parallel",
|
||||
"--port",
|
||||
str(port),
|
||||
"--compilation-config",
|
||||
json.dumps({"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [8]}),
|
||||
"--additional-config",
|
||||
json.dumps(
|
||||
{
|
||||
"eplb_config": {
|
||||
"dynamic_eplb": True,
|
||||
"expert_heat_collection_interval": 100,
|
||||
"algorithm_execution_interval": 20,
|
||||
"num_redundant_experts": 2,
|
||||
}
|
||||
}
|
||||
),
|
||||
]
|
||||
|
||||
with RemoteOpenAIServer(model, server_args, server_port=port, auto_port=False, env_dict=env_dict) as server:
|
||||
response = requests.post(
|
||||
server.url_for("v1", "completions"),
|
||||
json={
|
||||
"model": model,
|
||||
"prompt": "What is deeplearning?",
|
||||
"max_tokens": 400,
|
||||
"temperature": 0.0,
|
||||
"top_p": 1.0,
|
||||
"n": 1,
|
||||
},
|
||||
timeout=600,
|
||||
)
|
||||
response.raise_for_status()
|
||||
output = response.json()
|
||||
|
||||
assert output["choices"][0]["text"]
|
||||
44
tests/e2e/pull_request/two_card/test_qwen3_5_35b_a3b_w8a8.py
Normal file
44
tests/e2e/pull_request/two_card/test_qwen3_5_35b_a3b_w8a8.py
Normal file
@@ -0,0 +1,44 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
from tests.e2e.conftest import VllmRunner, wait_until_npu_memory_free
|
||||
|
||||
EXAMPLE_PROMPTS = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "1024"})
|
||||
@wait_until_npu_memory_free()
|
||||
def test_qwen3_5_35b_a3b_w8a8_tp2_without_ep():
|
||||
with VllmRunner(
|
||||
"Eco-Tech/Qwen3.5-35B-A3B-w8a8-mtp",
|
||||
max_model_len=4096,
|
||||
tensor_parallel_size=2,
|
||||
enable_expert_parallel=False,
|
||||
quantization="ascend",
|
||||
gpu_memory_utilization=0.9,
|
||||
distributed_executor_backend="mp",
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
) as vllm_model:
|
||||
outputs = vllm_model.generate_greedy(EXAMPLE_PROMPTS, max_tokens=5)
|
||||
|
||||
assert outputs[0][1]
|
||||
103
tests/e2e/pull_request/two_card/test_qwen3_6_27b_fia.py
Normal file
103
tests/e2e/pull_request/two_card/test_qwen3_6_27b_fia.py
Normal file
@@ -0,0 +1,103 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
from vllm.assets.image import ImageAsset
|
||||
|
||||
from tests.e2e.conftest import VllmRunner, qwen_prompt, wait_until_npu_memory_free
|
||||
|
||||
MODEL = "Qwen/Qwen3.6-27B"
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "1024"})
|
||||
@wait_until_npu_memory_free()
|
||||
def test_qwen3_6_27b_multimodel_fia_eager():
|
||||
"""Verify multimodal generation with FIA op and eager mode."""
|
||||
image = ImageAsset("cherry_blossom").pil_image.convert("RGB")
|
||||
questions = [
|
||||
"What is the content of this image?",
|
||||
"Describe the content of this image in detail.",
|
||||
"What's in the image?",
|
||||
"Where is this image taken?",
|
||||
]
|
||||
|
||||
images = [image] * len(questions)
|
||||
prompts = qwen_prompt(questions)
|
||||
|
||||
with VllmRunner(
|
||||
MODEL,
|
||||
max_model_len=4096,
|
||||
tensor_parallel_size=2,
|
||||
language_model_only=False,
|
||||
gpu_memory_utilization=0.9,
|
||||
limit_mm_per_prompt={"image": 1},
|
||||
mm_processor_kwargs={
|
||||
"min_pixels": 28 * 28,
|
||||
"max_pixels": 1280 * 28 * 28,
|
||||
"fps": 1,
|
||||
},
|
||||
enforce_eager=True,
|
||||
) as vllm_model:
|
||||
outputs = vllm_model.generate_greedy(
|
||||
prompts=prompts,
|
||||
images=images,
|
||||
max_tokens=64,
|
||||
)
|
||||
|
||||
assert outputs[0][1]
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"HCCL_BUFFSIZE": "1024"})
|
||||
@wait_until_npu_memory_free()
|
||||
def test_qwen3_6_27b_multimodel_fia_acl_graph():
|
||||
"""Verify multimodal generation with FIA op and FULL_AND_PIECEWISE graph mode."""
|
||||
image = ImageAsset("cherry_blossom").pil_image.convert("RGB")
|
||||
questions = [
|
||||
"What is the content of this image?",
|
||||
"Describe the content of this image in detail.",
|
||||
"What's in the image?",
|
||||
"Where is this image taken?",
|
||||
]
|
||||
|
||||
images = [image] * len(questions)
|
||||
prompts = qwen_prompt(questions)
|
||||
|
||||
with VllmRunner(
|
||||
MODEL,
|
||||
max_model_len=4096,
|
||||
tensor_parallel_size=2,
|
||||
language_model_only=False,
|
||||
gpu_memory_utilization=0.9,
|
||||
limit_mm_per_prompt={"image": 1},
|
||||
mm_processor_kwargs={
|
||||
"min_pixels": 28 * 28,
|
||||
"max_pixels": 1280 * 28 * 28,
|
||||
"fps": 1,
|
||||
},
|
||||
compilation_config={
|
||||
"cudagraph_mm_encoder": True,
|
||||
"cudagraph_capture_sizes": [1],
|
||||
"encoder_cudagraph_token_budgets": [128, 256, 512, 1024, 1536, 2048, 2560, 3072, 3584, 4096],
|
||||
},
|
||||
) as vllm_model:
|
||||
outputs = vllm_model.generate_greedy(
|
||||
prompts=prompts,
|
||||
images=images,
|
||||
max_tokens=64,
|
||||
)
|
||||
|
||||
assert outputs[0][1]
|
||||
77
tests/e2e/pull_request/two_card/test_qwen3_moe_eplb.py
Normal file
77
tests/e2e/pull_request/two_card/test_qwen3_moe_eplb.py
Normal file
@@ -0,0 +1,77 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
import json
|
||||
|
||||
import openai
|
||||
import pytest
|
||||
from vllm.utils.network_utils import get_open_port
|
||||
|
||||
from tests.e2e.conftest import RemoteOpenAIServer
|
||||
from vllm_ascend.utils import vllm_version_is
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
not vllm_version_is("0.23.0"),
|
||||
reason="broken on main, fix me.",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_qwen3_moe_w8a8_distributed_tp2_ep_dynamic_eplb():
|
||||
model = "vllm-ascend/Qwen3-30B-A3B-W8A8"
|
||||
port = get_open_port()
|
||||
compilation_config = json.dumps({"cudagraph_capture_sizes": [8]})
|
||||
server_args = [
|
||||
"--max_model_len",
|
||||
"8192",
|
||||
"--tensor_parallel_size",
|
||||
"2",
|
||||
"--enable_expert_parallel",
|
||||
"--quantization",
|
||||
"ascend",
|
||||
"--port",
|
||||
str(port),
|
||||
"--compilation-config",
|
||||
compilation_config,
|
||||
]
|
||||
env_dict = {"HCCL_BUFFSIZE": "1024"}
|
||||
with RemoteOpenAIServer(model, server_args, server_port=port, auto_port=False, env_dict=env_dict) as server:
|
||||
client = server.get_async_client()
|
||||
batch = await client.completions.create(
|
||||
model=model, prompt="What is deeplearning?", max_tokens=400, temperature=0, top_p=1.0, n=1
|
||||
)
|
||||
gt_choices: list[openai.types.CompletionChoice] = batch.choices
|
||||
|
||||
env_dict.update({"DYNAMIC_EPLB": "true"})
|
||||
additional_config = {
|
||||
"eplb_config": {
|
||||
"dynamic_eplb": True,
|
||||
"expert_heat_collection_interval": 100,
|
||||
"algorithm_execution_interval": 20,
|
||||
"num_redundant_experts": 2,
|
||||
"eplb_policy_type": 2,
|
||||
}
|
||||
}
|
||||
server_args.extend(["--additional-config", json.dumps(additional_config)])
|
||||
with RemoteOpenAIServer(model, server_args, server_port=port, auto_port=False, env_dict=env_dict) as server:
|
||||
client = server.get_async_client()
|
||||
batch = await client.completions.create(
|
||||
model=model, prompt="What is deeplearning?", max_tokens=400, temperature=0, top_p=1.0, n=1
|
||||
)
|
||||
eplb_choices: list[openai.types.CompletionChoice] = batch.choices
|
||||
assert gt_choices[0].text == eplb_choices[0].text, f"{gt_choices[0].text=} \n {eplb_choices[0].text=}"
|
||||
97
tests/e2e/pull_request/two_card/test_qwen3_performance.py
Normal file
97
tests/e2e/pull_request/two_card/test_qwen3_performance.py
Normal file
@@ -0,0 +1,97 @@
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
from typing import Any
|
||||
|
||||
import openai
|
||||
import pytest
|
||||
from vllm.utils.network_utils import get_open_port
|
||||
|
||||
from tests.e2e.conftest import RemoteOpenAIServer
|
||||
from tools.vllm_bench import run_vllm_bench_case
|
||||
|
||||
MODELS = [
|
||||
"Qwen/Qwen3-8B",
|
||||
]
|
||||
|
||||
prompts = [
|
||||
"San Francisco is a",
|
||||
]
|
||||
|
||||
api_keyword_args = {
|
||||
"max_tokens": 10,
|
||||
}
|
||||
|
||||
vllm_bench_cases = {
|
||||
"dataset-name": "random",
|
||||
"num_prompts": 500,
|
||||
"request_rate": 20,
|
||||
"random_input_len": 128,
|
||||
"max_concurrency": 40,
|
||||
"random_output_len": 100,
|
||||
"temperature": 0.0,
|
||||
}
|
||||
|
||||
# NOTE: Any changes for the baseline throughput should be approved by team members.
|
||||
# The origin baseline: 1600.0. For some uncertain reasons, the throughput is decreased to 1514.0
|
||||
baseline_throughput = 1514.0 # baseline throughput for Qwen3-8B, measured with num_prompts=500
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="Temporarily skipped due to flaky failures, pending investigation.")
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.asyncio
|
||||
async def test_models(model: str) -> None:
|
||||
port = get_open_port()
|
||||
env_dict = {
|
||||
"TASK_QUEUE_ENABLE": "1",
|
||||
"HCCL_OP_EXPANSION_MODE": "AIV",
|
||||
}
|
||||
server_args = [
|
||||
"--distributed-executor-backend",
|
||||
"mp",
|
||||
"--tensor-parallel-size",
|
||||
"1",
|
||||
"--port",
|
||||
str(port),
|
||||
"--max-model-len",
|
||||
"5500",
|
||||
"--max-num-batched-tokens",
|
||||
"40960",
|
||||
"--compilation-config",
|
||||
'{"cudagraph_mode": "FULL_DECODE_ONLY"}',
|
||||
"--additional-config",
|
||||
'{"pa_shape_list":[48,64,72,80],"weight_prefetch_config":{"enabled":true}}',
|
||||
"--block-size",
|
||||
"128",
|
||||
"--trust-remote-code",
|
||||
"--gpu-memory-utilization",
|
||||
"0.9",
|
||||
]
|
||||
|
||||
request_keyword_args: dict[str, Any] = {
|
||||
**api_keyword_args,
|
||||
}
|
||||
with RemoteOpenAIServer(model, server_args, server_port=port, env_dict=env_dict, auto_port=False) as server:
|
||||
client = server.get_async_client()
|
||||
batch = await client.completions.create(
|
||||
model=model,
|
||||
prompt=prompts,
|
||||
**request_keyword_args,
|
||||
)
|
||||
choices: list[openai.types.CompletionChoice] = batch.choices
|
||||
assert choices[0].text, "empty response"
|
||||
# vllm bench test
|
||||
run_vllm_bench_case(model, port, vllm_bench_cases, baseline_throughput)
|
||||
@@ -0,0 +1,66 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
from vllm.assets.image import ImageAsset
|
||||
|
||||
from tests.e2e.conftest import VllmRunner, qwen_prompt, wait_until_npu_memory_free
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_multimodal_reasoning_pp_full_decode_only():
|
||||
"""Verify multimodal generation with PP and full decode only."""
|
||||
image = ImageAsset("cherry_blossom").pil_image.convert("RGB")
|
||||
|
||||
img_questions = [
|
||||
"What is the content of this image?",
|
||||
"Describe the content of this image in detail.",
|
||||
"What's in the image?",
|
||||
"Where is this image taken?",
|
||||
]
|
||||
|
||||
images = [image] * len(img_questions)
|
||||
prompts = qwen_prompt(img_questions)
|
||||
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3-VL-30B-A3B-Instruct",
|
||||
pipeline_parallel_size=2,
|
||||
max_model_len=4096,
|
||||
max_num_batched_tokens=1024,
|
||||
gpu_memory_utilization=0.9,
|
||||
limit_mm_per_prompt={"image": 1},
|
||||
mm_processor_kwargs={
|
||||
"min_pixels": 28 * 28,
|
||||
"max_pixels": 1280 * 28 * 28,
|
||||
"fps": 1,
|
||||
},
|
||||
hf_overrides={"text_config": {"architectures": ["Qwen3MoeForCausalLM"]}},
|
||||
compilation_config={
|
||||
"cudagraph_mode": "FULL_DECODE_ONLY",
|
||||
"cudagraph_capture_sizes": [1, 2, 4, 8],
|
||||
},
|
||||
) as vllm_model:
|
||||
outputs = vllm_model.generate_greedy(
|
||||
prompts=prompts,
|
||||
images=images,
|
||||
max_tokens=64,
|
||||
)
|
||||
|
||||
assert len(outputs) == len(prompts)
|
||||
|
||||
for _, output_str in outputs:
|
||||
assert output_str, "Generated output should not be empty."
|
||||
473
tests/e2e/pull_request/two_card/test_sequence_parallelism_moe.py
Normal file
473
tests/e2e/pull_request/two_card/test_sequence_parallelism_moe.py
Normal file
@@ -0,0 +1,473 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Two-card e2e tests for SequenceParallelismMoePass patterns:
|
||||
# - MiddleLayerAllgatherAddRMSNormPattern (all_gather + slice + RMSNorm)
|
||||
# - Qwen3VLMiddleLayerAllgatherAddRMSNormPattern (all_gather + slice + add + RMSNorm)
|
||||
# - AllGatherChunkNoOpPattern (all_gather + sequence_parallel_chunk_impl -> identity)
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
|
||||
import queue
|
||||
import traceback
|
||||
from collections.abc import Callable, Generator
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import vllm.config
|
||||
from vllm.compilation.passes.fx_utils import OpOverload
|
||||
from vllm.config import ModelConfig, VllmConfig
|
||||
from vllm.distributed import (
|
||||
get_tensor_model_parallel_world_size,
|
||||
get_tp_group,
|
||||
init_distributed_environment,
|
||||
tensor_model_parallel_all_gather,
|
||||
)
|
||||
from vllm.distributed.parallel_state import (
|
||||
destroy_distributed_environment,
|
||||
destroy_model_parallel,
|
||||
initialize_model_parallel,
|
||||
)
|
||||
from vllm.utils.system_utils import update_environment_variables
|
||||
|
||||
import vllm_ascend.ops.register_custom_ops # noqa
|
||||
from tests.e2e.pull_request.one_card.compile.backend import TestBackend as CompileTestBackend
|
||||
from vllm_ascend.compilation.passes.sequence_parallelism_moe import (
|
||||
SequenceParallelismMoePass,
|
||||
)
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
|
||||
MASTER_PORT = 29500
|
||||
WORLD_SIZE = 2
|
||||
WORKER_READY = "__ready__"
|
||||
WORKER_STOP = "__stop__"
|
||||
WORKER_RESULT_TIMEOUT_S = 180
|
||||
WORKER_JOIN_TIMEOUT_S = 30
|
||||
|
||||
|
||||
class BaseAllGatherRMSNormModel(nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
hidden_size: int,
|
||||
dtype: torch.dtype,
|
||||
eps: float = 1e-6,
|
||||
device: str = "npu",
|
||||
):
|
||||
super().__init__()
|
||||
self.eps = eps
|
||||
self.norm_w = torch.randn(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def _all_gather_sliced(self, x: torch.Tensor, num_tokens_helper: torch.Tensor) -> torch.Tensor:
|
||||
num_tokens = num_tokens_helper.shape[0]
|
||||
activated = torch.relu(x)
|
||||
gathered = tensor_model_parallel_all_gather(activated, 0)
|
||||
return gathered[:num_tokens]
|
||||
|
||||
@staticmethod
|
||||
def ops_in_model_after() -> tuple[tuple[OpOverload, int], ...]:
|
||||
return (
|
||||
(torch.ops.vllm.all_gather.default, 1),
|
||||
(torch.ops._C_ascend.npu_add_rms_norm_bias.default, 1),
|
||||
(torch.ops.vllm.maybe_chunk_residual.default, 1),
|
||||
)
|
||||
|
||||
|
||||
class AllGatherRMSNormModel(BaseAllGatherRMSNormModel):
|
||||
def forward(
|
||||
self,
|
||||
x: torch.Tensor,
|
||||
residual: torch.Tensor,
|
||||
num_tokens_helper: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
sliced = self._all_gather_sliced(x, num_tokens_helper)
|
||||
rms_out = torch.ops._C_ascend.npu_add_rms_norm_bias(sliced, residual, self.norm_w, None, self.eps)
|
||||
return rms_out[0]
|
||||
|
||||
@staticmethod
|
||||
def ops_in_model_before() -> tuple[tuple[OpOverload, int], ...]:
|
||||
return (
|
||||
(torch.ops.vllm.all_gather.default, 1),
|
||||
(torch.ops._C_ascend.npu_add_rms_norm_bias.default, 1),
|
||||
)
|
||||
|
||||
|
||||
class Qwen3VLAllGatherRMSNormModel(BaseAllGatherRMSNormModel):
|
||||
"""Exercises Qwen3VLMiddleLayerAllgatherAddRMSNormPattern (all_gather + slice + add + RMSNorm)."""
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x: torch.Tensor,
|
||||
residual: torch.Tensor,
|
||||
num_tokens_helper: torch.Tensor,
|
||||
deepstack_input_embeds: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
sliced = self._all_gather_sliced(x, num_tokens_helper)
|
||||
add_ = sliced + deepstack_input_embeds
|
||||
result, _, residual = torch.ops._C_ascend.npu_add_rms_norm_bias(add_, residual, self.norm_w, None, self.eps)
|
||||
# Keep the residual output live so the traced graph preserves the full pattern.
|
||||
result = result - residual
|
||||
return result
|
||||
|
||||
@staticmethod
|
||||
def ops_in_model_before() -> tuple[tuple[OpOverload, int], ...]:
|
||||
return (
|
||||
(torch.ops.vllm.all_gather.default, 1),
|
||||
(torch.ops.aten.add.Tensor, 1),
|
||||
(torch.ops._C_ascend.npu_add_rms_norm_bias.default, 1),
|
||||
)
|
||||
|
||||
|
||||
class AllGatherChunkNoOpModel(nn.Module):
|
||||
"""Exercises AllGatherChunkNoOpPattern (all_gather + sequence_parallel_chunk_impl -> identity)."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
|
||||
def forward(self, x: torch.Tensor) -> torch.Tensor:
|
||||
z = torch.relu(x)
|
||||
gathered = tensor_model_parallel_all_gather(z, 0)
|
||||
return torch.ops.vllm.sequence_parallel_chunk_impl(gathered)
|
||||
|
||||
@staticmethod
|
||||
def ops_in_model_before() -> tuple[tuple[OpOverload, int], ...]:
|
||||
return (
|
||||
(torch.ops.vllm.all_gather.default, 1),
|
||||
(torch.ops.vllm.sequence_parallel_chunk_impl.default, 1),
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def ops_in_model_after() -> tuple[tuple[OpOverload, int], ...]:
|
||||
return (
|
||||
(torch.ops.vllm.all_gather.default, 0),
|
||||
(torch.ops.vllm.sequence_parallel_chunk_impl.default, 0),
|
||||
)
|
||||
|
||||
|
||||
def _build_all_gather_rms_norm_inputs(
|
||||
batch_size: int,
|
||||
seq_len: int,
|
||||
hidden_size: int,
|
||||
dtype: torch.dtype,
|
||||
tp_size: int,
|
||||
) -> tuple[torch.Tensor, ...]:
|
||||
local_tokens = batch_size * seq_len
|
||||
num_tokens = local_tokens * tp_size
|
||||
x = torch.randn(local_tokens, hidden_size, dtype=dtype)
|
||||
residual = torch.zeros(num_tokens, hidden_size, dtype=dtype)
|
||||
num_tokens_helper = torch.empty(num_tokens, device=x.device, dtype=dtype)
|
||||
return (x, residual, num_tokens_helper)
|
||||
|
||||
|
||||
def _build_qwen3vl_inputs(
|
||||
batch_size: int,
|
||||
seq_len: int,
|
||||
hidden_size: int,
|
||||
dtype: torch.dtype,
|
||||
tp_size: int,
|
||||
) -> tuple[torch.Tensor, ...]:
|
||||
x, residual, num_tokens_helper = _build_all_gather_rms_norm_inputs(
|
||||
batch_size=batch_size,
|
||||
seq_len=seq_len,
|
||||
hidden_size=hidden_size,
|
||||
dtype=dtype,
|
||||
tp_size=tp_size,
|
||||
)
|
||||
deepstack = torch.randn(num_tokens_helper.shape[0], hidden_size, dtype=dtype)
|
||||
return (x, residual, num_tokens_helper, deepstack)
|
||||
|
||||
|
||||
def _build_allgather_chunk_noop_inputs(
|
||||
batch_size: int,
|
||||
seq_len: int,
|
||||
hidden_size: int,
|
||||
dtype: torch.dtype,
|
||||
tp_size: int,
|
||||
) -> tuple[torch.Tensor, ...]:
|
||||
del tp_size
|
||||
local_tokens = batch_size * seq_len
|
||||
x = torch.randn(local_tokens, hidden_size, dtype=dtype)
|
||||
return (x,)
|
||||
|
||||
|
||||
def _create_all_gather_rms_norm_model(
|
||||
hidden_size: int,
|
||||
dtype: torch.dtype,
|
||||
eps: float,
|
||||
device: str,
|
||||
) -> nn.Module:
|
||||
return AllGatherRMSNormModel(hidden_size=hidden_size, dtype=dtype, eps=eps, device=device)
|
||||
|
||||
|
||||
def _create_qwen3vl_model(
|
||||
hidden_size: int,
|
||||
dtype: torch.dtype,
|
||||
eps: float,
|
||||
device: str,
|
||||
) -> nn.Module:
|
||||
return Qwen3VLAllGatherRMSNormModel(hidden_size=hidden_size, dtype=dtype, eps=eps, device=device)
|
||||
|
||||
|
||||
def _create_allgather_chunk_noop_model(
|
||||
hidden_size: int,
|
||||
dtype: torch.dtype,
|
||||
eps: float,
|
||||
device: str,
|
||||
) -> nn.Module:
|
||||
del hidden_size, dtype, eps, device
|
||||
return AllGatherChunkNoOpModel()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PatternTestCase:
|
||||
model_factory: Any
|
||||
input_builder: Any
|
||||
dynamic_input_indices: tuple[int, ...]
|
||||
pre_pass_expected_counts_factory: Any
|
||||
post_pass_expected_counts_factory: Any
|
||||
|
||||
|
||||
PATTERN_TEST_CASES = {
|
||||
"middle_layer_allgather_add_rms_norm": PatternTestCase(
|
||||
model_factory=_create_all_gather_rms_norm_model,
|
||||
input_builder=_build_all_gather_rms_norm_inputs,
|
||||
dynamic_input_indices=(0, 2),
|
||||
pre_pass_expected_counts_factory=AllGatherRMSNormModel.ops_in_model_before,
|
||||
post_pass_expected_counts_factory=AllGatherRMSNormModel.ops_in_model_after,
|
||||
),
|
||||
"qwen3vl_middle_layer_allgather_add_rms_norm": PatternTestCase(
|
||||
model_factory=_create_qwen3vl_model,
|
||||
input_builder=_build_qwen3vl_inputs,
|
||||
dynamic_input_indices=(0, 2, 3),
|
||||
pre_pass_expected_counts_factory=Qwen3VLAllGatherRMSNormModel.ops_in_model_before,
|
||||
post_pass_expected_counts_factory=Qwen3VLAllGatherRMSNormModel.ops_in_model_after,
|
||||
),
|
||||
"allgather_chunk_noop": PatternTestCase(
|
||||
model_factory=_create_allgather_chunk_noop_model,
|
||||
input_builder=_build_allgather_chunk_noop_inputs,
|
||||
dynamic_input_indices=(0,),
|
||||
pre_pass_expected_counts_factory=AllGatherChunkNoOpModel.ops_in_model_before,
|
||||
post_pass_expected_counts_factory=AllGatherChunkNoOpModel.ops_in_model_after,
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _assert_op_counts(
|
||||
backend: CompileTestBackend,
|
||||
expected_counts: tuple[tuple[OpOverload, int], ...],
|
||||
before: bool = False,
|
||||
) -> None:
|
||||
for op, expected_count in expected_counts:
|
||||
actual_count = backend.op_count(op, before=before)
|
||||
stage = "before" if before else "after"
|
||||
assert actual_count == expected_count, (
|
||||
f"op {stage} pass: {op} expected {expected_count}, but got {actual_count}"
|
||||
)
|
||||
|
||||
|
||||
def _run_single_pattern_case(
|
||||
local_rank: int,
|
||||
case_name: str,
|
||||
vllm_config: VllmConfig,
|
||||
tp_size: int,
|
||||
batch_size: int = 8,
|
||||
seq_len: int = 16,
|
||||
hidden_size: int = 16,
|
||||
dtype: torch.dtype = torch.bfloat16,
|
||||
eps: float = 1e-5,
|
||||
) -> None:
|
||||
case = PATTERN_TEST_CASES[case_name]
|
||||
sp_moe_pass = SequenceParallelismMoePass(vllm_config)
|
||||
backend = CompileTestBackend(custom_passes=[sp_moe_pass])
|
||||
model = case.model_factory(
|
||||
hidden_size=hidden_size,
|
||||
dtype=dtype,
|
||||
eps=eps,
|
||||
device=f"npu:{local_rank}",
|
||||
)
|
||||
inputs = case.input_builder(
|
||||
batch_size=batch_size,
|
||||
seq_len=seq_len,
|
||||
hidden_size=hidden_size,
|
||||
dtype=dtype,
|
||||
tp_size=tp_size,
|
||||
)
|
||||
for dynamic_input_index in case.dynamic_input_indices:
|
||||
torch._dynamo.mark_dynamic(inputs[dynamic_input_index], 0)
|
||||
|
||||
unfused = model(*inputs)
|
||||
compiled = torch.compile(model, backend=backend)
|
||||
fused = compiled(*inputs)
|
||||
assert unfused.shape == fused.shape
|
||||
|
||||
assert sp_moe_pass.matched_count == 1
|
||||
_assert_op_counts(backend, case.pre_pass_expected_counts_factory(), before=True)
|
||||
_assert_op_counts(backend, case.post_pass_expected_counts_factory())
|
||||
|
||||
|
||||
def _run_sequence_parallelism_moe_test(
|
||||
local_rank: int,
|
||||
world_size: int,
|
||||
master_port: int,
|
||||
command_queue: Any,
|
||||
result_queue: Any,
|
||||
batch_size: int = 8,
|
||||
seq_len: int = 16,
|
||||
hidden_size: int = 16,
|
||||
dtype: torch.dtype = torch.bfloat16,
|
||||
eps: float = 1e-5,
|
||||
) -> None:
|
||||
torch.npu.set_device(local_rank)
|
||||
torch.set_default_device(f"npu:{local_rank}")
|
||||
torch.set_default_dtype(dtype)
|
||||
torch.manual_seed(0)
|
||||
|
||||
update_environment_variables(
|
||||
{
|
||||
"RANK": str(local_rank),
|
||||
"LOCAL_RANK": str(local_rank),
|
||||
"WORLD_SIZE": str(world_size),
|
||||
"MASTER_ADDR": "127.0.0.1",
|
||||
"MASTER_PORT": str(master_port),
|
||||
}
|
||||
)
|
||||
|
||||
vllm_config = VllmConfig(model_config=ModelConfig(dtype=dtype))
|
||||
|
||||
try:
|
||||
with vllm.config.set_current_vllm_config(vllm_config):
|
||||
init_distributed_environment(
|
||||
world_size=world_size,
|
||||
rank=local_rank,
|
||||
local_rank=local_rank,
|
||||
backend="hccl",
|
||||
)
|
||||
initialize_model_parallel(tensor_model_parallel_size=world_size)
|
||||
|
||||
if not enable_custom_op():
|
||||
raise RuntimeError("vllm_ascend custom ops are not available")
|
||||
|
||||
_ = get_tp_group().unique_name
|
||||
tp_size = get_tensor_model_parallel_world_size()
|
||||
result_queue.put((WORKER_READY, local_rank, "ok", ""))
|
||||
|
||||
while True:
|
||||
case_name = command_queue.get()
|
||||
if case_name == WORKER_STOP:
|
||||
return
|
||||
|
||||
try:
|
||||
_run_single_pattern_case(
|
||||
local_rank=local_rank,
|
||||
case_name=case_name,
|
||||
vllm_config=vllm_config,
|
||||
tp_size=tp_size,
|
||||
batch_size=batch_size,
|
||||
seq_len=seq_len,
|
||||
hidden_size=hidden_size,
|
||||
dtype=dtype,
|
||||
eps=eps,
|
||||
)
|
||||
except Exception:
|
||||
result_queue.put((case_name, local_rank, "error", traceback.format_exc()))
|
||||
else:
|
||||
result_queue.put((case_name, local_rank, "ok", ""))
|
||||
finally:
|
||||
destroy_model_parallel()
|
||||
destroy_distributed_environment()
|
||||
if torch.distributed.is_initialized():
|
||||
torch.distributed.destroy_process_group()
|
||||
|
||||
|
||||
def _worker_entrypoint(
|
||||
local_rank: int,
|
||||
world_size: int,
|
||||
master_port: int,
|
||||
command_queue: Any,
|
||||
result_queue: Any,
|
||||
) -> None:
|
||||
try:
|
||||
_run_sequence_parallelism_moe_test(
|
||||
local_rank=local_rank,
|
||||
world_size=world_size,
|
||||
master_port=master_port,
|
||||
command_queue=command_queue,
|
||||
result_queue=result_queue,
|
||||
)
|
||||
except Exception:
|
||||
result_queue.put((WORKER_READY, local_rank, "error", traceback.format_exc()))
|
||||
|
||||
|
||||
def _wait_for_worker_reports(
|
||||
result_queue: Any,
|
||||
case_name: str,
|
||||
expected_reports: int,
|
||||
) -> None:
|
||||
errors = []
|
||||
for _ in range(expected_reports):
|
||||
try:
|
||||
reported_case_name, local_rank, status, payload = result_queue.get(timeout=WORKER_RESULT_TIMEOUT_S)
|
||||
except queue.Empty as exc:
|
||||
raise TimeoutError(f"Timed out waiting for worker reports for {case_name}") from exc
|
||||
|
||||
assert reported_case_name == case_name, f"Expected worker report for {case_name}, but got {reported_case_name}"
|
||||
if status != "ok":
|
||||
errors.append(f"rank {local_rank}:\n{payload}")
|
||||
|
||||
if errors:
|
||||
raise AssertionError("\n\n".join(errors))
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def sequence_parallelism_moe_workers() -> Generator[Callable[[str], None], None, None]:
|
||||
ctx = torch.multiprocessing.get_context("spawn")
|
||||
command_queues = [ctx.Queue() for _ in range(WORLD_SIZE)]
|
||||
result_queue = ctx.Queue()
|
||||
workers = []
|
||||
|
||||
for local_rank in range(WORLD_SIZE):
|
||||
worker = ctx.Process(
|
||||
target=_worker_entrypoint,
|
||||
args=(local_rank, WORLD_SIZE, MASTER_PORT, command_queues[local_rank], result_queue),
|
||||
)
|
||||
worker.start()
|
||||
workers.append(worker)
|
||||
|
||||
try:
|
||||
_wait_for_worker_reports(result_queue, WORKER_READY, WORLD_SIZE)
|
||||
|
||||
def _run_case(case_name: str) -> None:
|
||||
for command_queue in command_queues:
|
||||
command_queue.put(case_name)
|
||||
_wait_for_worker_reports(result_queue, case_name, WORLD_SIZE)
|
||||
|
||||
yield _run_case
|
||||
finally:
|
||||
for command_queue in command_queues:
|
||||
command_queue.put(WORKER_STOP)
|
||||
for worker in workers:
|
||||
worker.join(timeout=WORKER_JOIN_TIMEOUT_S)
|
||||
if worker.is_alive():
|
||||
worker.terminate()
|
||||
worker.join()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("case_name", tuple(PATTERN_TEST_CASES), ids=tuple(PATTERN_TEST_CASES))
|
||||
def test_sequence_parallelism_moe_patterns(
|
||||
sequence_parallelism_moe_workers: Callable[[str], None], case_name: str
|
||||
) -> None:
|
||||
sequence_parallelism_moe_workers(case_name)
|
||||
59
tests/e2e/pull_request/two_card/test_shared_expert_dp.py
Normal file
59
tests/e2e/pull_request/two_card/test_shared_expert_dp.py
Normal file
@@ -0,0 +1,59 @@
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import wait_until_npu_memory_free
|
||||
from tests.e2e.pull_request.utils import compare_logprobs
|
||||
|
||||
MODELS = [
|
||||
"deepseek-ai/DeepSeek-V2-Lite",
|
||||
]
|
||||
|
||||
PROMPTS = [
|
||||
"Hello, my name is",
|
||||
"The capital of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
|
||||
@wait_until_npu_memory_free(0.7)
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
def test_deepseek_v2_lite_enable_shared_expert_dp_tp2(model: str, monkeypatch) -> None:
|
||||
# FlashComm v1 / shared-expert-DP require HCCL_OP_EXPANSION_MODE to be unset.
|
||||
monkeypatch.delenv("HCCL_OP_EXPANSION_MODE", raising=False)
|
||||
|
||||
# FlashComm1 + shared-expert-DP must stay numerically consistent with the
|
||||
# plain eager baseline. `additional_config` is excluded from the baseline
|
||||
# by compare_logprobs, so the baseline runs without either flag.
|
||||
shared_expert_dp_config = {
|
||||
"enable_flashcomm1": True,
|
||||
"enable_shared_expert_dp": True,
|
||||
}
|
||||
|
||||
# Eager mode: FlashComm1 + shared-expert-DP vs eager baseline.
|
||||
compare_logprobs(
|
||||
runner_kwargs={
|
||||
"model_name": model,
|
||||
"max_model_len": 1024,
|
||||
"enforce_eager": True,
|
||||
"tensor_parallel_size": 2,
|
||||
"enable_expert_parallel": True,
|
||||
"additional_config": shared_expert_dp_config,
|
||||
},
|
||||
prompts=PROMPTS,
|
||||
)
|
||||
|
||||
# ACLGraph (FULL_DECODE_ONLY): FlashComm1 + shared-expert-DP vs eager baseline.
|
||||
compare_logprobs(
|
||||
runner_kwargs={
|
||||
"model_name": model,
|
||||
"max_model_len": 1024,
|
||||
"tensor_parallel_size": 2,
|
||||
"enable_expert_parallel": True,
|
||||
"compilation_config": {
|
||||
"cudagraph_capture_sizes": [1, 4, 8, 16],
|
||||
"cudagraph_mode": "FULL_DECODE_ONLY",
|
||||
},
|
||||
"additional_config": shared_expert_dp_config,
|
||||
},
|
||||
prompts=PROMPTS,
|
||||
)
|
||||
61
tests/e2e/pull_request/two_card/test_sp_pass.py
Normal file
61
tests/e2e/pull_request/two_card/test_sp_pass.py
Normal file
@@ -0,0 +1,61 @@
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.model_utils import check_outputs_equal
|
||||
|
||||
MODELS = [
|
||||
"Qwen/Qwen3-VL-2B-Instruct",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
def test_qwen3_vl_sp_tp2(model: str) -> None:
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The capital of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
sampling_params = SamplingParams(max_tokens=10, temperature=0.0)
|
||||
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
tensor_parallel_size=2,
|
||||
compilation_config={
|
||||
"cudagraph_capture_sizes": [2, 4],
|
||||
"cudagraph_mode": "FULL_DECODE_ONLY",
|
||||
"pass_config": {"enable_sp": False},
|
||||
},
|
||||
additional_config={"ascend_compilation_config": {"enable_npugraph_ex": False}},
|
||||
) as runner:
|
||||
no_sp_outputs = runner.model.generate(prompts, sampling_params)
|
||||
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
tensor_parallel_size=2,
|
||||
compilation_config={
|
||||
"cudagraph_capture_sizes": [2, 4],
|
||||
"cudagraph_mode": "FULL_DECODE_ONLY",
|
||||
"pass_config": {"enable_sp": True, "sp_min_token_num": 10},
|
||||
},
|
||||
additional_config={"ascend_compilation_config": {"enable_npugraph_ex": False}},
|
||||
) as runner:
|
||||
sp_outputs = runner.model.generate(prompts, sampling_params)
|
||||
|
||||
no_sp_outputs_list = []
|
||||
for output in no_sp_outputs:
|
||||
no_sp_outputs_list.append((output.outputs[0].index, output.outputs[0].text))
|
||||
|
||||
sp_outputs_list = []
|
||||
for output in sp_outputs:
|
||||
sp_outputs_list.append((output.outputs[0].index, output.outputs[0].text))
|
||||
|
||||
check_outputs_equal(
|
||||
outputs_0_lst=no_sp_outputs_list,
|
||||
outputs_1_lst=sp_outputs_list,
|
||||
name_0="no_sp_outputs",
|
||||
name_1="sp_outputs",
|
||||
)
|
||||
Reference in New Issue
Block a user