0
tests/e2e/pull_request/one_card/_310p/__init__.py
Normal file
0
tests/e2e/pull_request/one_card/_310p/__init__.py
Normal file
@@ -0,0 +1,41 @@
|
||||
import pytest
|
||||
import torch
|
||||
from modelscope import snapshot_download # type: ignore[import-untyped]
|
||||
from transformers import AutoModelForSequenceClassification
|
||||
|
||||
from tests.e2e.conftest import HfRunner, VllmRunner
|
||||
|
||||
|
||||
@pytest.mark.skip("Probabilistic failure, need fix")
|
||||
def test_qwen_pooling_classify_correctness() -> None:
|
||||
model_name = snapshot_download("Howeee/Qwen2.5-1.5B-apeach")
|
||||
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is what",
|
||||
]
|
||||
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
max_model_len=1024,
|
||||
enforce_eager=True,
|
||||
dtype="float16",
|
||||
gpu_memory_utilization=0.6,
|
||||
) as vllm_runner:
|
||||
vllm_outputs = vllm_runner.classify(prompts)
|
||||
|
||||
with HfRunner(
|
||||
model_name,
|
||||
dtype="float16",
|
||||
model_kwargs={"attn_implementation": "eager"},
|
||||
auto_cls=AutoModelForSequenceClassification,
|
||||
) as hf_runner:
|
||||
hf_outputs = hf_runner.classify(prompts)
|
||||
|
||||
for hf_output, vllm_output in zip(hf_outputs, vllm_outputs):
|
||||
hf_output = torch.tensor(hf_output)
|
||||
vllm_output = torch.tensor(vllm_output)
|
||||
assert torch.allclose(hf_output, vllm_output, 1e-2)
|
||||
173
tests/e2e/pull_request/one_card/_310p/test_dense_model_310p.py
Normal file
173
tests/e2e/pull_request/one_card/_310p/test_dense_model_310p.py
Normal file
@@ -0,0 +1,173 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
|
||||
|
||||
from tests.e2e.conftest import VllmRunner, wait_until_npu_memory_free
|
||||
from tests.e2e.model_utils import check_outputs_equal
|
||||
|
||||
QWEN3_5_PREFIX_MAMBA_PROMPT = (
|
||||
"You are reading a compact synthetic operations ledger. "
|
||||
"Use only the rows below when answering the final question.\n"
|
||||
+ "\n".join(
|
||||
f"Row {i}: route R{i:03d} moves cargo from zone {i % 11} to zone {(i * 7) % 13}; priority is {i % 5}."
|
||||
for i in range(64)
|
||||
)
|
||||
+ "\n"
|
||||
)
|
||||
|
||||
QWEN3_5_PREFIX_MAMBA_PROMPTS = [
|
||||
QWEN3_5_PREFIX_MAMBA_PROMPT + "Question: What route is listed in row 17? Answer briefly.",
|
||||
QWEN3_5_PREFIX_MAMBA_PROMPT + "Question: What priority is listed in row 42? Answer briefly.",
|
||||
]
|
||||
|
||||
|
||||
def _generate_qwen3_5_prefix_mamba_outputs(enable_prefix_caching: bool) -> list[tuple[list[int], str]]:
|
||||
outputs: list[tuple[list[int], str]] = []
|
||||
|
||||
if enable_prefix_caching:
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3.5-4B",
|
||||
tensor_parallel_size=1,
|
||||
enforce_eager=True,
|
||||
dtype="float16",
|
||||
max_model_len=2048,
|
||||
max_num_batched_tokens=2048,
|
||||
enable_prefix_caching=True,
|
||||
mamba_cache_mode="align",
|
||||
mamba_ssm_cache_dtype="float16",
|
||||
) as vllm_model:
|
||||
for prompt in QWEN3_5_PREFIX_MAMBA_PROMPTS:
|
||||
outputs.extend(vllm_model.generate_greedy([prompt], max_tokens=8))
|
||||
else:
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3.5-4B",
|
||||
tensor_parallel_size=1,
|
||||
enforce_eager=True,
|
||||
dtype="float16",
|
||||
max_model_len=2048,
|
||||
max_num_batched_tokens=2048,
|
||||
enable_prefix_caching=False,
|
||||
mamba_ssm_cache_dtype="float16",
|
||||
) as vllm_model:
|
||||
for prompt in QWEN3_5_PREFIX_MAMBA_PROMPTS:
|
||||
outputs.extend(vllm_model.generate_greedy([prompt], max_tokens=8))
|
||||
return outputs
|
||||
|
||||
|
||||
def test_qwen3_dense_tp1_fp16():
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
max_tokens = 5
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3-8B",
|
||||
tensor_parallel_size=1,
|
||||
enforce_eager=True,
|
||||
dtype="float16",
|
||||
max_model_len=16384,
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
|
||||
|
||||
@wait_until_npu_memory_free(0.7)
|
||||
def test_qwen3_dense_tp1_fp16_aclgraph():
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
] * 8
|
||||
max_tokens = 2
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3-8B",
|
||||
tensor_parallel_size=1,
|
||||
dtype="float16",
|
||||
max_num_seqs=16,
|
||||
max_model_len=16384,
|
||||
gpu_memory_utilization=0.80,
|
||||
additional_config={"ascend_compilation_config": {"fuse_norm_quant": False}},
|
||||
compilation_config={
|
||||
"cudagraph_mode": "FULL_DECODE_ONLY",
|
||||
"cudagraph_capture_sizes": [1, 2, 4, 8, 16],
|
||||
},
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
|
||||
|
||||
def test_qwen3_dense_tp1_w8a8():
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
max_tokens = 5
|
||||
with VllmRunner(
|
||||
"vllm-ascend/Qwen3-8B-W8A8",
|
||||
tensor_parallel_size=1,
|
||||
enforce_eager=True,
|
||||
dtype="float16",
|
||||
quantization="ascend",
|
||||
max_model_len=16384,
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
|
||||
|
||||
def test_qwen3_5_dense_tp1_fp16():
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
max_tokens = 5
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3.5-4B",
|
||||
tensor_parallel_size=1,
|
||||
enforce_eager=True,
|
||||
dtype="float16",
|
||||
max_model_len=16384,
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
|
||||
|
||||
@wait_until_npu_memory_free(0.7)
|
||||
def test_qwen3_5_dense_prefix_mamba_cache_tp1_fp16():
|
||||
prefix_cache_outputs = _generate_qwen3_5_prefix_mamba_outputs(enable_prefix_caching=True)
|
||||
no_prefix_cache_outputs = _generate_qwen3_5_prefix_mamba_outputs(enable_prefix_caching=False)
|
||||
|
||||
assert len(prefix_cache_outputs) == len(no_prefix_cache_outputs) == len(QWEN3_5_PREFIX_MAMBA_PROMPTS)
|
||||
check_outputs_equal(
|
||||
outputs_0_lst=no_prefix_cache_outputs,
|
||||
outputs_1_lst=prefix_cache_outputs,
|
||||
name_0="no_prefix_cache_outputs",
|
||||
name_1="prefix_cache_outputs",
|
||||
)
|
||||
|
||||
|
||||
@wait_until_npu_memory_free(0.7)
|
||||
def test_qwen3_5_dense_tp1_fp16_aclgraph():
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
] * 8
|
||||
max_tokens = 2
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3.5-4B",
|
||||
tensor_parallel_size=1,
|
||||
dtype="float16",
|
||||
max_num_seqs=16,
|
||||
max_model_len=16384,
|
||||
gpu_memory_utilization=0.80,
|
||||
additional_config={"ascend_compilation_config": {"fuse_norm_quant": False}},
|
||||
compilation_config={
|
||||
"cudagraph_mode": "FULL_DECODE_ONLY",
|
||||
"cudagraph_capture_sizes": [1, 2, 4, 8, 16],
|
||||
},
|
||||
mamba_ssm_cache_dtype="float16",
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
106
tests/e2e/pull_request/one_card/_310p/test_embedding_310p.py
Normal file
106
tests/e2e/pull_request/one_card/_310p/test_embedding_310p.py
Normal file
@@ -0,0 +1,106 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
|
||||
#
|
||||
import pytest
|
||||
from modelscope import snapshot_download # type: ignore[import-untyped]
|
||||
|
||||
from tests.e2e.conftest import HfRunner, VllmRunner
|
||||
from tests.e2e.utils import check_embeddings_close
|
||||
|
||||
MODELS = [
|
||||
"Qwen/Qwen3-Embedding-0.6B", # lasttoken
|
||||
"intfloat/multilingual-e5-small", # mean_tokens
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
def test_embed_models_correctness(model: str):
|
||||
queries = ["What is the capital of China?", "Explain gravity"]
|
||||
|
||||
model_name = snapshot_download(model)
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
max_model_len=512,
|
||||
enforce_eager=True,
|
||||
dtype="float16",
|
||||
gpu_memory_utilization=0.6,
|
||||
) as vllm_runner:
|
||||
vllm_outputs = vllm_runner.embed(queries)
|
||||
|
||||
with HfRunner(
|
||||
model_name,
|
||||
dtype="float16",
|
||||
is_sentence_transformer=True,
|
||||
) as hf_runner:
|
||||
hf_outputs = hf_runner.encode(queries)
|
||||
|
||||
check_embeddings_close(
|
||||
embeddings_0_lst=hf_outputs,
|
||||
embeddings_1_lst=vllm_outputs,
|
||||
name_0="hf",
|
||||
name_1="vllm",
|
||||
tol=1e-2,
|
||||
)
|
||||
|
||||
|
||||
def test_bge_m3_correctness():
|
||||
queries = ["What is the capital of China?", "Explain gravity"]
|
||||
|
||||
model_name = snapshot_download("BAAI/bge-m3")
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
max_model_len=1024,
|
||||
dtype="float16",
|
||||
cudagraph_capture_sizes=[512, 1024],
|
||||
additional_config={"ascend_compilation_config": {"fuse_norm_quant": False}},
|
||||
) as vllm_aclgraph_runner:
|
||||
vllm_aclgraph_outputs = vllm_aclgraph_runner.embed(queries)
|
||||
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
max_model_len=1024,
|
||||
dtype="float16",
|
||||
enforce_eager=True,
|
||||
) as vllm_runner:
|
||||
vllm_eager_outputs = vllm_runner.embed(queries)
|
||||
|
||||
with HfRunner(
|
||||
model_name,
|
||||
dtype="float16",
|
||||
is_sentence_transformer=True,
|
||||
) as hf_runner:
|
||||
hf_outputs = hf_runner.encode(queries)
|
||||
|
||||
check_embeddings_close(
|
||||
embeddings_0_lst=hf_outputs,
|
||||
embeddings_1_lst=vllm_eager_outputs,
|
||||
name_0="hf",
|
||||
name_1="vllm",
|
||||
tol=1e-2,
|
||||
)
|
||||
|
||||
check_embeddings_close(
|
||||
embeddings_0_lst=vllm_eager_outputs,
|
||||
embeddings_1_lst=vllm_aclgraph_outputs,
|
||||
name_0="eager",
|
||||
name_1="aclgraph",
|
||||
tol=1e-2,
|
||||
)
|
||||
87
tests/e2e/pull_request/one_card/_310p/test_scoring_310p.py
Normal file
87
tests/e2e/pull_request/one_card/_310p/test_scoring_310p.py
Normal file
@@ -0,0 +1,87 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
import pytest
|
||||
from modelscope import snapshot_download # type: ignore[import-untyped]
|
||||
|
||||
from tests.e2e.conftest import HfRunner, VllmRunner
|
||||
|
||||
CROSS_ENCODER_MODELS = [
|
||||
"BAAI/bge-reranker-v2-m3", # Roberta
|
||||
]
|
||||
|
||||
|
||||
TEXTS_1 = [
|
||||
"What is the capital of France?",
|
||||
"What is the capital of Germany?",
|
||||
]
|
||||
|
||||
TEXTS_2 = [
|
||||
"The capital of France is Paris.",
|
||||
"The capital of Germany is Berlin.",
|
||||
]
|
||||
|
||||
DTYPE = "float16"
|
||||
|
||||
|
||||
@pytest.fixture(scope="module", params=CROSS_ENCODER_MODELS)
|
||||
def model_name(request):
|
||||
yield snapshot_download(request.param)
|
||||
|
||||
|
||||
def test_cross_encoder_score_1_to_1(model_name):
|
||||
text_pair = [TEXTS_1[0], TEXTS_2[0]]
|
||||
|
||||
with HfRunner(model_name, dtype=DTYPE, is_cross_encoder=True) as hf_model:
|
||||
hf_outputs = hf_model.predict([text_pair]).tolist()
|
||||
|
||||
with VllmRunner(
|
||||
model_name, runner="pooling", dtype=DTYPE, max_model_len=1024, enforce_eager=True, gpu_memory_utilization=0.6
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(text_pair[0], text_pair[1])
|
||||
|
||||
assert len(vllm_outputs) == 1
|
||||
assert len(hf_outputs) == 1
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
|
||||
|
||||
def test_cross_encoder_score_1_to_N(model_name):
|
||||
text_pairs = [
|
||||
[TEXTS_1[0], TEXTS_2[0]],
|
||||
[TEXTS_1[0], TEXTS_2[1]],
|
||||
]
|
||||
|
||||
with HfRunner(model_name, dtype=DTYPE, is_cross_encoder=True) as hf_model:
|
||||
hf_outputs = hf_model.predict(text_pairs).tolist()
|
||||
|
||||
with VllmRunner(
|
||||
model_name, runner="pooling", dtype=DTYPE, max_model_len=1024, enforce_eager=True, gpu_memory_utilization=0.6
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(TEXTS_1[0], TEXTS_2)
|
||||
|
||||
assert len(vllm_outputs) == 2
|
||||
assert len(hf_outputs) == 2
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
assert hf_outputs[1] == pytest.approx(vllm_outputs[1], rel=0.01)
|
||||
|
||||
|
||||
def test_cross_encoder_score_N_to_N(model_name):
|
||||
text_pairs = [
|
||||
[TEXTS_1[0], TEXTS_2[0]],
|
||||
[TEXTS_1[1], TEXTS_2[1]],
|
||||
]
|
||||
|
||||
with HfRunner(model_name, dtype=DTYPE, is_cross_encoder=True) as hf_model:
|
||||
hf_outputs = hf_model.predict(text_pairs).tolist()
|
||||
|
||||
with VllmRunner(
|
||||
model_name, runner="pooling", dtype=DTYPE, max_model_len=1024, enforce_eager=True, gpu_memory_utilization=0.6
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(TEXTS_1, TEXTS_2)
|
||||
|
||||
assert len(vllm_outputs) == 2
|
||||
assert len(hf_outputs) == 2
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
assert hf_outputs[1] == pytest.approx(vllm_outputs[1], rel=0.01)
|
||||
@@ -0,0 +1,35 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
|
||||
def test_qwen3_5_mtp_tp1_eager():
|
||||
example_prompts = ["Hello, my name is"]
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3.5-4B",
|
||||
tensor_parallel_size=1,
|
||||
enforce_eager=True,
|
||||
dtype="float16",
|
||||
max_model_len=2048,
|
||||
mamba_ssm_cache_dtype="float16",
|
||||
speculative_config={
|
||||
"method": "qwen3_5_mtp",
|
||||
"num_speculative_tokens": 1,
|
||||
},
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens=8)
|
||||
34
tests/e2e/pull_request/one_card/_310p/test_vl_model_310p.py
Normal file
34
tests/e2e/pull_request/one_card/_310p/test_vl_model_310p.py
Normal file
@@ -0,0 +1,34 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
from tests.e2e.conftest import wait_until_npu_memory_free
|
||||
|
||||
current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
full_dir = os.path.dirname(os.path.dirname(current_dir))
|
||||
sys.path.insert(0, full_dir)
|
||||
|
||||
# ruff: noqa: E402
|
||||
from tests.e2e.pull_request.utils_310p import run_vl_model_test
|
||||
|
||||
|
||||
@wait_until_npu_memory_free(target_free_percentage=0.7)
|
||||
def test_qwen3_vl_8b_tp1_fp16():
|
||||
"""Qwen3-VL-8B single-card FP16 test"""
|
||||
run_vl_model_test(model_name="Qwen/Qwen3-VL-8B-Instruct", tensor_parallel_size=1, max_tokens=5)
|
||||
0
tests/e2e/pull_request/one_card/__init__.py
Normal file
0
tests/e2e/pull_request/one_card/__init__.py
Normal file
0
tests/e2e/pull_request/one_card/compile/__init__.py
Normal file
0
tests/e2e/pull_request/one_card/compile/__init__.py
Normal file
127
tests/e2e/pull_request/one_card/compile/backend.py
Normal file
127
tests/e2e/pull_request/one_card/compile/backend.py
Normal file
@@ -0,0 +1,127 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
from collections.abc import Callable, Sequence
|
||||
from copy import deepcopy
|
||||
from typing import Any
|
||||
|
||||
import torch.fx as fx
|
||||
from torch._inductor.decomposition import select_decomp_table
|
||||
from vllm.compilation.passes.fx_utils import OpOverload
|
||||
from vllm.config import get_current_vllm_config
|
||||
|
||||
from vllm_ascend.compilation.compiler_interface import compile_fx
|
||||
|
||||
|
||||
class TestBackend:
|
||||
"""
|
||||
A custom compilation backend for testing operator fusion passes.
|
||||
It applies the AddRMSNormQuantFusionPass during graph compilation and
|
||||
records the FX graph before and after the transformation.
|
||||
"""
|
||||
|
||||
def __init__(self, custom_passes: list[Any] | None = None):
|
||||
vllm_config = get_current_vllm_config()
|
||||
compile_config = vllm_config.compilation_config
|
||||
self.inductor_config = compile_config.inductor_compile_config
|
||||
self.inductor_config["graph_fusion_manager"] = self.post_pass
|
||||
self.custom_passes = custom_passes
|
||||
|
||||
# Placeholders to store FX graphs for verification
|
||||
self.graph_pre_pass = None
|
||||
self.graph_post_pass = None
|
||||
|
||||
def post_pass(self, graph: fx.Graph, runtime_shape: int | None = None) -> fx.Graph:
|
||||
"""
|
||||
Apply custom graph transformation passes.
|
||||
"""
|
||||
self.graph_pre_pass = deepcopy(graph)
|
||||
if self.custom_passes is not None:
|
||||
for pass_ in self.custom_passes:
|
||||
pass_(graph)
|
||||
self.graph_post_pass = deepcopy(graph)
|
||||
return graph
|
||||
|
||||
def compile(
|
||||
self,
|
||||
graph: fx.GraphModule,
|
||||
example_inputs: list[Any],
|
||||
compiler_config: dict[str, Any],
|
||||
runtime_shape: int | None = None,
|
||||
key: str | None = None,
|
||||
) -> tuple[Callable | None, Any | None]:
|
||||
"""
|
||||
Compile the FX graph using vLLM's Ascend compiler interface.
|
||||
Wraps the post-pass logic into the inner_compile callback.
|
||||
"""
|
||||
|
||||
def compile_inner(graph, example_inputs):
|
||||
current_pass_manager = compiler_config["graph_fusion_manager"]
|
||||
return current_pass_manager(graph, runtime_shape)
|
||||
|
||||
decompositions = select_decomp_table()
|
||||
compiled_fn = compile_fx(
|
||||
graph=graph,
|
||||
example_inputs=example_inputs,
|
||||
inner_compile=compile_inner,
|
||||
decompositions=decompositions,
|
||||
)
|
||||
return compiled_fn, None
|
||||
|
||||
def __call__(self, gm: fx.GraphModule, example_inputs: list[Any] | None):
|
||||
"""
|
||||
Make the backend callable by torch.compile().
|
||||
Returns a compiled executable function.
|
||||
"""
|
||||
assert example_inputs is not None
|
||||
compiled_fn, _ = self.compile(
|
||||
gm,
|
||||
example_inputs,
|
||||
compiler_config={"graph_fusion_manager": self.post_pass},
|
||||
runtime_shape=None,
|
||||
key=None,
|
||||
)
|
||||
return compiled_fn
|
||||
|
||||
def find_nodes_by_target(self, graph: fx.GraphModule, target: OpOverload) -> list[fx.Node]:
|
||||
"""Helper to find all FX nodes that call a specific operator."""
|
||||
return [node for node in graph.graph.nodes if hasattr(node, "target") and node.target == target]
|
||||
|
||||
def op_count(self, op: OpOverload, before: bool = False) -> int:
|
||||
"""Return the number of nodes that call the given operator."""
|
||||
graph = self.graph_pre_pass if before else self.graph_post_pass
|
||||
return len(self.find_nodes_by_target(graph, op))
|
||||
|
||||
def check_before_ops(self, ops: Sequence[OpOverload], fully_replaced: bool = True):
|
||||
"""
|
||||
Verify that the original (unfused) operators exist before the pass
|
||||
and are fully removed afterward (if fully_replaced=True).
|
||||
"""
|
||||
for op in ops:
|
||||
num_pre = len(self.find_nodes_by_target(self.graph_pre_pass, op))
|
||||
num_post = len(self.find_nodes_by_target(self.graph_post_pass, op))
|
||||
print(f"Op {op}: pre={num_pre}, post={num_post}")
|
||||
|
||||
assert num_pre > 0, f"Op {op} not found in pre-pass graph"
|
||||
if fully_replaced:
|
||||
assert num_post == 0, f"Unexpected op {op} in post-pass graph: {num_post} nodes remain"
|
||||
|
||||
def check_after_ops(self, ops: Sequence[OpOverload]):
|
||||
"""Verify that the fused operator appears in the transformed graph."""
|
||||
for op in ops:
|
||||
num_post = len(self.find_nodes_by_target(self.graph_post_pass, op))
|
||||
print(f"Op {op}: post={num_post}")
|
||||
assert num_post > 0, f"Op {op} not found in post-pass graph"
|
||||
@@ -0,0 +1,309 @@
|
||||
import copy
|
||||
|
||||
import npugraph_ex as nge
|
||||
import pytest
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch_npu
|
||||
import vllm.config
|
||||
from vllm.config import ModelConfig, VllmConfig
|
||||
from vllm.distributed import ensure_model_parallel_initialized, init_distributed_environment
|
||||
from vllm.utils.system_utils import update_environment_variables
|
||||
|
||||
from vllm_ascend.ascend_forward_context import set_ascend_forward_context
|
||||
from vllm_ascend.compilation.passes.norm_quant_fusion_pass import (
|
||||
AddRMSNormQuantPattern,
|
||||
AddRMSNormQuantPatternWithBias,
|
||||
AddRMSNormQuantSPPattern,
|
||||
AddRMSNormQuantSPPatternWithBias,
|
||||
)
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
|
||||
|
||||
def find_op(gm, op_default):
|
||||
return any(node.op == "call_function" and node.target == op_default for node in gm.graph.nodes)
|
||||
|
||||
|
||||
def create_pattern_wrapper(assert_func):
|
||||
original_func = nge.npu_fx_compiler._optimize_fx
|
||||
|
||||
def wrapper(gm, example_inputs=None, config=None):
|
||||
ret = original_func(gm, example_inputs, config)
|
||||
graph_after = copy.deepcopy(gm)
|
||||
assert_func(graph_after)
|
||||
return ret
|
||||
|
||||
return wrapper
|
||||
|
||||
|
||||
class ModelWithoutBias(nn.Module):
|
||||
"""
|
||||
A minimal test model that simulates the pattern:
|
||||
AddRMSNorm → Quantization (without bias)
|
||||
"""
|
||||
|
||||
def __init__(self, hidden_size: int, dtype: torch.bfloat16, eps: float = 1e-6, device="npu"):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.eps = eps
|
||||
self.rms_norm_weight = nn.Parameter(torch.randn(hidden_size, dtype=dtype, device=device))
|
||||
self.quant_scale = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_scale_reciprocal = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_offset = torch.zeros(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def forward(self, x):
|
||||
"""
|
||||
Forward pass:
|
||||
1. Perform npu_add_rms_norm
|
||||
2. Quantize the normalized output to int8
|
||||
Returns both quantized output and updated residual.
|
||||
"""
|
||||
residual = torch.zeros_like(x)
|
||||
|
||||
norm_output, _, new_residual = torch_npu.npu_add_rms_norm(x, residual, self.rms_norm_weight, self.eps)
|
||||
|
||||
quantized_output = torch.ops.vllm.quantize(
|
||||
norm_output, self.quant_scale, self.quant_scale_reciprocal, self.quant_offset
|
||||
)
|
||||
|
||||
return quantized_output, new_residual
|
||||
|
||||
|
||||
class ModelWithBias(nn.Module):
|
||||
"""
|
||||
A test model that simulates the pattern:
|
||||
AddRMSNorm → Add Bias → Quantization (with bias)
|
||||
"""
|
||||
|
||||
def __init__(self, hidden_size: int, dtype: torch.bfloat16, eps: float = 1e-6, device="npu"):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.eps = eps
|
||||
self.rms_norm_weight = nn.Parameter(torch.randn(hidden_size, dtype=dtype, device=device))
|
||||
self.bias = nn.Parameter(torch.randn(hidden_size, dtype=dtype, device=device))
|
||||
self.quant_scale = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_scale_reciprocal = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_offset = torch.zeros(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def forward(self, x):
|
||||
"""
|
||||
Forward pass:
|
||||
1. Perform npu_add_rms_norm
|
||||
2. Add bias
|
||||
3. Quantize to int8
|
||||
Returns both quantized output and updated residual.
|
||||
"""
|
||||
residual = torch.zeros_like(x)
|
||||
|
||||
norm_output, _, new_residual = torch_npu.npu_add_rms_norm(x, residual, self.rms_norm_weight, self.eps)
|
||||
|
||||
# Add bias
|
||||
norm_output_with_bias = norm_output + self.bias
|
||||
|
||||
quantized_output = torch.ops.vllm.quantize(
|
||||
norm_output_with_bias, self.quant_scale, self.quant_scale_reciprocal, self.quant_offset
|
||||
)
|
||||
|
||||
return quantized_output, new_residual
|
||||
|
||||
|
||||
class ModelSPWithoutBias(nn.Module):
|
||||
"""
|
||||
A minimal test model that simulates the pattern:
|
||||
AddRMSNorm → maybe_allgather → Quantization (without bias)
|
||||
"""
|
||||
|
||||
def __init__(self, hidden_size: int, dtype: torch.bfloat16, eps: float = 1e-6, device="npu"):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.eps = eps
|
||||
self.rms_norm_weight = nn.Parameter(torch.randn(hidden_size, dtype=dtype, device=device))
|
||||
self.quant_scale = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_scale_reciprocal = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_offset = torch.zeros(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def forward(self, x):
|
||||
"""
|
||||
Forward pass:
|
||||
1. Perform npu_add_rms_norm
|
||||
2. Perform a fake maybe_all_gather_and_maybe_unpad
|
||||
3. Quantize the normalized output to int8
|
||||
Returns both quantized output and updated residual.
|
||||
"""
|
||||
residual = torch.zeros_like(x)
|
||||
|
||||
norm_output, _, new_residual = torch_npu.npu_add_rms_norm(x, residual, self.rms_norm_weight, self.eps)
|
||||
|
||||
norm_output = torch.ops.vllm.maybe_all_gather_and_maybe_unpad(norm_output, True)
|
||||
|
||||
quantized_output = torch.ops.vllm.quantize(
|
||||
norm_output, self.quant_scale, self.quant_scale_reciprocal, self.quant_offset
|
||||
)
|
||||
|
||||
return quantized_output, new_residual
|
||||
|
||||
|
||||
class ModelSPWithBias(nn.Module):
|
||||
"""
|
||||
A minimal test model that simulates the pattern:
|
||||
AddRMSNorm → Add bias → maybe_allgather → Quantization (without bias)
|
||||
"""
|
||||
|
||||
def __init__(self, hidden_size: int, dtype: torch.bfloat16, eps: float = 1e-6, device="npu"):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.eps = eps
|
||||
self.rms_norm_weight = nn.Parameter(torch.randn(hidden_size, dtype=dtype, device=device))
|
||||
self.bias = nn.Parameter(torch.randn(hidden_size, dtype=dtype, device=device))
|
||||
self.quant_scale = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_scale_reciprocal = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_offset = torch.zeros(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def forward(self, x):
|
||||
"""
|
||||
Forward pass:
|
||||
1. Perform npu_add_rms_norm
|
||||
2. Add bias
|
||||
3. Perform a fake maybe_all_gather_and_maybe_unpad
|
||||
4. Quantize the normalized output to int8
|
||||
Returns both quantized output and updated residual.
|
||||
"""
|
||||
residual = torch.zeros_like(x)
|
||||
|
||||
norm_output, _, new_residual = torch_npu.npu_add_rms_norm(x, residual, self.rms_norm_weight, self.eps)
|
||||
|
||||
# Add bias
|
||||
norm_output_with_bias = norm_output + self.bias
|
||||
|
||||
norm_output_with_bias = torch.ops.vllm.maybe_all_gather_and_maybe_unpad(norm_output_with_bias, True)
|
||||
|
||||
quantized_output = torch.ops.vllm.quantize(
|
||||
norm_output_with_bias, self.quant_scale, self.quant_scale_reciprocal, self.quant_offset
|
||||
)
|
||||
|
||||
return quantized_output, new_residual
|
||||
|
||||
|
||||
def assert_addrmsnorm_quant(after_gm, expect_fused=True, use_bias=False, sp_enable=False):
|
||||
check_rules = [
|
||||
(torch.ops.npu.npu_add_rms_norm_quant.default, expect_fused),
|
||||
(torch.ops.npu.npu_add_rms_norm.default, not expect_fused),
|
||||
(torch.ops.npu.npu_quantize.default, not expect_fused),
|
||||
]
|
||||
if use_bias:
|
||||
check_rules.append((torch.ops.aten.add.Tensor, not expect_fused))
|
||||
if sp_enable:
|
||||
check_rules.append((torch.ops.vllm.maybe_all_gather_and_maybe_unpad.default, expect_fused))
|
||||
for torch_op, expect_exist in check_rules:
|
||||
found = find_op(after_gm, torch_op)
|
||||
if expect_exist:
|
||||
assert found, f"Expected operator '{torch_op}' but not find"
|
||||
else:
|
||||
assert not found, f"Not expected operator '{torch_op}' but find"
|
||||
|
||||
|
||||
_registered_patterns = set()
|
||||
|
||||
|
||||
def register_pattern_safe(pattern_class, vllm_config, eps, pattern_key):
|
||||
global _registered_patterns
|
||||
if pattern_key in _registered_patterns:
|
||||
print(f"Pattern {pattern_key} already registered, skipping...")
|
||||
return None
|
||||
|
||||
pattern = pattern_class(vllm_config=vllm_config, eps=eps)
|
||||
try:
|
||||
# Import the required pass class
|
||||
from torch._inductor.pattern_matcher import PatternMatcherPass
|
||||
|
||||
pm_pass = PatternMatcherPass()
|
||||
pattern.register(pm_pass)
|
||||
_registered_patterns.add(pattern_key)
|
||||
print(f"Successfully registered pattern: {pattern_key}")
|
||||
except RuntimeError as e:
|
||||
if "Duplicate pattern" in str(e):
|
||||
print(f"Pattern {pattern_key} already exists (caught from RuntimeError), skipping...")
|
||||
_registered_patterns.add(pattern_key)
|
||||
else:
|
||||
raise e
|
||||
return pattern
|
||||
|
||||
|
||||
@pytest.mark.parametrize("dtype", [torch.bfloat16])
|
||||
@pytest.mark.parametrize("hidden_size", [64])
|
||||
@pytest.mark.parametrize("num_tokens", [257])
|
||||
@pytest.mark.parametrize("eps", [1e-5])
|
||||
@pytest.mark.parametrize("use_bias", [False, True])
|
||||
@pytest.mark.parametrize("sp_enable", [False, True])
|
||||
def test_rmsnorm_quant_fusion(
|
||||
dtype: torch.dtype,
|
||||
hidden_size: int,
|
||||
num_tokens: int,
|
||||
eps: float,
|
||||
use_bias: bool,
|
||||
sp_enable: bool,
|
||||
):
|
||||
# Check if fusion operator is available
|
||||
if not hasattr(torch.ops.npu, "npu_add_rms_norm_quant"):
|
||||
pytest.skip("Fusion operator npu_add_rms_norm_quant not available, skipping test")
|
||||
|
||||
vllm_config = VllmConfig(model_config=ModelConfig(dtype=dtype))
|
||||
with vllm.config.set_current_vllm_config(vllm_config):
|
||||
update_environment_variables(
|
||||
{
|
||||
"RANK": "0",
|
||||
"LOCAL_RANK": "0",
|
||||
"WORLD_SIZE": "1",
|
||||
"MASTER_ADDR": "localhost",
|
||||
"MASTER_PORT": "12345",
|
||||
}
|
||||
)
|
||||
init_distributed_environment()
|
||||
ensure_model_parallel_initialized(1, 1)
|
||||
|
||||
with vllm.config.set_current_vllm_config(vllm_config), set_ascend_forward_context(None, vllm_config):
|
||||
if use_bias:
|
||||
# Skip test if custom ops are not available
|
||||
if not enable_custom_op():
|
||||
pytest.skip("Custom ops not available, skipping bias test")
|
||||
# Check if the bias operator exists
|
||||
if not hasattr(torch.ops._C_ascend, "npu_add_rms_norm_bias"):
|
||||
pytest.skip("Operator npu_add_rms_norm_bias not available, skipping bias test")
|
||||
if sp_enable:
|
||||
model = ModelSPWithBias(hidden_size, dtype, eps, device="npu")
|
||||
register_pattern_safe(
|
||||
AddRMSNormQuantSPPatternWithBias, vllm_config, eps, "GraphEXAddRMSNormQuantSPPatternWithBias"
|
||||
)
|
||||
else:
|
||||
model = ModelWithBias(hidden_size, dtype, eps, device="npu")
|
||||
register_pattern_safe(
|
||||
AddRMSNormQuantPatternWithBias, vllm_config, eps, "GraphEXAddRMSNormQuantPatternWithBias"
|
||||
)
|
||||
else:
|
||||
# The non-bias patterns currently use npu_add_rms_norm_bias in their pattern matching
|
||||
# so we need to skip if it's not available
|
||||
if not hasattr(torch.ops._C_ascend, "npu_add_rms_norm_bias"):
|
||||
pytest.skip("Operator npu_add_rms_norm_bias not available, skipping test")
|
||||
if sp_enable:
|
||||
model = ModelSPWithoutBias(hidden_size, dtype, eps, device="npu")
|
||||
register_pattern_safe(AddRMSNormQuantSPPattern, vllm_config, eps, "GraphEXAddRMSNormQuantSPPattern")
|
||||
else:
|
||||
model = ModelWithoutBias(hidden_size, dtype, eps, device="npu")
|
||||
register_pattern_safe(AddRMSNormQuantPattern, vllm_config, eps, "GraphEXAddRMSNormQuantPattern")
|
||||
|
||||
model = model.to("npu")
|
||||
x = torch.randn(num_tokens, hidden_size, device="npu", dtype=dtype, requires_grad=False)
|
||||
|
||||
with torch.no_grad():
|
||||
# Don't expect fusion since patterns are not properly integrated into the compilation pipeline
|
||||
# Just test that the model compiles and runs without errors
|
||||
compiled_model = torch.compile(model, backend="npugraph_ex", fullgraph=True, dynamic=True)
|
||||
compiled_out, compiled_res = compiled_model(x)
|
||||
|
||||
# Verify output shapes are correct
|
||||
assert compiled_out.shape == (num_tokens, hidden_size), (
|
||||
f"Expected shape {(num_tokens, hidden_size)}, got {compiled_out.shape}"
|
||||
)
|
||||
assert compiled_res.shape == (num_tokens, hidden_size), (
|
||||
f"Expected shape {(num_tokens, hidden_size)}, got {compiled_res.shape}"
|
||||
)
|
||||
@@ -0,0 +1,228 @@
|
||||
import copy
|
||||
|
||||
import npugraph_ex as nge
|
||||
import numpy as np
|
||||
import pytest
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import vllm.config
|
||||
from vllm.config import ModelConfig, VllmConfig
|
||||
from vllm.distributed import ensure_model_parallel_initialized, init_distributed_environment
|
||||
from vllm.utils.system_utils import update_environment_variables
|
||||
|
||||
from vllm_ascend.ascend_forward_context import set_ascend_forward_context
|
||||
from vllm_ascend.compilation.passes.qknorm_rope_fusion_pass import (
|
||||
QKNormRopeFusionPattern,
|
||||
QKNormRopeFusionPatternWithBias,
|
||||
)
|
||||
from vllm_ascend.ops.triton.triton_utils import init_device_properties_triton
|
||||
|
||||
MAX_POSITION_EMBEDDING = 262144
|
||||
|
||||
|
||||
def find_op(gm, op_default):
|
||||
return any(node.op == "call_function" and node.target == op_default for node in gm.graph.nodes)
|
||||
|
||||
|
||||
def create_pattern_wrapper(assert_func):
|
||||
original_func = nge.npu_fx_compiler._optimize_fx
|
||||
|
||||
def wrapper(gm, example_inputs=None, config=None):
|
||||
ret = original_func(gm, example_inputs, config)
|
||||
graph_after = copy.deepcopy(gm)
|
||||
assert_func(graph_after)
|
||||
return ret
|
||||
|
||||
return wrapper
|
||||
|
||||
|
||||
@pytest.fixture(scope="module", autouse=True)
|
||||
def init_triton():
|
||||
init_device_properties_triton()
|
||||
|
||||
|
||||
class ModelQKNormRopeWithoutBias(nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
head_dim: int,
|
||||
num_heads: int,
|
||||
num_kv_heads: int,
|
||||
dtype: torch.dtype = torch.bfloat16,
|
||||
eps: float = 1e-6,
|
||||
device="npu",
|
||||
):
|
||||
super().__init__()
|
||||
self.head_dim = head_dim
|
||||
self.num_heads = num_heads
|
||||
self.num_kv_heads = num_kv_heads
|
||||
self.q_size = num_heads * head_dim
|
||||
self.kv_size = num_kv_heads * head_dim
|
||||
self.eps = eps
|
||||
|
||||
# RMSNorm weight per head (shared across heads of same type)
|
||||
self.q_weight = nn.Parameter(torch.randn(head_dim, dtype=dtype, device=device))
|
||||
self.k_weight = nn.Parameter(torch.randn(head_dim, dtype=dtype, device=device))
|
||||
|
||||
def forward(self, qkv, cos_sin_cache, positions):
|
||||
"""
|
||||
Args:
|
||||
qkv: [T, q_size + 2*kv_size]
|
||||
cos: [1, T, 1, head_dim]
|
||||
sin: [1, T, 1, head_dim]
|
||||
Returns:
|
||||
q_rope, k_rope, v
|
||||
"""
|
||||
# Split QKV
|
||||
q, k, v = qkv.split([self.q_size, self.kv_size, self.kv_size], dim=-1)
|
||||
|
||||
# Q RMSNorm (per-head)
|
||||
q_by_head = q.view(*q.shape[:-1], self.num_heads, self.head_dim)
|
||||
q_norm_out, _ = torch.ops.npu.npu_rms_norm(q_by_head, self.q_weight, self.eps)
|
||||
|
||||
# K RMSNorm (per-head)
|
||||
k_by_head = k.view(*k.shape[:-1], self.num_kv_heads, self.head_dim)
|
||||
k_norm_out, _ = torch.ops.npu.npu_rms_norm(k_by_head, self.k_weight, self.eps)
|
||||
|
||||
# Reshape for RoPE: [T, num_heads, head_dim] -> [1, T, num_heads, head_dim]
|
||||
q_flat = q_norm_out.view(q.shape)
|
||||
k_flat = k_norm_out.view(k.shape)
|
||||
|
||||
# Apply RoPE
|
||||
q_rope, k_rope = torch.ops.vllm.npu_rotary_embedding(
|
||||
positions, q_flat, k_flat, cos_sin_cache, self.head_dim, self.head_dim, True
|
||||
)
|
||||
|
||||
return q_rope, k_rope, v
|
||||
|
||||
|
||||
class ModelQKNormRopeWithBias(nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
head_dim: int,
|
||||
num_heads: int,
|
||||
num_kv_heads: int,
|
||||
dtype: torch.dtype = torch.bfloat16,
|
||||
eps: float = 1e-6,
|
||||
device="npu",
|
||||
):
|
||||
super().__init__()
|
||||
self.head_dim = head_dim
|
||||
self.num_heads = num_heads
|
||||
self.num_kv_heads = num_kv_heads
|
||||
self.q_size = num_heads * head_dim
|
||||
self.kv_size = num_kv_heads * head_dim
|
||||
self.eps = eps
|
||||
|
||||
self.q_weight = nn.Parameter(torch.randn(head_dim, dtype=dtype, device=device))
|
||||
self.k_weight = nn.Parameter(torch.randn(head_dim, dtype=dtype, device=device))
|
||||
self.q_bias = nn.Parameter(torch.randn(head_dim, dtype=dtype, device=device))
|
||||
self.k_bias = nn.Parameter(torch.randn(head_dim, dtype=dtype, device=device))
|
||||
|
||||
def forward(self, qkv, cos_sin_cache, positions):
|
||||
# Split QKV
|
||||
q, k, v = qkv.split([self.q_size, self.kv_size, self.kv_size], dim=-1)
|
||||
|
||||
# Q RMSNorm + Bias
|
||||
q_by_head = q.view(*q.shape[:-1], self.num_heads, self.head_dim)
|
||||
q_norm_out, _ = torch.ops.npu.npu_rms_norm(q_by_head, self.q_weight, self.eps)
|
||||
q_normed = q_norm_out + self.q_bias
|
||||
|
||||
# K RMSNorm + Bias
|
||||
k_by_head = k.view(*k.shape[:-1], self.num_kv_heads, self.head_dim)
|
||||
k_norm_out, _ = torch.ops.npu.npu_rms_norm(k_by_head, self.k_weight, self.eps)
|
||||
k_normed = k_norm_out + self.k_bias
|
||||
|
||||
# Reshape for RoPE
|
||||
q_flat = q_normed.view(q.shape)
|
||||
k_flat = k_normed.view(k.shape)
|
||||
|
||||
# Apply RoPE
|
||||
q_rope, k_rope = torch.ops.vllm.npu_rotary_embedding(
|
||||
positions, q_flat, k_flat, cos_sin_cache, self.head_dim, self.head_dim, True
|
||||
)
|
||||
|
||||
return q_rope, k_rope, v
|
||||
|
||||
|
||||
def assert_qknorm_rope_fusion(after_gm, expect_fused=True, use_bias=False):
|
||||
check_rules = [
|
||||
(torch.ops.vllm.qkv_rmsnorm_rope.default, expect_fused),
|
||||
(torch.ops.npu.npu_rms_norm.default, not expect_fused),
|
||||
(torch.ops.vllm.npu_rotary_embedding.default, not expect_fused),
|
||||
]
|
||||
if use_bias:
|
||||
check_rules.append((torch.ops.aten.add.Tensor, not expect_fused))
|
||||
for torch_op, expect_exist in check_rules:
|
||||
found = find_op(after_gm, torch_op)
|
||||
if expect_exist:
|
||||
assert found, f"Expected operator '{torch_op}' but not find"
|
||||
else:
|
||||
assert not found, f"Not expected operator '{torch_op}' but find"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("dtype", [torch.bfloat16])
|
||||
@pytest.mark.parametrize("hidden_size", [64])
|
||||
@pytest.mark.parametrize("num_tokens", [257])
|
||||
@pytest.mark.parametrize("eps", [1e-5])
|
||||
@pytest.mark.parametrize("use_bias", [False, True])
|
||||
def test_rmsnorm_quant_fusion(
|
||||
dtype: torch.dtype,
|
||||
hidden_size: int,
|
||||
num_tokens: int,
|
||||
eps: float,
|
||||
use_bias: bool,
|
||||
):
|
||||
vllm_config = VllmConfig(model_config=ModelConfig(dtype=dtype))
|
||||
with vllm.config.set_current_vllm_config(vllm_config):
|
||||
update_environment_variables(
|
||||
{
|
||||
"RANK": "0",
|
||||
"LOCAL_RANK": "0",
|
||||
"WORLD_SIZE": "1",
|
||||
"MASTER_ADDR": "localhost",
|
||||
"MASTER_PORT": "12345",
|
||||
}
|
||||
)
|
||||
init_distributed_environment()
|
||||
ensure_model_parallel_initialized(1, 1)
|
||||
num_heads = 16
|
||||
num_kv_heads = 8
|
||||
head_dim = 128
|
||||
with vllm.config.set_current_vllm_config(vllm_config), set_ascend_forward_context(None, vllm_config):
|
||||
fusion_pattern = None
|
||||
q_size = num_heads * head_dim
|
||||
kv_size = num_kv_heads * head_dim
|
||||
qkv_size = q_size + 2 * kv_size
|
||||
if use_bias:
|
||||
model = ModelQKNormRopeWithBias(head_dim, num_heads, num_kv_heads, dtype, eps, device="npu")
|
||||
fusion_pattern = QKNormRopeFusionPatternWithBias(
|
||||
vllm_config=vllm_config, head_dim=head_dim, num_heads=num_heads, num_kv_heads=num_kv_heads, eps=eps
|
||||
)
|
||||
else:
|
||||
model = ModelQKNormRopeWithoutBias(head_dim, num_heads, num_kv_heads, dtype, eps, device="npu")
|
||||
fusion_pattern = QKNormRopeFusionPattern(
|
||||
vllm_config=vllm_config, head_dim=head_dim, num_heads=num_heads, num_kv_heads=num_kv_heads, eps=eps
|
||||
)
|
||||
from torch._inductor.pattern_matcher import PatternMatcherPass
|
||||
|
||||
pm_pass = PatternMatcherPass()
|
||||
fusion_pattern.register(pm_pass)
|
||||
model = model.to("npu")
|
||||
seq_len = num_tokens
|
||||
qkv = torch.randn(seq_len, qkv_size, device="npu", dtype=dtype)
|
||||
cos_sin_cache = torch.from_numpy(np.random.uniform(0, 1, [MAX_POSITION_EMBEDDING, head_dim])).to(dtype).npu()
|
||||
positions = torch.randint(
|
||||
low=0, high=MAX_POSITION_EMBEDDING, size=(num_tokens,), dtype=torch.int64, device="npu"
|
||||
)
|
||||
|
||||
with torch.no_grad():
|
||||
original_optimize = nge.npu_fx_compiler._optimize_fx
|
||||
nge.npu_fx_compiler._optimize_fx = create_pattern_wrapper(
|
||||
lambda gm: assert_qknorm_rope_fusion(gm, expect_fused=True, use_bias=use_bias)
|
||||
)
|
||||
|
||||
compiled_model = torch.compile(model, backend="npugraph_ex", fullgraph=True, dynamic=True)
|
||||
|
||||
compiled_model(qkv, cos_sin_cache, positions)
|
||||
|
||||
nge.npu_fx_compiler._optimize_fx = original_optimize
|
||||
@@ -0,0 +1,298 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import vllm.config
|
||||
from vllm.compilation.passes.fx_utils import OpOverload
|
||||
from vllm.config import ModelConfig, VllmConfig
|
||||
from vllm.distributed import ensure_model_parallel_initialized, init_distributed_environment
|
||||
from vllm.utils.system_utils import update_environment_variables
|
||||
|
||||
import vllm_ascend.ops.register_custom_ops # noqa
|
||||
from vllm_ascend.ascend_forward_context import set_ascend_forward_context
|
||||
from vllm_ascend.compilation.passes.norm_quant_fusion_pass import AddRMSNormQuantFusionPass
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
|
||||
from .backend import TestBackend
|
||||
|
||||
# Cache backend to avoid duplicate pattern registration
|
||||
_backend_cache = None
|
||||
|
||||
|
||||
def get_or_create_backend(vllm_config):
|
||||
"""Get or create backend with fusion passes (cached to avoid duplicate pattern registration)."""
|
||||
global _backend_cache
|
||||
if _backend_cache is None:
|
||||
_backend_cache = TestBackend(custom_passes=[AddRMSNormQuantFusionPass(vllm_config=vllm_config)])
|
||||
return _backend_cache
|
||||
|
||||
|
||||
class TestModelWithoutBias(nn.Module):
|
||||
"""
|
||||
A minimal test model that simulates the pattern:
|
||||
AddRMSNorm → Quantization (without bias)
|
||||
"""
|
||||
|
||||
def __init__(self, hidden_size: int, dtype: torch.dtype, eps: float = 1e-6, device="npu"):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.eps = eps
|
||||
self.rms_norm_weight = nn.Parameter(torch.randn(hidden_size, device=device))
|
||||
self.quant_scale = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_scale_reciprocal = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_offset = torch.zeros(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def forward(self, x):
|
||||
"""
|
||||
Forward pass:
|
||||
1. Perform npu_add_rms_norm
|
||||
2. Quantize the normalized output to int8
|
||||
Returns both quantized output and updated residual.
|
||||
"""
|
||||
residual = torch.zeros_like(x)
|
||||
|
||||
norm_output, _, new_residual = torch.ops._C_ascend.npu_add_rms_norm_bias(
|
||||
x, residual, self.rms_norm_weight, None, self.eps
|
||||
)
|
||||
|
||||
quantized_output = torch.ops.vllm.quantize(
|
||||
norm_output, self.quant_scale, self.quant_scale_reciprocal, self.quant_offset
|
||||
)
|
||||
|
||||
return quantized_output, new_residual
|
||||
|
||||
def ops_in_model_before(self) -> list[OpOverload]:
|
||||
"""Return the list of expected operators BEFORE fusion."""
|
||||
return [torch.ops._C_ascend.npu_add_rms_norm_bias.default, torch.ops.vllm.quantize.default]
|
||||
|
||||
def ops_in_model_after(self) -> list[OpOverload]:
|
||||
"""Return the list of expected operators AFTER successful fusion."""
|
||||
return [torch.ops.npu.npu_add_rms_norm_quant.default]
|
||||
|
||||
|
||||
class TestModelWithBias(nn.Module):
|
||||
"""
|
||||
A test model that simulates the pattern:
|
||||
AddRMSNorm → Add Bias → Quantization (with bias)
|
||||
"""
|
||||
|
||||
def __init__(self, hidden_size: int, dtype: torch.dtype, eps: float = 1e-6, device="npu"):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.eps = eps
|
||||
self.rms_norm_weight = nn.Parameter(torch.randn(hidden_size, device=device))
|
||||
self.bias = nn.Parameter(torch.randn(hidden_size, device=device))
|
||||
self.quant_scale = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_scale_reciprocal = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_offset = torch.zeros(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def forward(self, x):
|
||||
"""
|
||||
Forward pass:
|
||||
1. Perform npu_add_rms_norm
|
||||
2. Add bias
|
||||
3. Quantize to int8
|
||||
Returns both quantized output and updated residual.
|
||||
"""
|
||||
residual = torch.zeros_like(x)
|
||||
|
||||
norm_output_with_bias, _, new_residual = torch.ops._C_ascend.npu_add_rms_norm_bias(
|
||||
x, residual, self.rms_norm_weight, self.bias, self.eps
|
||||
)
|
||||
|
||||
quantized_output = torch.ops.vllm.quantize(
|
||||
norm_output_with_bias, self.quant_scale, self.quant_scale_reciprocal, self.quant_offset
|
||||
)
|
||||
|
||||
return quantized_output, new_residual
|
||||
|
||||
def ops_in_model_before(self) -> list[OpOverload]:
|
||||
"""Return the list of expected operators BEFORE fusion."""
|
||||
return [torch.ops._C_ascend.npu_add_rms_norm_bias.default, torch.ops.vllm.quantize.default]
|
||||
|
||||
def ops_in_model_after(self) -> list[OpOverload]:
|
||||
"""Return the list of expected operators AFTER successful fusion."""
|
||||
return [torch.ops.npu.npu_add_rms_norm_quant.default]
|
||||
|
||||
|
||||
class TestModelSPWithoutBias(nn.Module):
|
||||
"""
|
||||
A minimal test model that simulates the pattern:
|
||||
AddRMSNorm → maybe_allgather → Quantization (without bias)
|
||||
"""
|
||||
|
||||
def __init__(self, hidden_size: int, dtype: torch.dtype, eps: float = 1e-6, device="npu"):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.eps = eps
|
||||
self.rms_norm_weight = nn.Parameter(torch.randn(hidden_size, device=device))
|
||||
self.quant_scale = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_scale_reciprocal = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_offset = torch.zeros(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def forward(self, x):
|
||||
"""
|
||||
Forward pass:
|
||||
1. Perform npu_add_rms_norm
|
||||
2. Perform a fake maybe_all_gather_and_maybe_unpad
|
||||
3. Quantize the normalized output to int8
|
||||
Returns both quantized output and updated residual.
|
||||
"""
|
||||
residual = torch.zeros_like(x)
|
||||
|
||||
norm_output, _, new_residual = torch.ops._C_ascend.npu_add_rms_norm_bias(
|
||||
x, residual, self.rms_norm_weight, None, self.eps
|
||||
)
|
||||
|
||||
norm_output = torch.ops.vllm.maybe_all_gather_and_maybe_unpad(norm_output, True)
|
||||
|
||||
quantized_output = torch.ops.vllm.quantize(
|
||||
norm_output, self.quant_scale, self.quant_scale_reciprocal, self.quant_offset
|
||||
)
|
||||
|
||||
return quantized_output, new_residual
|
||||
|
||||
def ops_in_model_before(self) -> list[OpOverload]:
|
||||
"""Return the list of expected operators BEFORE fusion."""
|
||||
return [
|
||||
torch.ops._C_ascend.npu_add_rms_norm_bias.default,
|
||||
torch.ops.vllm.maybe_all_gather_and_maybe_unpad.default,
|
||||
torch.ops.vllm.quantize.default,
|
||||
]
|
||||
|
||||
def ops_in_model_after(self) -> list[OpOverload]:
|
||||
"""Return the list of expected operators AFTER successful fusion."""
|
||||
return [torch.ops.npu.npu_add_rms_norm_quant.default, torch.ops.vllm.maybe_all_gather_and_maybe_unpad.default]
|
||||
|
||||
|
||||
class TestModelSPWithBias(nn.Module):
|
||||
"""
|
||||
A minimal test model that simulates the pattern:
|
||||
AddRMSNorm → Add bias → maybe_allgather → Quantization (without bias)
|
||||
"""
|
||||
|
||||
def __init__(self, hidden_size: int, dtype: torch.dtype, eps: float = 1e-6, device="npu"):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.eps = eps
|
||||
self.rms_norm_weight = nn.Parameter(torch.randn(hidden_size, device=device))
|
||||
self.bias = nn.Parameter(torch.randn(hidden_size, device=device))
|
||||
self.quant_scale = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_scale_reciprocal = torch.ones(hidden_size, dtype=dtype, device=device)
|
||||
self.quant_offset = torch.zeros(hidden_size, dtype=dtype, device=device)
|
||||
|
||||
def forward(self, x):
|
||||
"""
|
||||
Forward pass:
|
||||
1. Perform npu_add_rms_norm
|
||||
2. Add bias
|
||||
3. Perform a fake maybe_all_gather_and_maybe_unpad
|
||||
4. Quantize the normalized output to int8
|
||||
Returns both quantized output and updated residual.
|
||||
"""
|
||||
residual = torch.zeros_like(x)
|
||||
|
||||
norm_output_with_bias, _, new_residual = torch.ops._C_ascend.npu_add_rms_norm_bias(
|
||||
x, residual, self.rms_norm_weight, self.bias, self.eps
|
||||
)
|
||||
|
||||
norm_output_with_bias = torch.ops.vllm.maybe_all_gather_and_maybe_unpad(norm_output_with_bias, True)
|
||||
|
||||
quantized_output = torch.ops.vllm.quantize(
|
||||
norm_output_with_bias, self.quant_scale, self.quant_scale_reciprocal, self.quant_offset
|
||||
)
|
||||
|
||||
return quantized_output, new_residual
|
||||
|
||||
def ops_in_model_before(self) -> list[OpOverload]:
|
||||
"""Return the list of expected operators BEFORE fusion."""
|
||||
return [
|
||||
torch.ops._C_ascend.npu_add_rms_norm_bias.default,
|
||||
torch.ops.vllm.maybe_all_gather_and_maybe_unpad.default,
|
||||
torch.ops.vllm.quantize.default,
|
||||
]
|
||||
|
||||
def ops_in_model_after(self) -> list[OpOverload]:
|
||||
"""Return the list of expected operators AFTER successful fusion."""
|
||||
return [torch.ops.npu.npu_add_rms_norm_quant.default, torch.ops.vllm.maybe_all_gather_and_maybe_unpad.default]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("dtype", [torch.bfloat16])
|
||||
@pytest.mark.parametrize("hidden_size", [64])
|
||||
@pytest.mark.parametrize("num_tokens", [257])
|
||||
@pytest.mark.parametrize("eps", [1e-5, 1e-6])
|
||||
@pytest.mark.parametrize("use_bias", [False, True])
|
||||
@pytest.mark.parametrize("sp_enable", [False, True])
|
||||
def test_rmsnorm_quant_fusion(
|
||||
dtype: torch.dtype,
|
||||
hidden_size: int,
|
||||
num_tokens: int,
|
||||
eps: float,
|
||||
use_bias: bool,
|
||||
sp_enable: bool,
|
||||
):
|
||||
"""
|
||||
End-to-end test for AddRMSNorm+Quantize fusion.
|
||||
Compares: Operator presence/absence before and after graph transformation
|
||||
"""
|
||||
torch.set_default_dtype(dtype)
|
||||
torch.manual_seed(1)
|
||||
|
||||
vllm_config = VllmConfig(model_config=ModelConfig(dtype=dtype))
|
||||
|
||||
with vllm.config.set_current_vllm_config(vllm_config):
|
||||
update_environment_variables(
|
||||
{
|
||||
"RANK": "0",
|
||||
"LOCAL_RANK": "0",
|
||||
"WORLD_SIZE": "1",
|
||||
"MASTER_ADDR": "localhost",
|
||||
"MASTER_PORT": "12345",
|
||||
}
|
||||
)
|
||||
init_distributed_environment()
|
||||
ensure_model_parallel_initialized(1, 1)
|
||||
|
||||
with vllm.config.set_current_vllm_config(vllm_config), set_ascend_forward_context(None, vllm_config):
|
||||
backend = get_or_create_backend(vllm_config)
|
||||
if use_bias:
|
||||
if not enable_custom_op():
|
||||
return
|
||||
if sp_enable:
|
||||
model = TestModelSPWithBias(hidden_size, dtype, eps, device="npu")
|
||||
else:
|
||||
model = TestModelWithBias(hidden_size, dtype, eps, device="npu")
|
||||
else:
|
||||
if sp_enable:
|
||||
model = TestModelSPWithoutBias(hidden_size, dtype, eps, device="npu")
|
||||
else:
|
||||
model = TestModelWithoutBias(hidden_size, dtype, eps, device="npu")
|
||||
model = model.to("npu")
|
||||
|
||||
x = torch.rand(num_tokens, hidden_size, device="npu", dtype=dtype, requires_grad=False)
|
||||
|
||||
result_unfused = model(x)
|
||||
print("Unfused result:", [t.shape for t in result_unfused])
|
||||
model_fused = torch.compile(model, backend=backend)
|
||||
result_fused = model_fused(x)
|
||||
print("Fused result:", [t.shape for t in result_fused])
|
||||
|
||||
print("=== Checking operator fusion ===")
|
||||
backend.check_before_ops(model.ops_in_model_before(), fully_replaced=not sp_enable)
|
||||
backend.check_after_ops(model.ops_in_model_after())
|
||||
0
tests/e2e/pull_request/one_card/lora/__init__.py
Normal file
0
tests/e2e/pull_request/one_card/lora/__init__.py
Normal file
59
tests/e2e/pull_request/one_card/lora/test_ilama_lora.py
Normal file
59
tests/e2e/pull_request/one_card/lora/test_ilama_lora.py
Normal file
@@ -0,0 +1,59 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
import vllm
|
||||
from vllm.lora.request import LoRARequest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
MODEL_PATH = "vllm-ascend/ilama-3.2-1B"
|
||||
|
||||
PROMPT_TEMPLATE = """I want you to act as a SQL terminal in front of an example database, you need only to return the sql command to me.Below is an instruction that describes a task, Write a response that appropriately completes the request.\n"\n##Instruction:\nconcert_singer contains tables such as stadium, singer, concert, singer_in_concert. Table stadium has columns such as Stadium_ID, Location, Name, Capacity, Highest, Lowest, Average. Stadium_ID is the primary key.\nTable singer has columns such as Singer_ID, Name, Country, Song_Name, Song_release_year, Age, Is_male. Singer_ID is the primary key.\nTable concert has columns such as concert_ID, concert_Name, Theme, Stadium_ID, Year. concert_ID is the primary key.\nTable singer_in_concert has columns such as concert_ID, Singer_ID. concert_ID is the primary key.\nThe Stadium_ID of concert is the foreign key of Stadium_ID of stadium.\nThe Singer_ID of singer_in_concert is the foreign key of Singer_ID of singer.\nThe concert_ID of singer_in_concert is the foreign key of concert_ID of concert.\n\n###Input:\n{query}\n\n###Response:""" # noqa: E501
|
||||
|
||||
EXPECTED_LORA_OUTPUT = [
|
||||
"SELECT count(*) FROM singer",
|
||||
"SELECT avg(age) , min(age) , max(age) FROM singer WHERE country = 'France'", # noqa: E501
|
||||
"SELECT DISTINCT Country FROM singer WHERE Age > 20",
|
||||
]
|
||||
|
||||
|
||||
def do_sample(llm: vllm.LLM, lora_path: str, lora_id: int) -> list[str]:
|
||||
prompts = [
|
||||
PROMPT_TEMPLATE.format(query="How many singers do we have?"),
|
||||
PROMPT_TEMPLATE.format(
|
||||
query="What is the average, minimum, and maximum age of all singers from France?" # noqa: E501
|
||||
),
|
||||
PROMPT_TEMPLATE.format(
|
||||
query="What are all distinct countries where singers above age 20 are from?" # noqa: E501
|
||||
),
|
||||
]
|
||||
sampling_params = vllm.SamplingParams(temperature=0, max_tokens=32)
|
||||
outputs = llm.generate(
|
||||
prompts, sampling_params, lora_request=LoRARequest(str(lora_id), lora_id, lora_path) if lora_id else None
|
||||
)
|
||||
# Print the outputs.
|
||||
generated_texts: list[str] = []
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text.strip()
|
||||
generated_texts.append(generated_text)
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
return generated_texts
|
||||
|
||||
|
||||
def test_ilama_lora(ilama_lora_files):
|
||||
with VllmRunner(
|
||||
MODEL_PATH,
|
||||
enable_lora=True,
|
||||
dtype="half",
|
||||
max_loras=4,
|
||||
max_model_len=1024,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
max_num_seqs=16,
|
||||
enforce_eager=True,
|
||||
) as vllm_model:
|
||||
output1 = do_sample(vllm_model.model, ilama_lora_files, lora_id=1)
|
||||
for i in range(len(EXPECTED_LORA_OUTPUT)):
|
||||
assert output1[i] == EXPECTED_LORA_OUTPUT[i]
|
||||
|
||||
output2 = do_sample(vllm_model.model, ilama_lora_files, lora_id=2)
|
||||
for i in range(len(EXPECTED_LORA_OUTPUT)):
|
||||
assert output2[i] == EXPECTED_LORA_OUTPUT[i]
|
||||
140
tests/e2e/pull_request/one_card/lora/test_llama32_lora.py
Normal file
140
tests/e2e/pull_request/one_card/lora/test_llama32_lora.py
Normal file
@@ -0,0 +1,140 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
import vllm
|
||||
import vllm.config
|
||||
from vllm.lora.request import LoRARequest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
|
||||
enable_custom_op()
|
||||
|
||||
PROMPT_TEMPLATE = """<|eot_id|><|start_header_id|>user<|end_header_id|>
|
||||
I want you to act as a SQL terminal in front of an example database, you need only to return the sql command to me.Below is an instruction that describes a task, Write a response that appropriately completes the request.
|
||||
"
|
||||
##Instruction:
|
||||
candidate_poll contains tables such as candidate, people. Table candidate has columns such as Candidate_ID, People_ID, Poll_Source, Date, Support_rate, Consider_rate, Oppose_rate, Unsure_rate. Candidate_ID is the primary key.
|
||||
Table people has columns such as People_ID, Sex, Name, Date_of_Birth, Height, Weight. People_ID is the primary key.
|
||||
The People_ID of candidate is the foreign key of People_ID of people.
|
||||
###Input:
|
||||
{context}
|
||||
###Response:<|eot_id|><|start_header_id|>assistant<|end_header_id|>
|
||||
""" # noqa: E501
|
||||
|
||||
EXPECTED_LORA_OUTPUT = [
|
||||
"SELECT count(*) FROM candidate",
|
||||
"SELECT count(*) FROM candidate",
|
||||
"SELECT poll_source FROM candidate GROUP BY poll_source ORDER BY count(*) DESC LIMIT 1", # noqa: E501
|
||||
"SELECT poll_source FROM candidate GROUP BY poll_source ORDER BY count(*) DESC LIMIT 1", # noqa: E501
|
||||
]
|
||||
|
||||
EXPECTED_BASE_MODEL_OUTPUT = [
|
||||
"SELECT COUNT(*) FROM candidate",
|
||||
"`SELECT COUNT(*) FROM candidate;`",
|
||||
"SELECT Poll_Source FROM candidate GROUP BY Poll_Source ORDER BY COUNT(*) DESC LIMIT 1;",
|
||||
"SELECT * FROM candidate ORDER BY Candidate_ID DESC LIMIT 1",
|
||||
]
|
||||
|
||||
# For hk region, we need to use the model from hf to avoid the network issue
|
||||
MODEL_PATH = "meta-llama/Llama-3.2-3B-Instruct"
|
||||
|
||||
|
||||
def do_sample(
|
||||
llm: vllm.LLM,
|
||||
lora_path: str,
|
||||
lora_id: int,
|
||||
tensorizer_config_dict: dict | None = None,
|
||||
) -> list[str]:
|
||||
prompts = [
|
||||
PROMPT_TEMPLATE.format(context="How many candidates are there?"),
|
||||
PROMPT_TEMPLATE.format(context="Count the number of candidates."),
|
||||
PROMPT_TEMPLATE.format(
|
||||
context="Which poll resource provided the most number of candidate information?" # noqa: E501
|
||||
),
|
||||
PROMPT_TEMPLATE.format(context="Return the poll resource associated with the most candidates."),
|
||||
]
|
||||
|
||||
sampling_params = vllm.SamplingParams(temperature=0, max_tokens=64, stop=["<|im_end|>"])
|
||||
if tensorizer_config_dict is not None:
|
||||
outputs = llm.generate(
|
||||
prompts,
|
||||
sampling_params,
|
||||
lora_request=LoRARequest(
|
||||
str(lora_id),
|
||||
lora_id,
|
||||
lora_path,
|
||||
tensorizer_config_dict=tensorizer_config_dict,
|
||||
)
|
||||
if lora_id
|
||||
else None,
|
||||
)
|
||||
else:
|
||||
outputs = llm.generate(
|
||||
prompts,
|
||||
sampling_params,
|
||||
lora_request=LoRARequest(str(lora_id), lora_id, lora_path) if lora_id else None,
|
||||
)
|
||||
|
||||
generated_texts: list[str] = []
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
generated_texts.append(generated_text)
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
return generated_texts
|
||||
|
||||
|
||||
def generate_and_test(llm, llama32_lora_files, tensorizer_config_dict: dict | None = None):
|
||||
print("lora adapter created")
|
||||
print("lora 1")
|
||||
assert (
|
||||
do_sample(
|
||||
llm,
|
||||
llama32_lora_files,
|
||||
tensorizer_config_dict=tensorizer_config_dict,
|
||||
lora_id=1,
|
||||
)
|
||||
== EXPECTED_LORA_OUTPUT
|
||||
)
|
||||
|
||||
print("lora 2")
|
||||
assert (
|
||||
do_sample(
|
||||
llm,
|
||||
llama32_lora_files,
|
||||
tensorizer_config_dict=tensorizer_config_dict,
|
||||
lora_id=2,
|
||||
)
|
||||
== EXPECTED_LORA_OUTPUT
|
||||
)
|
||||
|
||||
print("base model")
|
||||
assert (
|
||||
do_sample(
|
||||
llm,
|
||||
llama32_lora_files,
|
||||
tensorizer_config_dict=tensorizer_config_dict,
|
||||
lora_id=0,
|
||||
)
|
||||
== EXPECTED_BASE_MODEL_OUTPUT
|
||||
)
|
||||
|
||||
print("removing lora")
|
||||
|
||||
|
||||
@patch.dict("os.environ", {"VLLM_USE_MODELSCOPE": "False"})
|
||||
def test_llama_lora(llama32_lora_files):
|
||||
vllm_model = VllmRunner(
|
||||
MODEL_PATH,
|
||||
enable_lora=True,
|
||||
# also test odd max_num_seqs
|
||||
max_num_seqs=7,
|
||||
max_model_len=1024,
|
||||
max_loras=4,
|
||||
compilation_config={"cudagraph_mode": "PIECEWISE"},
|
||||
)
|
||||
llm = vllm_model.model
|
||||
generate_and_test(llm, llama32_lora_files)
|
||||
@@ -0,0 +1,116 @@
|
||||
"""
|
||||
This script contains:
|
||||
1. test lora with speculative decoding for batch inference
|
||||
"""
|
||||
|
||||
import random
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
import torch
|
||||
from vllm import LLM, SamplingParams
|
||||
from vllm.lora.request import LoRARequest
|
||||
|
||||
LORA_TEST_PROMPT_MAP: dict[str, str] = {}
|
||||
|
||||
LORA_TEST_PROMPT_MAP["vllm-ascend/qwen-linear-algebra-coder"] = """
|
||||
### INSTRUCTION:
|
||||
You are an AI assistant that generates Python code to solve linear
|
||||
algebra problems.
|
||||
|
||||
### PROBLEM:
|
||||
Find the eigenvalues and eigenvectors of the following 3x3 matrix:
|
||||
[[3, 2, 0],
|
||||
[2, 3, 0],
|
||||
[0, 0, 2]]
|
||||
|
||||
### OUTPUT FORMAT (STRICT):
|
||||
Numbers should be represented as integers only.
|
||||
|
||||
### PYTHON SOLUTION:
|
||||
"""
|
||||
|
||||
SEED = 42
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_setup",
|
||||
[
|
||||
(
|
||||
"eagle3",
|
||||
"Qwen/Qwen3-1.7B",
|
||||
"vllm-ascend/Qwen3-1.7B_eagle3",
|
||||
"vllm-ascend/qwen-linear-algebra-coder",
|
||||
1,
|
||||
)
|
||||
],
|
||||
)
|
||||
def test_batch_inference_correctness(
|
||||
model_setup: tuple[str, str, str, str, int],
|
||||
):
|
||||
"""
|
||||
Compare the outputs of a LLM with only Lora and a LLM with both SD and Lora.
|
||||
Should be the same and no failure when doing batch inference.
|
||||
model_setup: (method, model_name, spec_model_name, lora_path, tp_size)
|
||||
"""
|
||||
# Disable randomness
|
||||
torch.manual_seed(SEED)
|
||||
np.random.seed(SEED)
|
||||
random.seed(SEED)
|
||||
torch.use_deterministic_algorithms(True)
|
||||
|
||||
method, model_name, spec_model_name, lora_path, tp_size = model_setup
|
||||
prompts = [LORA_TEST_PROMPT_MAP[lora_path]] * 100
|
||||
lora_request = LoRARequest("adapter", 1, lora_path)
|
||||
sampling_params = SamplingParams(temperature=0.0, top_p=1.0, top_k=-1, seed=SEED, max_tokens=128)
|
||||
|
||||
# without speculative decoding
|
||||
ref_llm = LLM(
|
||||
model=model_name,
|
||||
trust_remote_code=True,
|
||||
tensor_parallel_size=tp_size,
|
||||
max_model_len=2048,
|
||||
max_num_seqs=4,
|
||||
enable_lora=True,
|
||||
max_loras=1,
|
||||
max_cpu_loras=1,
|
||||
max_lora_rank=16,
|
||||
)
|
||||
ref_outputs = ref_llm.generate(prompts, sampling_params, lora_request=lora_request)
|
||||
del ref_llm
|
||||
|
||||
# speculative decoding
|
||||
lora_spec_llm = LLM(
|
||||
model=model_name,
|
||||
trust_remote_code=True,
|
||||
tensor_parallel_size=tp_size,
|
||||
speculative_config={
|
||||
"method": method,
|
||||
"model": spec_model_name,
|
||||
"num_speculative_tokens": 3,
|
||||
"max_model_len": 2048,
|
||||
},
|
||||
max_model_len=2048,
|
||||
max_num_seqs=4,
|
||||
enable_lora=True,
|
||||
max_loras=1,
|
||||
max_cpu_loras=1,
|
||||
max_lora_rank=16,
|
||||
)
|
||||
lora_spec_outputs = lora_spec_llm.generate(prompts, sampling_params, lora_request=lora_request)
|
||||
del lora_spec_llm
|
||||
|
||||
matches = 0
|
||||
misses = 0
|
||||
for ref_output, spec_output in zip(ref_outputs, lora_spec_outputs):
|
||||
if ref_output.outputs[0].text == spec_output.outputs[0].text:
|
||||
matches += 1
|
||||
else:
|
||||
misses += 1
|
||||
print(f"ref_output: {ref_output.outputs[0].text}")
|
||||
print(f"spec_output: {spec_output.outputs[0].text}")
|
||||
|
||||
# Heuristic: expect at least 90% of the prompts to match exactly
|
||||
# Upon failure, inspect the outputs to check for inaccuracy.
|
||||
print(f"match ratio: {matches}/{len(ref_outputs)}")
|
||||
assert matches > int(0.90 * len(ref_outputs))
|
||||
106
tests/e2e/pull_request/one_card/lora/test_olmoe_lora.py
Normal file
106
tests/e2e/pull_request/one_card/lora/test_olmoe_lora.py
Normal file
@@ -0,0 +1,106 @@
|
||||
from collections.abc import Sequence
|
||||
|
||||
import vllm
|
||||
from vllm.lora.request import LoRARequest
|
||||
|
||||
MODEL_PATH = "allenai/OLMoE-1B-7B-0125-Instruct"
|
||||
|
||||
PROMPT_TEMPLATE = """I want you to act as a SQL terminal in front of an example database, you need only to return the sql command to me. Do not return any additional explanation. Below is an instruction that describes a task, Write a response that appropriately completes the request.
|
||||
"
|
||||
##Instruction:
|
||||
candidate_poll contains tables such as candidate, people. Table candidate has columns such as Candidate_ID, People_ID, Poll_Source, Date, Support_rate, Consider_rate, Oppose_rate, Unsure_rate. Candidate_ID is the primary key.
|
||||
Table people has columns such as People_ID, Sex, Name, Date_of_Birth, Height, Weight. People_ID is the primary key.
|
||||
The People_ID of candidate is the foreign key of People_ID of people.
|
||||
|
||||
|
||||
###Input:
|
||||
{context}
|
||||
|
||||
###Response:""" # noqa: E501
|
||||
|
||||
EXPECTED_LORA_OUTPUT = [
|
||||
"SELECT count(*) FROM candidate",
|
||||
"SELECT count(*) FROM candidate",
|
||||
"SELECT poll_source FROM candidate GROUP BY poll_source ORDER BY count(*) DESC LIMIT 1", # noqa: E501
|
||||
"SELECT poll_source FROM candidate GROUP BY poll_source ORDER BY count(*) DESC LIMIT 1", # noqa: E501
|
||||
]
|
||||
|
||||
EXPECTED_BASE_MODEL_OUTPUT = [
|
||||
"SELECT COUNT(Candidate_ID) FROM candidate",
|
||||
"SELECT COUNT(Candidate_ID) FROM candidate",
|
||||
"SELECT Candidate_ID, COUNT(*) as Total_Candidates\nFROM candidate\nINNER JOIN people ON candidate.People_ID = people.People_ID", # noqa: E501
|
||||
# There are multiple acceptable responses
|
||||
(
|
||||
"SELECT Candidate_ID, Poll_Source FROM candidate WHERE People_ID IN (SELECT People_ID FROM people) ORDER BY COUNT(*) DESC LIMIT 1", # noqa: E501
|
||||
"SELECT Candidate_ID, Poll_Source FROM candidate WHERE COUNT(People_ID) = (SELECT COUNT(People_ID) FROM people) ORDER BY Candidate_ID DESC LIMIT 1", # noqa: E501
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
def _output_matches(generated: str, accepted: str | Sequence[str]) -> bool:
|
||||
if isinstance(accepted, str):
|
||||
accepted = (accepted,)
|
||||
return any(generated.startswith(s) for s in accepted)
|
||||
|
||||
|
||||
def generate_and_test(
|
||||
llm: vllm.LLM,
|
||||
lora_path: str,
|
||||
lora_id: list[int | None] | int | None,
|
||||
compare_lower: bool = False,
|
||||
) -> None:
|
||||
prompts = [
|
||||
PROMPT_TEMPLATE.format(context="How many candidates are there?"),
|
||||
PROMPT_TEMPLATE.format(context="Count the number of candidates."),
|
||||
PROMPT_TEMPLATE.format(
|
||||
context="Which poll resource provided the most number of candidate information?" # noqa: E501
|
||||
),
|
||||
PROMPT_TEMPLATE.format(context="Return the poll resource associated with the most candidates."),
|
||||
]
|
||||
|
||||
lora_request = None
|
||||
if isinstance(lora_id, int):
|
||||
lora_request = LoRARequest(str(lora_id), lora_id, lora_path)
|
||||
elif isinstance(lora_id, list):
|
||||
lora_request = [LoRARequest(str(i), i, lora_path) if i is not None else None for i in lora_id]
|
||||
|
||||
sampling_params = vllm.SamplingParams(temperature=0, max_tokens=64)
|
||||
outputs = llm.generate(prompts, sampling_params, lora_request=lora_request)
|
||||
# Print the outputs.
|
||||
generated_texts: list[str] = []
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text.strip()
|
||||
generated_texts.append(generated_text)
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
|
||||
for i in range(len(EXPECTED_LORA_OUTPUT)):
|
||||
req_lora_id = lora_id[i] if isinstance(lora_id, list) else lora_id
|
||||
generated_text = generated_texts[i]
|
||||
expected_output = EXPECTED_LORA_OUTPUT[i] if req_lora_id is not None else EXPECTED_BASE_MODEL_OUTPUT[i]
|
||||
|
||||
if compare_lower:
|
||||
generated_text = generated_text.lower()
|
||||
if isinstance(expected_output, str):
|
||||
expected_output = (expected_output.lower(),)
|
||||
else:
|
||||
expected_output = tuple(s.lower() for s in expected_output)
|
||||
assert _output_matches(generated_text, expected_output), (
|
||||
f"Output {i}: {generated_text!r} does not match any of {expected_output!r}"
|
||||
)
|
||||
|
||||
|
||||
def test_olmoe_lora(olmoe_lora_files):
|
||||
# We enable enforce_eager=True here to reduce VRAM usage for lora-test CI,
|
||||
# Otherwise, the lora-test will fail due to CUDA OOM.
|
||||
llm = vllm.LLM(
|
||||
MODEL_PATH,
|
||||
max_model_len=1024,
|
||||
enable_lora=True,
|
||||
max_loras=4,
|
||||
enforce_eager=False,
|
||||
trust_remote_code=True,
|
||||
enable_chunked_prefill=True,
|
||||
)
|
||||
|
||||
generate_and_test(llm, olmoe_lora_files, lora_id=1)
|
||||
@@ -0,0 +1,90 @@
|
||||
import vllm
|
||||
from transformers import AutoTokenizer
|
||||
from vllm.lora.request import LoRARequest
|
||||
|
||||
MODEL_PATH = "Qwen/Qwen3.5-4B"
|
||||
TEXT_LORA_ID = 1
|
||||
|
||||
# text-only task
|
||||
TEXT_PROMPT_TEMPLATE = """Write a SQL query for the given database.\nSchema:\nTables:\n - stadium(Stadium_ID, Location, Name, Capacity, Highest, Lowest, Average)\n - singer(Singer_ID, Name, Country, Song_Name, Song_release_year, Age, Is_male)\n - concert(concert_ID, concert_Name, Theme, Stadium_ID, Year)\n - singer_in_concert(concert_ID, Singer_ID)\n\nQuestion:\n{query}""" # noqa: E501
|
||||
|
||||
TEXT_EXPECTED_LORA_OUTPUT = [
|
||||
"SELECT count(*) FROM singer",
|
||||
"SELECT avg(age) , min(age) , max(age) FROM singer WHERE country = 'France'",
|
||||
"SELECT name FROM stadium WHERE stadium_id NOT IN (SELECT stadium_id FROM concert)",
|
||||
]
|
||||
|
||||
|
||||
TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True)
|
||||
|
||||
|
||||
def _assert_exact_outputs(generated_texts: list[str], expected_outputs: list[str]) -> None:
|
||||
assert generated_texts == expected_outputs
|
||||
|
||||
|
||||
def _run_text_lora_sample(
|
||||
llm: vllm.LLM,
|
||||
lora_path: str,
|
||||
lora_id: int,
|
||||
) -> list[str]:
|
||||
prompts = [
|
||||
TEXT_PROMPT_TEMPLATE.format(query="How many singers do we have?"),
|
||||
TEXT_PROMPT_TEMPLATE.format(
|
||||
query=("What is the average, minimum, and maximum age of all singers from France?")
|
||||
),
|
||||
TEXT_PROMPT_TEMPLATE.format(query="What are the names of the stadiums without any concerts?"),
|
||||
]
|
||||
input_templates = []
|
||||
for prompt_text in prompts:
|
||||
messages = [{"role": "user", "content": prompt_text}]
|
||||
prompt = TOKENIZER.apply_chat_template(
|
||||
messages,
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
enable_thinking=False, # disable thinking
|
||||
)
|
||||
input_templates.append(prompt)
|
||||
|
||||
outputs = llm.generate(
|
||||
input_templates,
|
||||
vllm.SamplingParams(temperature=0.01, max_tokens=512),
|
||||
lora_request=LoRARequest(str(lora_id), lora_id, lora_path),
|
||||
)
|
||||
|
||||
generated_texts: list[str] = []
|
||||
for output in outputs:
|
||||
generated_text = output.outputs[0].text.strip()
|
||||
generated_texts.append(generated_text)
|
||||
print(f"Prompt: {output.prompt!r}, Generated text: {generated_text!r}")
|
||||
return generated_texts
|
||||
|
||||
|
||||
def _assert_qwen35_text_lora(
|
||||
llm: vllm.LLM,
|
||||
qwen35_text_lora_files: str,
|
||||
) -> None:
|
||||
generated_texts = _run_text_lora_sample(
|
||||
llm,
|
||||
qwen35_text_lora_files,
|
||||
TEXT_LORA_ID,
|
||||
)
|
||||
|
||||
_assert_exact_outputs(generated_texts, TEXT_EXPECTED_LORA_OUTPUT)
|
||||
|
||||
|
||||
def test_qwen35_text_lora(qwen35_text_lora_files):
|
||||
llm = vllm.LLM(
|
||||
model=MODEL_PATH,
|
||||
max_model_len=4096,
|
||||
enable_lora=True,
|
||||
max_loras=2,
|
||||
max_num_seqs=4,
|
||||
max_lora_rank=8,
|
||||
enforce_eager=True,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
|
||||
_assert_qwen35_text_lora(
|
||||
llm,
|
||||
qwen35_text_lora_files,
|
||||
)
|
||||
150
tests/e2e/pull_request/one_card/lora/test_qwen3_multi_loras.py
Normal file
150
tests/e2e/pull_request/one_card/lora/test_qwen3_multi_loras.py
Normal file
@@ -0,0 +1,150 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
from unittest.mock import patch
|
||||
|
||||
from vllm import SamplingParams
|
||||
from vllm.lora.request import LoRARequest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
|
||||
enable_custom_op()
|
||||
|
||||
MODEL_PATH = "Qwen/Qwen3-0.6B"
|
||||
LORA_NAME_PATH_MAP = {
|
||||
"Alice": "charent/self_cognition_Alice",
|
||||
"Bob": "charent/self_cognition_Bob",
|
||||
"Cat": "charent/self_cognition_Bob", # same as Bob
|
||||
}
|
||||
|
||||
LORA_RANK = 8
|
||||
|
||||
LORA_TEST_PROMPTS = ["What is GitHub?", "Hi, tell me about you"]
|
||||
LORA_TEST_EXPECTED = [
|
||||
"GitHub is an open-source platform that provides a way to manage and develop software projects. It allows developers to store and manage code, collaborate on projects, and automate tasks.", # noqa: E501
|
||||
"I am Alice, an AI assistant developed by GitHub/Charent.", # noqa: E501
|
||||
]
|
||||
|
||||
|
||||
def format_chatml_messages(prompt: str):
|
||||
return [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{"role": "user", "content": prompt},
|
||||
]
|
||||
|
||||
|
||||
@patch.dict("os.environ", {"VLLM_USE_MODELSCOPE": "False"})
|
||||
def test_multi_loras_with_tp_sync():
|
||||
lora_name_id_map = {}
|
||||
increase_lora_id = 0
|
||||
|
||||
def make_add_lora_request(name: str, path: str):
|
||||
nonlocal increase_lora_id
|
||||
increase_lora_id += 1
|
||||
lora_name_id_map[name] = increase_lora_id
|
||||
|
||||
return LoRARequest(
|
||||
lora_name=name,
|
||||
lora_int_id=increase_lora_id,
|
||||
lora_path=path,
|
||||
)
|
||||
|
||||
vllm_model = VllmRunner(
|
||||
MODEL_PATH,
|
||||
enable_lora=True,
|
||||
# dtype="half",
|
||||
max_loras=2, # ensure max_loras < max_cpu_loras
|
||||
max_lora_rank=LORA_RANK,
|
||||
max_model_len=512,
|
||||
gpu_memory_utilization=0.9,
|
||||
enforce_eager=True,
|
||||
# tensor_parallel_size=2, # ensure tp >= 2
|
||||
max_cpu_loras=4, # ensure max_cpu_loras >= 2
|
||||
)
|
||||
llm = vllm_model.model
|
||||
|
||||
def run_check_lora(fn, args, expected: list):
|
||||
fn(args)
|
||||
assert set(llm.llm_engine.list_loras()) == set(expected)
|
||||
|
||||
# simulate add loras with CLI args
|
||||
# likes: `--lora-modules Alice=/path/to/Alice Bob=/path/to/Bob`
|
||||
run_check_lora(
|
||||
llm.llm_engine.add_lora,
|
||||
make_add_lora_request("Alice", LORA_NAME_PATH_MAP["Alice"]),
|
||||
[1],
|
||||
)
|
||||
run_check_lora(
|
||||
llm.llm_engine.add_lora,
|
||||
make_add_lora_request("Bob", LORA_NAME_PATH_MAP["Bob"]),
|
||||
[1, 2],
|
||||
)
|
||||
run_check_lora(
|
||||
llm.llm_engine.add_lora,
|
||||
make_add_lora_request("Cat", LORA_NAME_PATH_MAP["Cat"]),
|
||||
[1, 2, 3],
|
||||
)
|
||||
|
||||
# set temperature = 0 for greedy search
|
||||
sampling_params = SamplingParams(temperature=0, max_tokens=64)
|
||||
|
||||
def call_llm_get_outputs(prompt: str, lora_name: str):
|
||||
lora_request = LoRARequest(
|
||||
lora_name=lora_name,
|
||||
lora_int_id=lora_name_id_map[lora_name],
|
||||
lora_path=LORA_NAME_PATH_MAP[lora_name],
|
||||
)
|
||||
messages = format_chatml_messages(prompt)
|
||||
outputs = llm.chat(
|
||||
[messages],
|
||||
sampling_params,
|
||||
chat_template_kwargs={"enable_thinking": False}, # for those loras, ensure enable_thinking=False
|
||||
lora_request=lora_request,
|
||||
use_tqdm=False,
|
||||
)
|
||||
output_text = outputs[0].outputs[0].text
|
||||
return output_text
|
||||
|
||||
def reload_lora(name: str):
|
||||
"""
|
||||
reload a lora to simulate the case:
|
||||
setting `VLLM_ALLOW_RUNTIME_LORA_UPDATING=true`
|
||||
for dynamic lora loading and unloading
|
||||
"""
|
||||
remove_lora_response = llm.llm_engine.remove_lora(lora_id=lora_name_id_map[name])
|
||||
|
||||
add_lora_response = llm.llm_engine.add_lora(make_add_lora_request(name, LORA_NAME_PATH_MAP[name]))
|
||||
|
||||
print(f"{remove_lora_response=}, {add_lora_response=}")
|
||||
|
||||
def check_outputs(outputs: str, expected: str, prompt: str):
|
||||
print(f"{prompt=}.\n{expected=}\n{outputs=}")
|
||||
print("\n----------------------------\n")
|
||||
assert outputs == expected
|
||||
|
||||
for prompt, expected_output in zip(LORA_TEST_PROMPTS, LORA_TEST_EXPECTED):
|
||||
output_text = call_llm_get_outputs(prompt, "Alice")
|
||||
check_outputs(output_text, expected_output, prompt)
|
||||
|
||||
# call Bob, ignore what it is output
|
||||
call_llm_get_outputs(prompt, "Bob")
|
||||
print("After call Bob:")
|
||||
|
||||
# call Alice
|
||||
output_text = call_llm_get_outputs(prompt, "Alice")
|
||||
check_outputs(output_text, expected_output, prompt)
|
||||
|
||||
# reload Bob Lora
|
||||
reload_lora("Bob")
|
||||
print("After reload Bob:")
|
||||
|
||||
# call Alice
|
||||
output_text = call_llm_get_outputs(prompt, "Alice")
|
||||
check_outputs(output_text, expected_output, prompt)
|
||||
|
||||
# reload Alice Lora
|
||||
reload_lora("Alice")
|
||||
print("After reload Alice:")
|
||||
|
||||
output_text = call_llm_get_outputs(prompt, "Alice")
|
||||
check_outputs(output_text, expected_output, prompt)
|
||||
74
tests/e2e/pull_request/one_card/lora/test_qwen3_reranker_lora.py
Executable file
74
tests/e2e/pull_request/one_card/lora/test_qwen3_reranker_lora.py
Executable file
@@ -0,0 +1,74 @@
|
||||
from pathlib import Path
|
||||
|
||||
from vllm import LLM
|
||||
|
||||
model_name = "Qwen/Qwen3-Reranker-0.6B"
|
||||
|
||||
|
||||
def get_llm() -> LLM:
|
||||
"""
|
||||
Initializes and returns the LLM model for Qwen3-Reranker.
|
||||
|
||||
Returns:
|
||||
LLM: Configured vLLM instance for reranking tasks.
|
||||
|
||||
Note:
|
||||
This function loads the ORIGINAL Qwen3-Reranker model with specific
|
||||
overrides to make it compatible with vLLM's score API.
|
||||
"""
|
||||
return LLM(
|
||||
# Specify the original model from HuggingFace
|
||||
model=model_name,
|
||||
# Use pooling runner for score task
|
||||
runner="pooling",
|
||||
# HuggingFace model configuration overrides required for compatibility
|
||||
hf_overrides={
|
||||
# Manually route to sequence classification architecture
|
||||
# This tells vLLM to use Qwen3ForSequenceClassification instead of
|
||||
# the default Qwen3ForCausalLM
|
||||
"architectures": ["Qwen3ForSequenceClassification"],
|
||||
# Specify which token logits to extract from the language model head
|
||||
# The original reranker uses "no" and "yes" token logits for scoring
|
||||
"classifier_from_token": ["no", "yes"],
|
||||
# Enable special handling for original Qwen3-Reranker models
|
||||
# This flag triggers conversion logic that transforms the two token
|
||||
# vectors into a single classification vector
|
||||
"is_original_qwen3_reranker": True,
|
||||
},
|
||||
enable_lora=True,
|
||||
)
|
||||
|
||||
|
||||
def test_reranker_models_lora():
|
||||
# Load the Jinja template for formatting query-document pairs
|
||||
# The template ensures proper formatting for the reranker model
|
||||
template_home = Path(__file__).parents[1] / "pooling" / "template"
|
||||
template_path = "qwen3_reranker.jinja"
|
||||
chat_template = (template_home / template_path).read_text()
|
||||
|
||||
# Sample queries for testing the reranker
|
||||
queries = [
|
||||
"What is the capital of China?",
|
||||
"Explain gravity",
|
||||
]
|
||||
|
||||
# Corresponding documents to be scored against each query
|
||||
documents = [
|
||||
"The capital of China is Beijing.",
|
||||
"Gravity is a force that attracts two bodies towards each other. It gives weight to physical objects and is "
|
||||
"responsible for the movement of planets around the sun.",
|
||||
]
|
||||
|
||||
# Initialize the LLM model with the original Qwen3-Reranker configuration
|
||||
llm = get_llm()
|
||||
|
||||
# Compute relevance scores for each query-document pair
|
||||
# The score() method returns a relevance score for each pair
|
||||
# Higher scores indicate better relevance
|
||||
outputs = llm.score(queries, documents, chat_template=chat_template)
|
||||
|
||||
# Extract and print the relevance scores from the outputs
|
||||
# Each output contains a score representing query-document relevance
|
||||
print("-" * 30)
|
||||
print("Relevance scores:", [output.outputs.score for output in outputs])
|
||||
print("-" * 30)
|
||||
171
tests/e2e/pull_request/one_card/model_runner_v2/test_basic.py
Normal file
171
tests/e2e/pull_request/one_card/model_runner_v2/test_basic.py
Normal file
@@ -0,0 +1,171 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from vllm_ascend.utils import vllm_version_is
|
||||
|
||||
MODELS = ["Qwen/Qwen3-0.6B", "vllm-ascend/DeepSeek-V2-Lite-W8A8"]
|
||||
|
||||
MAIN_MODELS = ["LLM-Research/Meta-Llama-3.1-8B-Instruct"]
|
||||
EGALE_MODELS = ["vllm-ascend/EAGLE-LLaMA3.1-Instruct-8B"]
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
vllm_version_is("0.23.0"),
|
||||
reason="v2 model runner patches not supported on v0.23.0",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.skipif(True, reason="Fix me, it's broken after CANN and trition-ascend are upgraded.")
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [32])
|
||||
@pytest.mark.parametrize("enforce_eager", [True])
|
||||
@patch.dict(os.environ, {"VLLM_USE_V2_MODEL_RUNNER": "1"})
|
||||
def test_qwen3_dense_eager_mode(
|
||||
model: str,
|
||||
max_tokens: int,
|
||||
enforce_eager: bool,
|
||||
) -> None:
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
sampling_params = SamplingParams(
|
||||
max_tokens=max_tokens,
|
||||
temperature=0.5,
|
||||
logprobs=2,
|
||||
prompt_logprobs=2,
|
||||
logit_bias={0: -1.0, 1: 0.5},
|
||||
min_p=0.01,
|
||||
bad_words=["the", " the"],
|
||||
)
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
enforce_eager=enforce_eager,
|
||||
async_scheduling=True,
|
||||
) as runner:
|
||||
runner.model.generate(prompts, sampling_params)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MAIN_MODELS)
|
||||
@pytest.mark.parametrize("eagle_model", EGALE_MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [32])
|
||||
@pytest.mark.parametrize("enforce_eager", [False])
|
||||
@pytest.mark.parametrize(
|
||||
"compilation_config",
|
||||
[
|
||||
pytest.param(
|
||||
{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [4, 8]},
|
||||
id="full_decode_only",
|
||||
),
|
||||
pytest.param({}, id="default_full_and_piecewise"),
|
||||
],
|
||||
)
|
||||
@patch.dict(os.environ, {"VLLM_USE_V2_MODEL_RUNNER": "1"})
|
||||
def test_egale_spec_decoding(
|
||||
model: str,
|
||||
eagle_model: str,
|
||||
max_tokens: int,
|
||||
enforce_eager: bool,
|
||||
compilation_config: dict,
|
||||
) -> None:
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
sampling_params = SamplingParams(max_tokens=max_tokens, temperature=0.0)
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
enforce_eager=enforce_eager,
|
||||
async_scheduling=True,
|
||||
speculative_config={
|
||||
"model": eagle_model,
|
||||
"method": "eagle",
|
||||
"num_speculative_tokens": 3,
|
||||
},
|
||||
compilation_config=compilation_config,
|
||||
) as runner:
|
||||
runner.model.generate(prompts, sampling_params)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [32])
|
||||
@pytest.mark.parametrize("enforce_eager", [False])
|
||||
@pytest.mark.parametrize(
|
||||
"compilation_config",
|
||||
[
|
||||
pytest.param({"cudagraph_mode": "FULL_DECODE_ONLY"}, id="full_decode_only"),
|
||||
pytest.param({}, id="default_full_and_piecewise"),
|
||||
],
|
||||
)
|
||||
@patch.dict(os.environ, {"VLLM_USE_V2_MODEL_RUNNER": "1"})
|
||||
def test_qwen3_dense_graph_mode(
|
||||
model: str,
|
||||
max_tokens: int,
|
||||
enforce_eager: bool,
|
||||
compilation_config: dict,
|
||||
) -> None:
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
sampling_params = SamplingParams(max_tokens=max_tokens, temperature=0.0)
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
enforce_eager=enforce_eager,
|
||||
compilation_config=compilation_config,
|
||||
) as runner:
|
||||
outputs = runner.model.generate(prompts, sampling_params)
|
||||
|
||||
if model != "Qwen/Qwen3-0.6B":
|
||||
return
|
||||
|
||||
expected_outputs = [
|
||||
" Lina. I'm a 22-year-old student from China.",
|
||||
" the same as the president of the United Nations. This is because the president",
|
||||
" Paris. The capital of France is also the capital of the Republic of France",
|
||||
" not just about the technology itself but also about the human aspect-how we",
|
||||
]
|
||||
|
||||
matches = 0
|
||||
misses = 0
|
||||
for output, expected_output in zip(outputs, expected_outputs):
|
||||
if output.outputs[0].text[:10] == expected_output[:10]:
|
||||
matches += 1
|
||||
else:
|
||||
misses += 1
|
||||
print(f"output: {output.outputs[0].text}")
|
||||
print(f"expected_output: {expected_output}")
|
||||
|
||||
assert misses == 0
|
||||
171
tests/e2e/pull_request/one_card/model_runner_v2/test_uva.py
Normal file
171
tests/e2e/pull_request/one_card/model_runner_v2/test_uva.py
Normal file
@@ -0,0 +1,171 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from vllm_ascend.utils import vllm_version_is
|
||||
|
||||
MODELS = ["Qwen/Qwen3-0.6B", "vllm-ascend/DeepSeek-V2-Lite-W8A8"]
|
||||
|
||||
MAIN_MODELS = ["LLM-Research/Meta-Llama-3.1-8B-Instruct"]
|
||||
EGALE_MODELS = ["vllm-ascend/EAGLE-LLaMA3.1-Instruct-8B"]
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
vllm_version_is("0.23.0"),
|
||||
reason="v2 model runner patches not supported on v0.23.0",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.skipif(True, reason="Fix me, it's broken after CANN and trition-ascend are upgraded.")
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [32])
|
||||
@pytest.mark.parametrize("enforce_eager", [True])
|
||||
@patch.dict(os.environ, {"VLLM_USE_V2_MODEL_RUNNER": "1", "PYTORCH_NPU_ALLOC_CONF": "pinned_mem_register:True"})
|
||||
def test_qwen3_dense_eager_mode(
|
||||
model: str,
|
||||
max_tokens: int,
|
||||
enforce_eager: bool,
|
||||
) -> None:
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
sampling_params = SamplingParams(
|
||||
max_tokens=max_tokens,
|
||||
temperature=0.5,
|
||||
logprobs=2,
|
||||
prompt_logprobs=2,
|
||||
logit_bias={0: -1.0, 1: 0.5},
|
||||
min_p=0.01,
|
||||
bad_words=["the", " the"],
|
||||
)
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
enforce_eager=enforce_eager,
|
||||
async_scheduling=True,
|
||||
) as runner:
|
||||
runner.model.generate(prompts, sampling_params)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MAIN_MODELS)
|
||||
@pytest.mark.parametrize("eagle_model", EGALE_MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [32])
|
||||
@pytest.mark.parametrize("enforce_eager", [False])
|
||||
@pytest.mark.parametrize(
|
||||
"compilation_config",
|
||||
[
|
||||
pytest.param(
|
||||
{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [4, 8]},
|
||||
id="full_decode_only",
|
||||
),
|
||||
pytest.param({}, id="default_full_and_piecewise"),
|
||||
],
|
||||
)
|
||||
@patch.dict(os.environ, {"VLLM_USE_V2_MODEL_RUNNER": "1", "PYTORCH_NPU_ALLOC_CONF": "pinned_mem_register:True"})
|
||||
def test_egale_spec_decoding(
|
||||
model: str,
|
||||
eagle_model: str,
|
||||
max_tokens: int,
|
||||
enforce_eager: bool,
|
||||
compilation_config: dict,
|
||||
) -> None:
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
sampling_params = SamplingParams(max_tokens=max_tokens, temperature=0.0)
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
enforce_eager=enforce_eager,
|
||||
async_scheduling=True,
|
||||
speculative_config={
|
||||
"model": eagle_model,
|
||||
"method": "eagle",
|
||||
"num_speculative_tokens": 3,
|
||||
},
|
||||
compilation_config=compilation_config,
|
||||
) as runner:
|
||||
runner.model.generate(prompts, sampling_params)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [32])
|
||||
@pytest.mark.parametrize("enforce_eager", [False])
|
||||
@pytest.mark.parametrize(
|
||||
"compilation_config",
|
||||
[
|
||||
pytest.param({"cudagraph_mode": "FULL_DECODE_ONLY"}, id="full_decode_only"),
|
||||
pytest.param({}, id="default_full_and_piecewise"),
|
||||
],
|
||||
)
|
||||
@patch.dict(os.environ, {"VLLM_USE_V2_MODEL_RUNNER": "1", "PYTORCH_NPU_ALLOC_CONF": "pinned_mem_register:True"})
|
||||
def test_qwen3_dense_graph_mode(
|
||||
model: str,
|
||||
max_tokens: int,
|
||||
enforce_eager: bool,
|
||||
compilation_config: dict,
|
||||
) -> None:
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
sampling_params = SamplingParams(max_tokens=max_tokens, temperature=0.0)
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
enforce_eager=enforce_eager,
|
||||
compilation_config=compilation_config,
|
||||
) as runner:
|
||||
outputs = runner.model.generate(prompts, sampling_params)
|
||||
|
||||
if model != "Qwen/Qwen3-0.6B":
|
||||
return
|
||||
|
||||
expected_outputs = [
|
||||
" Lina. I'm a 22-year-old student from China.",
|
||||
" the same as the president of the United Nations. This is because the president",
|
||||
" Paris. The capital of France is also the capital of the Republic of France",
|
||||
" not just about the technology itself but also about the human aspect-how we",
|
||||
]
|
||||
|
||||
matches = 0
|
||||
misses = 0
|
||||
for output, expected_output in zip(outputs, expected_outputs):
|
||||
if output.outputs[0].text[:10] == expected_output[:10]:
|
||||
matches += 1
|
||||
else:
|
||||
misses += 1
|
||||
print(f"output: {output.outputs[0].text}")
|
||||
print(f"expected_output: {expected_output}")
|
||||
|
||||
assert misses == 0
|
||||
0
tests/e2e/pull_request/one_card/pooling/__init__.py
Normal file
0
tests/e2e/pull_request/one_card/pooling/__init__.py
Normal file
11
tests/e2e/pull_request/one_card/pooling/template/qwen3_reranker.jinja
Executable file
11
tests/e2e/pull_request/one_card/pooling/template/qwen3_reranker.jinja
Executable file
@@ -0,0 +1,11 @@
|
||||
<|im_start|>system
|
||||
Judge whether the Document meets the requirements based on the Query and the Instruct provided. Note that the answer can only be "yes" or "no".<|im_end|>
|
||||
<|im_start|>user
|
||||
<Instruct>: {{ messages | selectattr("role", "eq", "system") | map(attribute="content") | first | default("Given a web search query, retrieve relevant passages that answer the query") }}
|
||||
<Query>: {{ messages | selectattr("role", "eq", "query") | map(attribute="content") | first }}
|
||||
<Document>: {{ messages | selectattr("role", "eq", "document") | map(attribute="content") | first }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
<think>
|
||||
|
||||
</think>
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
import huggingface_hub
|
||||
import torch
|
||||
from modelscope import snapshot_download # type: ignore[import-untyped]
|
||||
from transformers import AutoModelForSequenceClassification
|
||||
|
||||
from tests.e2e.conftest import (
|
||||
HfRunner,
|
||||
VllmRunner,
|
||||
cleanup_dist_env_and_memory,
|
||||
wait_until_npu_memory_free,
|
||||
)
|
||||
|
||||
|
||||
@wait_until_npu_memory_free(target_free_percentage=0.7)
|
||||
def test_qwen_pooling_classify_correctness() -> None:
|
||||
model_name = snapshot_download(
|
||||
"Howeee/Qwen2.5-1.5B-apeach",
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is what",
|
||||
]
|
||||
|
||||
with HfRunner(model_name, dtype="float32", auto_cls=AutoModelForSequenceClassification) as hf_runner:
|
||||
hf_outputs = hf_runner.classify(prompts)
|
||||
cleanup_dist_env_and_memory()
|
||||
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
max_model_len=None,
|
||||
cudagraph_capture_sizes=[4],
|
||||
) as vllm_runner:
|
||||
vllm_outputs = vllm_runner.classify(prompts)
|
||||
|
||||
for hf_output, vllm_output in zip(hf_outputs, vllm_outputs):
|
||||
hf_output = torch.tensor(hf_output)
|
||||
vllm_output = torch.tensor(vllm_output)
|
||||
assert torch.allclose(hf_output, vllm_output, 1e-2)
|
||||
135
tests/e2e/pull_request/one_card/pooling/test_embedding.py
Normal file
135
tests/e2e/pull_request/one_card/pooling/test_embedding.py
Normal file
@@ -0,0 +1,135 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
|
||||
#
|
||||
import huggingface_hub
|
||||
import pytest
|
||||
from modelscope import snapshot_download # type: ignore[import-untyped]
|
||||
|
||||
from tests.e2e.conftest import HfRunner, VllmRunner
|
||||
from tests.e2e.utils import check_embeddings_close
|
||||
|
||||
MODELS = [
|
||||
"Qwen/Qwen3-Embedding-0.6B", # lasttoken
|
||||
"intfloat/multilingual-e5-small", # mean_tokens
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
def test_embed_models_correctness(model: str):
|
||||
queries = ["What is the capital of China?", "Explain gravity"]
|
||||
|
||||
model_name = snapshot_download(
|
||||
model,
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
max_model_len=None,
|
||||
cudagraph_capture_sizes=[4],
|
||||
) as vllm_runner:
|
||||
vllm_outputs = vllm_runner.embed(queries)
|
||||
|
||||
with HfRunner(
|
||||
model_name,
|
||||
dtype="float32",
|
||||
is_sentence_transformer=True,
|
||||
) as hf_runner:
|
||||
hf_outputs = hf_runner.encode(queries)
|
||||
|
||||
check_embeddings_close(
|
||||
embeddings_0_lst=hf_outputs,
|
||||
embeddings_1_lst=vllm_outputs,
|
||||
name_0="hf",
|
||||
name_1="vllm",
|
||||
tol=1e-2,
|
||||
)
|
||||
|
||||
|
||||
def test_causal_embed_models_using_prefix_caching_correctness():
|
||||
# This test is to verify the correctness of prefix caching for embedding models.
|
||||
# We compare the outputs of vLLM with and without prefix caching enabled, and check if they are close enough.
|
||||
# We set the input query to be very long to make sure prefix caching is triggered.
|
||||
queries = ["What is the capital of China?" * 256, "Explain gravity"]
|
||||
|
||||
model_name = snapshot_download(
|
||||
"Qwen/Qwen3-Embedding-0.6B",
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
max_model_len=None,
|
||||
cudagraph_capture_sizes=[4],
|
||||
enable_prefix_caching=True,
|
||||
) as vllm_runner_using_caching:
|
||||
vllm_outputs_without_caching = vllm_runner_using_caching.embed(queries)
|
||||
vllm_outputs_with_caching = vllm_runner_using_caching.embed(queries)
|
||||
|
||||
check_embeddings_close(
|
||||
embeddings_0_lst=vllm_outputs_without_caching,
|
||||
embeddings_1_lst=vllm_outputs_with_caching,
|
||||
name_0="without_caching",
|
||||
name_1="with_caching",
|
||||
tol=1e-2,
|
||||
)
|
||||
|
||||
|
||||
def test_bge_m3_correctness():
|
||||
queries = ["What is the capital of China?", "Explain gravity"]
|
||||
|
||||
model_name = snapshot_download(
|
||||
"BAAI/bge-m3",
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
cudagraph_capture_sizes=[4],
|
||||
) as vllm_aclgraph_runner:
|
||||
vllm_aclgraph_outputs = vllm_aclgraph_runner.embed(queries)
|
||||
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
runner="pooling",
|
||||
enforce_eager=True,
|
||||
) as vllm_runner:
|
||||
vllm_eager_outputs = vllm_runner.embed(queries)
|
||||
|
||||
with HfRunner(
|
||||
model_name,
|
||||
dtype="float32",
|
||||
is_sentence_transformer=True,
|
||||
) as hf_runner:
|
||||
hf_outputs = hf_runner.encode(queries)
|
||||
|
||||
check_embeddings_close(
|
||||
embeddings_0_lst=hf_outputs,
|
||||
embeddings_1_lst=vllm_eager_outputs,
|
||||
name_0="hf",
|
||||
name_1="vllm",
|
||||
tol=1e-2,
|
||||
)
|
||||
|
||||
check_embeddings_close(
|
||||
embeddings_0_lst=vllm_eager_outputs,
|
||||
embeddings_1_lst=vllm_aclgraph_outputs,
|
||||
name_0="eager",
|
||||
name_1="aclgraph",
|
||||
tol=1e-2,
|
||||
)
|
||||
167
tests/e2e/pull_request/one_card/pooling/test_scoring.py
Normal file
167
tests/e2e/pull_request/one_card/pooling/test_scoring.py
Normal file
@@ -0,0 +1,167 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
import huggingface_hub
|
||||
import pytest
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
from modelscope import snapshot_download # type: ignore[import-untyped]
|
||||
|
||||
from tests.e2e.conftest import HfRunner, VllmRunner
|
||||
|
||||
CROSS_ENCODER_MODELS = [
|
||||
"dengcao/ms-marco-MiniLM-L6-v2", # Bert
|
||||
"BAAI/bge-reranker-v2-m3", # Roberta
|
||||
]
|
||||
|
||||
EMBEDDING_MODELS = [
|
||||
"sentence-transformers/all-MiniLM-L12-v2",
|
||||
]
|
||||
|
||||
TEXTS_1 = [
|
||||
"What is the capital of France?",
|
||||
"What is the capital of Germany?",
|
||||
]
|
||||
|
||||
TEXTS_2 = [
|
||||
"The capital of France is Paris.",
|
||||
"The capital of Germany is Berlin.",
|
||||
]
|
||||
|
||||
DTYPE = "half"
|
||||
|
||||
|
||||
@pytest.fixture(scope="module", params=CROSS_ENCODER_MODELS)
|
||||
def model_name(request):
|
||||
yield snapshot_download(
|
||||
request.param,
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
|
||||
|
||||
def test_cross_encoder_score_1_to_1(model_name):
|
||||
text_pair = [TEXTS_1[0], TEXTS_2[0]]
|
||||
|
||||
with HfRunner(model_name, dtype=DTYPE, is_cross_encoder=True) as hf_model:
|
||||
hf_outputs = hf_model.predict([text_pair]).tolist()
|
||||
|
||||
with VllmRunner(
|
||||
model_name, runner="pooling", dtype=DTYPE, cudagraph_capture_sizes=[4], max_model_len=None
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(text_pair[0], text_pair[1])
|
||||
|
||||
assert len(vllm_outputs) == 1
|
||||
assert len(hf_outputs) == 1
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
|
||||
|
||||
def test_cross_encoder_score_1_to_N(model_name):
|
||||
text_pairs = [
|
||||
[TEXTS_1[0], TEXTS_2[0]],
|
||||
[TEXTS_1[0], TEXTS_2[1]],
|
||||
]
|
||||
|
||||
with HfRunner(model_name, dtype=DTYPE, is_cross_encoder=True) as hf_model:
|
||||
hf_outputs = hf_model.predict(text_pairs).tolist()
|
||||
|
||||
with VllmRunner(
|
||||
model_name, runner="pooling", dtype=DTYPE, cudagraph_capture_sizes=[4], max_model_len=None
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(TEXTS_1[0], TEXTS_2)
|
||||
|
||||
assert len(vllm_outputs) == 2
|
||||
assert len(hf_outputs) == 2
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
assert hf_outputs[1] == pytest.approx(vllm_outputs[1], rel=0.01)
|
||||
|
||||
|
||||
def test_cross_encoder_score_N_to_N(model_name):
|
||||
text_pairs = [
|
||||
[TEXTS_1[0], TEXTS_2[0]],
|
||||
[TEXTS_1[1], TEXTS_2[1]],
|
||||
]
|
||||
|
||||
with HfRunner(model_name, dtype=DTYPE, is_cross_encoder=True) as hf_model:
|
||||
hf_outputs = hf_model.predict(text_pairs).tolist()
|
||||
|
||||
with VllmRunner(
|
||||
model_name, runner="pooling", dtype=DTYPE, cudagraph_capture_sizes=[4], max_model_len=None
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(TEXTS_1, TEXTS_2)
|
||||
|
||||
assert len(vllm_outputs) == 2
|
||||
assert len(hf_outputs) == 2
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
assert hf_outputs[1] == pytest.approx(vllm_outputs[1], rel=0.01)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module", params=EMBEDDING_MODELS)
|
||||
def emb_model_name(request):
|
||||
yield snapshot_download(
|
||||
request.param,
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
|
||||
|
||||
def test_embedding_score_1_to_1(emb_model_name):
|
||||
text_pair = [TEXTS_1[0], TEXTS_2[0]]
|
||||
|
||||
with HfRunner(emb_model_name, dtype=DTYPE, is_sentence_transformer=True) as hf_model:
|
||||
hf_embeddings = hf_model.encode(text_pair)
|
||||
hf_outputs = [F.cosine_similarity(*map(torch.tensor, hf_embeddings), dim=0)]
|
||||
|
||||
with VllmRunner(
|
||||
emb_model_name, runner="pooling", dtype=DTYPE, cudagraph_capture_sizes=[4], max_model_len=None
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(text_pair[0], text_pair[1])
|
||||
|
||||
assert len(vllm_outputs) == 1
|
||||
assert len(hf_outputs) == 1
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
|
||||
|
||||
def test_embedding_score_1_to_N(emb_model_name):
|
||||
text_pairs = [
|
||||
[TEXTS_1[0], TEXTS_2[0]],
|
||||
[TEXTS_1[0], TEXTS_2[1]],
|
||||
]
|
||||
|
||||
with HfRunner(emb_model_name, dtype=DTYPE, is_sentence_transformer=True) as hf_model:
|
||||
hf_embeddings = [hf_model.encode(text_pair) for text_pair in text_pairs]
|
||||
hf_outputs = [F.cosine_similarity(*map(torch.tensor, pair), dim=0) for pair in hf_embeddings]
|
||||
|
||||
with VllmRunner(
|
||||
emb_model_name, runner="pooling", dtype=DTYPE, cudagraph_capture_sizes=[4], max_model_len=None
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(TEXTS_1[0], TEXTS_2)
|
||||
|
||||
assert len(vllm_outputs) == 2
|
||||
assert len(hf_outputs) == 2
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
assert hf_outputs[1] == pytest.approx(vllm_outputs[1], rel=0.01)
|
||||
|
||||
|
||||
def test_embedding_score_N_to_N(emb_model_name):
|
||||
text_pairs = [
|
||||
[TEXTS_1[0], TEXTS_2[0]],
|
||||
[TEXTS_1[1], TEXTS_2[1]],
|
||||
]
|
||||
|
||||
with HfRunner(emb_model_name, dtype=DTYPE, is_sentence_transformer=True) as hf_model:
|
||||
hf_embeddings = [hf_model.encode(text_pair) for text_pair in text_pairs]
|
||||
hf_outputs = [F.cosine_similarity(*map(torch.tensor, pair), dim=0) for pair in hf_embeddings]
|
||||
|
||||
with VllmRunner(
|
||||
emb_model_name, runner="pooling", dtype=DTYPE, cudagraph_capture_sizes=[4], max_model_len=None
|
||||
) as vllm_model:
|
||||
vllm_outputs = vllm_model.score(TEXTS_1, TEXTS_2)
|
||||
|
||||
assert len(vllm_outputs) == 2
|
||||
assert len(hf_outputs) == 2
|
||||
|
||||
assert hf_outputs[0] == pytest.approx(vllm_outputs[0], rel=0.01)
|
||||
assert hf_outputs[1] == pytest.approx(vllm_outputs[1], rel=0.01)
|
||||
53
tests/e2e/pull_request/one_card/spec_decode/conftest.py
Normal file
53
tests/e2e/pull_request/one_card/spec_decode/conftest.py
Normal file
@@ -0,0 +1,53 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import random
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def test_prompts() -> list[list[dict[str, Any]]]:
|
||||
prompt_types = ["repeat", "sentence"]
|
||||
num_prompts = 100
|
||||
prompts = []
|
||||
|
||||
random.seed(0)
|
||||
random_prompt_type_choices = random.choices(prompt_types, k=num_prompts)
|
||||
|
||||
for kind in random_prompt_type_choices:
|
||||
word_choices = ["test", "temp", "hello", "where"]
|
||||
word = random.choice(word_choices)
|
||||
if kind == "repeat":
|
||||
prompt = f"""
|
||||
please repeat the word '{word}' 10 times.
|
||||
give no other output than the word at least ten times in a row,
|
||||
in lowercase with spaces between each word and without quotes.
|
||||
"""
|
||||
elif kind == "sentence":
|
||||
prompt = f"""
|
||||
please give a ten-word sentence that
|
||||
uses the word {word} at least once.
|
||||
give no other output than that simple sentence without quotes.
|
||||
"""
|
||||
else:
|
||||
raise ValueError(f"Unknown prompt type: {kind}")
|
||||
prompts.append([{"role": "user", "content": prompt}])
|
||||
|
||||
return prompts
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sampling_config() -> SamplingParams:
|
||||
return SamplingParams(temperature=0, max_tokens=10, ignore_eos=False)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def model_name() -> str:
|
||||
return "LLM-Research/Meta-Llama-3.1-8B-Instruct"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def vl_model_name() -> str:
|
||||
return "Qwen/Qwen3-VL-8B-Instruct"
|
||||
81
tests/e2e/pull_request/one_card/spec_decode/test_dflash.py
Normal file
81
tests/e2e/pull_request/one_card/spec_decode/test_dflash.py
Normal file
@@ -0,0 +1,81 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from transformers import AutoTokenizer
|
||||
from vllm import SamplingParams
|
||||
from vllm.config import CompilationConfig
|
||||
from vllm.v1.metrics.reader import Counter, Vector
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.pull_request.one_card.spec_decode.utils import BASELINES, DFLASH, calculate_acceptance_per_pos
|
||||
|
||||
|
||||
@pytest.mark.parametrize("method", DFLASH.keys())
|
||||
@pytest.mark.parametrize("num_speculative_tokens", [8])
|
||||
def test_dflash_acceptance(
|
||||
method: str,
|
||||
num_speculative_tokens: int,
|
||||
):
|
||||
main_model_name = DFLASH[method]["main"]
|
||||
spec_model_name = DFLASH[method]["spec"]
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
main_model_name,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0,
|
||||
ignore_eos=False,
|
||||
max_tokens=256,
|
||||
)
|
||||
|
||||
prompts = [{"role": "user", "content": "Hello, your name is"}]
|
||||
prompts = [
|
||||
tokenizer.apply_chat_template(
|
||||
[prompt],
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
enable_thinking=False,
|
||||
)
|
||||
for prompt in prompts
|
||||
]
|
||||
|
||||
speculative_config = {
|
||||
"method": "dflash",
|
||||
"model": spec_model_name,
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
}
|
||||
|
||||
compilation_config = CompilationConfig(cudagraph_mode="FULL_DECODE_ONLY", cudagraph_capture_sizes=[9, 18])
|
||||
|
||||
with VllmRunner(
|
||||
main_model_name,
|
||||
max_model_len=4096,
|
||||
disable_log_stats=False,
|
||||
tensor_parallel_size=1,
|
||||
max_num_seqs=256,
|
||||
distributed_executor_backend="mp",
|
||||
gpu_memory_utilization=0.8,
|
||||
speculative_config=speculative_config,
|
||||
compilation_config=compilation_config,
|
||||
enable_prefix_caching=False,
|
||||
) as llm:
|
||||
outputs = llm.model.generate(prompts, sampling_params)
|
||||
metrics = llm.model.get_metrics()
|
||||
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
output_tokens = output.outputs[0].token_ids
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
print(f"Output tokens: {output_tokens}")
|
||||
|
||||
acceptance_per_pos = calculate_acceptance_per_pos(metrics, num_speculative_tokens, Counter, Vector)
|
||||
golden = BASELINES[method]
|
||||
|
||||
match = all(abs(a - b) < 0.1 for a, b in zip(acceptance_per_pos, golden))
|
||||
if not match:
|
||||
print(f"acceptance_per_pos: {acceptance_per_pos}")
|
||||
print(f"golden: {golden}")
|
||||
|
||||
assert match
|
||||
@@ -0,0 +1,93 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from transformers import AutoTokenizer
|
||||
from vllm import SamplingParams
|
||||
from vllm.config import CompilationConfig
|
||||
from vllm.tokenizers.registry import resolve_tokenizer_args
|
||||
from vllm.v1.metrics.reader import Counter, Vector
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.pull_request.one_card.spec_decode.utils import (
|
||||
BASELINES,
|
||||
DRAFT_PARALLEL_MODELS,
|
||||
calculate_acceptance_per_pos,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("method", DRAFT_PARALLEL_MODELS.keys())
|
||||
@pytest.mark.parametrize("num_speculative_tokens", [8])
|
||||
@pytest.mark.parametrize("draft_tensor_parallel_size", [1])
|
||||
def test_parallel_drafting_acceptance(
|
||||
method: str,
|
||||
num_speculative_tokens: int,
|
||||
draft_tensor_parallel_size: None | int,
|
||||
):
|
||||
main_model_name = DRAFT_PARALLEL_MODELS[method]["main"]
|
||||
spec_model_name = DRAFT_PARALLEL_MODELS[method]["spec"]
|
||||
|
||||
tokenizer_path = resolve_tokenizer_args(main_model_name)[1]
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
tokenizer_path,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0,
|
||||
ignore_eos=False,
|
||||
max_tokens=256,
|
||||
)
|
||||
|
||||
prompts = [{"role": "user", "content": "Hello, your name is"}]
|
||||
prompts = [
|
||||
tokenizer.apply_chat_template(
|
||||
[prompt],
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
)
|
||||
for prompt in prompts
|
||||
]
|
||||
|
||||
speculative_config = {
|
||||
"method": "draft_model",
|
||||
"model": spec_model_name,
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
"draft_tensor_parallel_size": draft_tensor_parallel_size,
|
||||
"parallel_drafting": True,
|
||||
}
|
||||
|
||||
compilation_config = CompilationConfig(
|
||||
cudagraph_mode="PIECEWISE",
|
||||
cudagraph_capture_sizes=[12],
|
||||
)
|
||||
|
||||
with VllmRunner(
|
||||
main_model_name,
|
||||
max_model_len=4096,
|
||||
disable_log_stats=False,
|
||||
tensor_parallel_size=1,
|
||||
max_num_seqs=256,
|
||||
distributed_executor_backend="mp",
|
||||
gpu_memory_utilization=0.8,
|
||||
speculative_config=speculative_config,
|
||||
compilation_config=compilation_config,
|
||||
enable_prefix_caching=False,
|
||||
) as llm:
|
||||
outputs = llm.model.generate(prompts, sampling_params)
|
||||
metrics = llm.model.get_metrics()
|
||||
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
output_tokens = output.outputs[0].token_ids
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
print(f"Output tokens: {output_tokens}")
|
||||
|
||||
acceptance_per_pos = calculate_acceptance_per_pos(metrics, num_speculative_tokens, Counter, Vector)
|
||||
golden = BASELINES[method]
|
||||
|
||||
match = all(abs(a - b) < 0.1 for a, b in zip(acceptance_per_pos, golden))
|
||||
if not match:
|
||||
print(f"acceptance_per_pos: {acceptance_per_pos}")
|
||||
print(f"golden: {golden}")
|
||||
|
||||
assert match
|
||||
110
tests/e2e/pull_request/one_card/spec_decode/test_eagle.py
Normal file
110
tests/e2e/pull_request/one_card/spec_decode/test_eagle.py
Normal file
@@ -0,0 +1,110 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from transformers import AutoTokenizer
|
||||
from vllm import SamplingParams
|
||||
from vllm.config import CompilationConfig
|
||||
from vllm.tokenizers.registry import resolve_tokenizer_args
|
||||
from vllm.v1.metrics.reader import Counter, Vector
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.pull_request.one_card.spec_decode.utils import MODELS, calculate_acceptance_per_pos
|
||||
|
||||
|
||||
def test_qwen3_vl_eagle(
|
||||
test_prompts: list[list[dict[str, Any]]],
|
||||
sampling_config: SamplingParams,
|
||||
vl_model_name: str,
|
||||
):
|
||||
with VllmRunner(
|
||||
vl_model_name,
|
||||
max_model_len=1024,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
) as ref_llm:
|
||||
ref_llm.model.chat(test_prompts, sampling_config)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("method", MODELS.keys())
|
||||
@pytest.mark.parametrize("num_speculative_tokens", [3])
|
||||
@pytest.mark.parametrize("draft_tensor_parallel_size", [1])
|
||||
@pytest.mark.parametrize("disable_padded_drafter_batch", [False])
|
||||
@pytest.mark.parametrize("async_scheduling", [True])
|
||||
def test_qwen_eagle3_acceptance(
|
||||
method: str,
|
||||
num_speculative_tokens: int,
|
||||
draft_tensor_parallel_size: None | int,
|
||||
disable_padded_drafter_batch: bool,
|
||||
async_scheduling: bool,
|
||||
):
|
||||
main_model_name = MODELS[method]["main"]
|
||||
spec_model_name = MODELS[method]["spec"]
|
||||
|
||||
tokenizer_path = resolve_tokenizer_args(main_model_name)[1]
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
tokenizer_path,
|
||||
trust_remote_code=True,
|
||||
)
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0,
|
||||
ignore_eos=False,
|
||||
max_tokens=256,
|
||||
)
|
||||
|
||||
prompts = [
|
||||
{"role": "user", "content": "Hello, my name is"},
|
||||
{"role": "user", "content": "The president of the United States is"},
|
||||
{"role": "user", "content": "The capital of France is"},
|
||||
{"role": "user", "content": "The future of AI is"},
|
||||
]
|
||||
prompts = [
|
||||
tokenizer.apply_chat_template(
|
||||
[prompt],
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
)
|
||||
for prompt in prompts
|
||||
]
|
||||
|
||||
speculative_config = {
|
||||
"method": method,
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
"draft_tensor_parallel_size": draft_tensor_parallel_size,
|
||||
"disable_padded_drafter_batch": disable_padded_drafter_batch,
|
||||
"model": spec_model_name,
|
||||
}
|
||||
|
||||
compilation_config = CompilationConfig(cudagraph_mode="FULL_DECODE_ONLY", cudagraph_capture_sizes=[12])
|
||||
|
||||
with VllmRunner(
|
||||
main_model_name,
|
||||
max_model_len=2048,
|
||||
disable_log_stats=False,
|
||||
tensor_parallel_size=1,
|
||||
max_num_seqs=256,
|
||||
distributed_executor_backend="mp",
|
||||
gpu_memory_utilization=0.7,
|
||||
speculative_config=speculative_config,
|
||||
compilation_config=compilation_config,
|
||||
async_scheduling=async_scheduling,
|
||||
) as llm:
|
||||
outputs = llm.model.generate(prompts, sampling_params)
|
||||
metrics = llm.model.get_metrics()
|
||||
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
output_tokens = output.outputs[0].token_ids
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
print(f"Output tokens: {output_tokens}")
|
||||
|
||||
acceptance_per_pos = calculate_acceptance_per_pos(metrics, num_speculative_tokens, Counter, Vector)
|
||||
golden = [0.68, 0.40, 0.18]
|
||||
|
||||
match = all(abs(a - b) < 0.08 for a, b in zip(acceptance_per_pos, golden))
|
||||
if not match:
|
||||
print(f"acceptance_per_pos: {acceptance_per_pos}")
|
||||
print(f"golden: {golden}")
|
||||
|
||||
assert match
|
||||
@@ -0,0 +1,213 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""E2E tests for the extract_hidden_states speculative decoding method.
|
||||
|
||||
Follows the pattern from vllm's test_extraction.py, validating that hidden
|
||||
states are correctly extracted and saved on the Ascend NPU. Parametrized over:
|
||||
|
||||
* a dense model (Qwen3-8B) in both eager and ACL graph modes, using real
|
||||
weights so outputs can be checked to be non-zero, and
|
||||
* a hybrid attention model (Qwen3.5-0.8B, GatedDeltaNet + full_attention)
|
||||
loaded with dummy weights as a shape/round-trip smoke test. The hybrid case
|
||||
mirrors upstream vLLM PR #39949.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
from vllm import LLM, SamplingParams
|
||||
|
||||
from vllm_ascend.utils import vllm_version_is
|
||||
|
||||
if vllm_version_is("0.23.0"):
|
||||
from safetensors import safe_open
|
||||
else:
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1 import example_hidden_states_connector
|
||||
|
||||
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
|
||||
DENSE_MODEL = "Qwen/Qwen3-8B"
|
||||
# Qwen3-8B has 36 layers; pick a spread of layer indices to extract.
|
||||
DENSE_AUX_HIDDEN_STATE_LAYER_IDS = [2, 18, 34]
|
||||
|
||||
HYBRID_MODEL = "Qwen/Qwen3.5-0.8B"
|
||||
HYBRID_AUX_HIDDEN_STATE_LAYER_IDS = [5, 11, 17]
|
||||
|
||||
|
||||
@dataclass
|
||||
class ExtractHiddenStatesCase:
|
||||
model_name: str
|
||||
aux_hidden_state_layer_ids: list[int]
|
||||
prompts: list[str]
|
||||
enforce_eager: bool
|
||||
# ``None`` means "do not pass the argument", preserving each model's
|
||||
# original defaults.
|
||||
gpu_memory_utilization: float | None = None
|
||||
max_num_seqs: int | None = None
|
||||
max_model_len: int | None = None
|
||||
load_format: str | None = None
|
||||
# Dummy-weight runs can't assert non-zero outputs; real-weight runs can.
|
||||
verify_nonzero: bool = True
|
||||
# Hybrid smoke test additionally checks the token_ids round-trip.
|
||||
verify_token_ids: bool = False
|
||||
|
||||
|
||||
CASES = [
|
||||
pytest.param(
|
||||
ExtractHiddenStatesCase(
|
||||
model_name=DENSE_MODEL,
|
||||
aux_hidden_state_layer_ids=DENSE_AUX_HIDDEN_STATE_LAYER_IDS,
|
||||
prompts=[
|
||||
"Hello, how are you?",
|
||||
"What is machine learning?",
|
||||
"Explain quantum computing briefly.",
|
||||
],
|
||||
enforce_eager=True,
|
||||
gpu_memory_utilization=0.8,
|
||||
max_num_seqs=16,
|
||||
),
|
||||
id="dense_eager",
|
||||
),
|
||||
pytest.param(
|
||||
ExtractHiddenStatesCase(
|
||||
model_name=DENSE_MODEL,
|
||||
aux_hidden_state_layer_ids=DENSE_AUX_HIDDEN_STATE_LAYER_IDS,
|
||||
prompts=[
|
||||
"Hello, how are you?",
|
||||
"What is machine learning?",
|
||||
],
|
||||
enforce_eager=False,
|
||||
max_num_seqs=16,
|
||||
),
|
||||
id="dense_aclgraph",
|
||||
),
|
||||
pytest.param(
|
||||
ExtractHiddenStatesCase(
|
||||
model_name=HYBRID_MODEL,
|
||||
aux_hidden_state_layer_ids=HYBRID_AUX_HIDDEN_STATE_LAYER_IDS,
|
||||
prompts=[
|
||||
"Hello world",
|
||||
"Test prompt with several tokens",
|
||||
],
|
||||
enforce_eager=True,
|
||||
gpu_memory_utilization=0.4,
|
||||
max_model_len=256,
|
||||
load_format="dummy",
|
||||
verify_nonzero=False,
|
||||
verify_token_ids=True,
|
||||
),
|
||||
id="hybrid_dummy_eager",
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sampling_config():
|
||||
return SamplingParams(temperature=0, max_tokens=1)
|
||||
|
||||
|
||||
def _verify_output(output, expected_shape, *, verify_nonzero, verify_token_ids):
|
||||
"""Verify a single hidden-states dump (matches vllm's check pattern)."""
|
||||
assert output.kv_transfer_params is not None
|
||||
hidden_states_path = output.kv_transfer_params.get("hidden_states_path")
|
||||
assert hidden_states_path is not None
|
||||
|
||||
if vllm_version_is("0.23.0"):
|
||||
assert os.path.exists(hidden_states_path)
|
||||
with safe_open(hidden_states_path, "pt") as f:
|
||||
tensor_names = f.keys()
|
||||
assert "hidden_states" in tensor_names
|
||||
hidden_states = f.get_tensor("hidden_states")
|
||||
assert hidden_states.shape == expected_shape
|
||||
|
||||
if verify_token_ids:
|
||||
token_ids = f.get_tensor("token_ids")
|
||||
assert torch.equal(token_ids, torch.tensor(output.prompt_token_ids))
|
||||
|
||||
if verify_nonzero:
|
||||
assert not torch.allclose(hidden_states, torch.zeros_like(hidden_states))
|
||||
else:
|
||||
obj = example_hidden_states_connector.load_hidden_states(hidden_states_path)
|
||||
example_hidden_states_connector.cleanup_hidden_states(hidden_states_path)
|
||||
hidden_states = obj["hidden_states"]
|
||||
assert hidden_states.shape == expected_shape
|
||||
|
||||
if verify_token_ids:
|
||||
token_ids = obj["token_ids"]
|
||||
assert torch.equal(token_ids, torch.tensor(output.prompt_token_ids))
|
||||
|
||||
if verify_nonzero:
|
||||
assert not torch.allclose(hidden_states, torch.zeros_like(hidden_states))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("case", CASES)
|
||||
def test_extract_hidden_states(case: ExtractHiddenStatesCase, sampling_config):
|
||||
"""Extract hidden states from the target model and validate the dump."""
|
||||
with tempfile.TemporaryDirectory() as tmpdirname:
|
||||
llm_kwargs = dict(
|
||||
model=case.model_name,
|
||||
tensor_parallel_size=1,
|
||||
enforce_eager=case.enforce_eager,
|
||||
enable_chunked_prefill=False,
|
||||
speculative_config={
|
||||
"method": "extract_hidden_states",
|
||||
"num_speculative_tokens": 1,
|
||||
"draft_model_config": {
|
||||
"hf_config": {
|
||||
"eagle_aux_hidden_state_layer_ids": case.aux_hidden_state_layer_ids,
|
||||
}
|
||||
},
|
||||
},
|
||||
kv_transfer_config={
|
||||
"kv_connector": "ExampleHiddenStatesConnector",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_connector_extra_config": {
|
||||
"shared_storage_path": tmpdirname,
|
||||
},
|
||||
},
|
||||
)
|
||||
if case.gpu_memory_utilization is not None:
|
||||
llm_kwargs["gpu_memory_utilization"] = case.gpu_memory_utilization
|
||||
if case.max_num_seqs is not None:
|
||||
llm_kwargs["max_num_seqs"] = case.max_num_seqs
|
||||
if case.max_model_len is not None:
|
||||
llm_kwargs["max_model_len"] = case.max_model_len
|
||||
if case.load_format is not None:
|
||||
llm_kwargs["load_format"] = case.load_format
|
||||
|
||||
llm = LLM(**llm_kwargs)
|
||||
|
||||
outputs = llm.generate(case.prompts, sampling_config)
|
||||
hidden_size = llm.llm_engine.model_config.get_hidden_size()
|
||||
num_layers = len(case.aux_hidden_state_layer_ids)
|
||||
|
||||
assert len(outputs) == len(case.prompts)
|
||||
|
||||
for output in outputs:
|
||||
num_tokens = len(output.prompt_token_ids)
|
||||
expected_shape = (num_tokens, num_layers, hidden_size)
|
||||
_verify_output(
|
||||
output,
|
||||
expected_shape,
|
||||
verify_nonzero=case.verify_nonzero,
|
||||
verify_token_ids=case.verify_token_ids,
|
||||
)
|
||||
@@ -0,0 +1,75 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2025 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
|
||||
#
|
||||
"""Compare the short outputs of HF and vLLM when using greedy sampling."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
from vllm.config import CompilationConfig
|
||||
|
||||
from tests.e2e.conftest import VllmRunner, cleanup_dist_env_and_memory
|
||||
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
|
||||
MODELS = ["wemaster/deepseek_mtp_main_random_bf16"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model_name", MODELS)
|
||||
@pytest.mark.parametrize("num_speculative_tokens", [3])
|
||||
@pytest.mark.parametrize("cudagraph_mode", ["FULL_DECODE_ONLY"])
|
||||
@pytest.mark.parametrize("disable_padded_drafter_batch", [False])
|
||||
def test_deepseek_mtp(
|
||||
model_name: str, num_speculative_tokens: int, cudagraph_mode: str, disable_padded_drafter_batch: bool
|
||||
):
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
"""
|
||||
Compare the outputs of a original LLM and a speculative LLM
|
||||
should be the same when using mtp speculative decoding.
|
||||
"""
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
tensor_parallel_size=1,
|
||||
max_num_seqs=256,
|
||||
gpu_memory_utilization=0.7,
|
||||
distributed_executor_backend="mp",
|
||||
enable_expert_parallel=True,
|
||||
speculative_config={
|
||||
"method": "mtp",
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
"disable_padded_drafter_batch": disable_padded_drafter_batch,
|
||||
},
|
||||
max_model_len=2000,
|
||||
compilation_config=CompilationConfig(
|
||||
cudagraph_mode=cudagraph_mode,
|
||||
cudagraph_capture_sizes=[20],
|
||||
),
|
||||
) as spec_llm:
|
||||
sampling_config = SamplingParams(temperature=0, max_tokens=256, ignore_eos=False)
|
||||
spec_llm.generate(example_prompts, sampling_config)
|
||||
|
||||
cleanup_dist_env_and_memory()
|
||||
del spec_llm
|
||||
26
tests/e2e/pull_request/one_card/spec_decode/test_ngram.py
Normal file
26
tests/e2e/pull_request/one_card/spec_decode/test_ngram.py
Normal file
@@ -0,0 +1,26 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from vllm import SamplingParams
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
|
||||
def test_ngram(
|
||||
test_prompts: list[list[dict[str, Any]]],
|
||||
sampling_config: SamplingParams,
|
||||
model_name: str,
|
||||
):
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
speculative_config={
|
||||
"method": "ngram",
|
||||
"prompt_lookup_max": 5,
|
||||
"prompt_lookup_min": 3,
|
||||
"num_speculative_tokens": 3,
|
||||
},
|
||||
max_model_len=1024,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
) as runner:
|
||||
runner.model.chat(test_prompts, sampling_config)
|
||||
@@ -0,0 +1,68 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
from vllm.config import CompilationConfig
|
||||
from vllm.v1.metrics.reader import Counter, Vector
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.pull_request.one_card.spec_decode.utils import calculate_acceptance_per_pos
|
||||
|
||||
|
||||
@pytest.mark.parametrize("num_speculative_tokens", [3])
|
||||
def test_ngram_npu_async_acceptance(
|
||||
test_prompts: list[list[dict[str, Any]]],
|
||||
num_speculative_tokens: int,
|
||||
model_name: str,
|
||||
):
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0,
|
||||
ignore_eos=False,
|
||||
max_tokens=256,
|
||||
)
|
||||
|
||||
speculative_config = {
|
||||
"method": "ngram_gpu",
|
||||
"prompt_lookup_max": 2,
|
||||
"prompt_lookup_min": 2,
|
||||
"num_speculative_tokens": num_speculative_tokens,
|
||||
}
|
||||
|
||||
compilation_config = CompilationConfig(
|
||||
cudagraph_mode="PIECEWISE",
|
||||
cudagraph_capture_sizes=[12],
|
||||
)
|
||||
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
max_model_len=2048,
|
||||
disable_log_stats=False,
|
||||
tensor_parallel_size=1,
|
||||
max_num_seqs=256,
|
||||
distributed_executor_backend="mp",
|
||||
gpu_memory_utilization=0.7,
|
||||
speculative_config=speculative_config,
|
||||
compilation_config=compilation_config,
|
||||
async_scheduling=True,
|
||||
) as llm:
|
||||
outputs = llm.model.chat(test_prompts, sampling_params)
|
||||
metrics = llm.model.get_metrics()
|
||||
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
output_tokens = output.outputs[0].token_ids
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
print(f"Output tokens: {output_tokens}")
|
||||
|
||||
acceptance_per_pos = calculate_acceptance_per_pos(metrics, num_speculative_tokens, Counter, Vector)
|
||||
golden = [0.50, 0.30, 0.20]
|
||||
|
||||
match = all(abs(a - b) < 1.0 for a, b in zip(acceptance_per_pos, golden))
|
||||
if not match:
|
||||
print(f"acceptance_per_pos: {acceptance_per_pos}")
|
||||
print(f"golden: {golden}")
|
||||
|
||||
assert match
|
||||
49
tests/e2e/pull_request/one_card/spec_decode/test_suffix.py
Normal file
49
tests/e2e/pull_request/one_card/spec_decode/test_suffix.py
Normal file
@@ -0,0 +1,49 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from vllm import SamplingParams
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
|
||||
def test_suffix_acceptance(
|
||||
test_prompts: list[list[dict[str, Any]]],
|
||||
sampling_config: SamplingParams,
|
||||
model_name: str,
|
||||
):
|
||||
num_draft = []
|
||||
num_accept = []
|
||||
with VllmRunner(
|
||||
model_name,
|
||||
speculative_config={
|
||||
"method": "suffix",
|
||||
"suffix_decoding_max_spec_factor": 2.0,
|
||||
"suffix_decoding_max_cached_requests": 1000,
|
||||
"num_speculative_tokens": 10,
|
||||
},
|
||||
max_model_len=1024,
|
||||
max_cudagraph_capture_size=16,
|
||||
disable_log_stats=False,
|
||||
) as runner:
|
||||
for i in range(10):
|
||||
runner.model.chat(test_prompts[i], sampling_config)
|
||||
metrics = runner.model.get_metrics()
|
||||
for metric in metrics:
|
||||
print(metric)
|
||||
if metric.name == "vllm:spec_decode_num_draft_tokens":
|
||||
num_draft.append(metric.value)
|
||||
if metric.name == "vllm:spec_decode_num_accepted_tokens":
|
||||
num_accept.append(metric.value)
|
||||
|
||||
first_accept_tokens = num_accept[0]
|
||||
first_draft_tokens = num_draft[0]
|
||||
first_accept_rate = first_accept_tokens / first_draft_tokens
|
||||
|
||||
last_accept_tokens = num_accept[-1] - num_accept[-2]
|
||||
last_draft_tokens = num_draft[-1] - num_draft[-2]
|
||||
last_accept_rate = last_accept_tokens / last_draft_tokens
|
||||
|
||||
assert first_accept_tokens < last_accept_tokens
|
||||
assert first_accept_rate < last_accept_rate
|
||||
assert last_accept_rate > 0.60
|
||||
61
tests/e2e/pull_request/one_card/spec_decode/utils.py
Normal file
61
tests/e2e/pull_request/one_card/spec_decode/utils.py
Normal file
@@ -0,0 +1,61 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Any
|
||||
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
|
||||
MODELS = {
|
||||
"eagle3": {
|
||||
"main": "Qwen/Qwen3-8B",
|
||||
"spec": "RedHatAI/Qwen3-8B-speculator.eagle3",
|
||||
},
|
||||
}
|
||||
|
||||
DRAFT_PARALLEL_MODELS = {
|
||||
"draft_parallel": {
|
||||
"main": "LLM-Research/Meta-Llama-3.1-8B-Instruct",
|
||||
"spec": "amd/PARD-Llama-3.2-1B",
|
||||
},
|
||||
}
|
||||
|
||||
DFLASH = {
|
||||
"dflash": {
|
||||
"main": "Qwen/Qwen3-8B",
|
||||
"spec": "z-lab/Qwen3-8B-DFlash-b16",
|
||||
}
|
||||
}
|
||||
|
||||
BASELINES = {
|
||||
"eagle": [0.74, 0.44, 0.29],
|
||||
"eagle3": [0.68, 0.40, 0.18],
|
||||
"draft_parallel": [0.83, 0.50, 0.33, 0.17, 0.17, 0.17, 0.17, 0.00],
|
||||
"dflash": [0.60, 0.50, 0.30, 0.20, 0.20, 0.10, 0.00, 0.00],
|
||||
}
|
||||
|
||||
|
||||
def eagle_model_name():
|
||||
return "vllm-ascend/EAGLE-LLaMA3.1-Instruct-8B"
|
||||
|
||||
|
||||
def eagle3_model_name():
|
||||
return "vllm-ascend/EAGLE3-LLaMA3.1-Instruct-8B"
|
||||
|
||||
|
||||
def vl_eagle3_model_name():
|
||||
return "MNN/Qwen3-VL-8B-Instruct-Eagle3"
|
||||
|
||||
|
||||
def calculate_acceptance_per_pos(metrics: list[Any], num_speculative_tokens: int, counter_type: Any, vector_type: Any):
|
||||
num_drafts = 0
|
||||
num_accepted_tokens_per_pos = [0] * num_speculative_tokens
|
||||
for metric in metrics:
|
||||
if metric.name == "vllm:spec_decode_num_drafts":
|
||||
assert isinstance(metric, counter_type)
|
||||
num_drafts += metric.value
|
||||
elif metric.name == "vllm:spec_decode_num_accepted_tokens_per_pos":
|
||||
assert isinstance(metric, vector_type)
|
||||
for pos in range(len(metric.values)):
|
||||
num_accepted_tokens_per_pos[pos] += metric.values[pos]
|
||||
|
||||
return [num_accepted_tokens / num_drafts for num_accepted_tokens in num_accepted_tokens_per_pos]
|
||||
153
tests/e2e/pull_request/one_card/test_attention_fa3.py
Normal file
153
tests/e2e/pull_request/one_card/test_attention_fa3.py
Normal file
@@ -0,0 +1,153 @@
|
||||
"""
|
||||
End-to-end test for Flash Attention 3 (FA3) on Ascend.
|
||||
|
||||
This test verifies that FA3 produces correct results by comparing its output
|
||||
with the default Fused Infer Attention (FIA) backend. Both backends are run
|
||||
with the same model and prompts in eager mode, and the generated token ids
|
||||
are compared for consistency.
|
||||
"""
|
||||
|
||||
import os
|
||||
from importlib import import_module, util
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
os.environ["VLLM_BATCH_INVARIANT"] = "1"
|
||||
|
||||
MODEL_NAME = "Qwen/Qwen3-0.6B"
|
||||
MAX_MODEL_LEN = 512
|
||||
MAX_TOKENS = 32
|
||||
|
||||
SHORT_PROMPTS = [
|
||||
"The capital of France is",
|
||||
"In a hole in the ground there lived a hobbit.",
|
||||
"The meaning of life is",
|
||||
"To be or not to be, that is the",
|
||||
]
|
||||
|
||||
LONG_PROMPT = "The quick brown fox jumps over the lazy dog. " * 50
|
||||
|
||||
|
||||
def _fa3_available() -> bool:
|
||||
try:
|
||||
if util.find_spec("flash_attn_npu_v3") is None:
|
||||
return False
|
||||
mod = import_module("flash_attn_npu_v3")
|
||||
return hasattr(mod, "flash_attn_with_kvcache")
|
||||
except ImportError:
|
||||
return False
|
||||
|
||||
|
||||
def _generate_with_backend(prompts, max_tokens=MAX_TOKENS, **runner_kwargs):
|
||||
with VllmRunner(
|
||||
MODEL_NAME,
|
||||
max_model_len=MAX_MODEL_LEN,
|
||||
enforce_eager=True,
|
||||
gpu_memory_utilization=0.7,
|
||||
**runner_kwargs,
|
||||
) as runner:
|
||||
return runner.generate_greedy(prompts, max_tokens)
|
||||
|
||||
|
||||
def _generate_logprobs_with_backend(prompts, max_tokens=5, num_logprobs=5, **runner_kwargs):
|
||||
with VllmRunner(
|
||||
MODEL_NAME,
|
||||
max_model_len=MAX_MODEL_LEN,
|
||||
enforce_eager=True,
|
||||
gpu_memory_utilization=0.7,
|
||||
**runner_kwargs,
|
||||
) as runner:
|
||||
return runner.generate_greedy_logprobs(prompts, max_tokens=max_tokens, num_logprobs=num_logprobs)
|
||||
|
||||
|
||||
def _assert_outputs_match(fia_outputs, fa3_outputs, label=""):
|
||||
for i, (fia_out, fa3_out) in enumerate(zip(fia_outputs, fa3_outputs)):
|
||||
fia_ids, fia_text = fia_out
|
||||
fa3_ids, fa3_text = fa3_out
|
||||
assert fia_ids == fa3_ids, (
|
||||
f"{label}Prompt {i}: FA3 and FIA token ids differ.\n"
|
||||
f" FIA ids: {fia_ids}\n"
|
||||
f" FA3 ids: {fa3_ids}\n"
|
||||
f" FIA text: {fia_text}\n"
|
||||
f" FA3 text: {fa3_text}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.skipif(not _fa3_available(), reason="flash_attn_npu_v3 is not installed")
|
||||
def test_fa3_vs_fia_single_prompt():
|
||||
"""Compare FA3 and FIA with a single prompt (minimal batch size).
|
||||
|
||||
This is an edge case that verifies FA3 works correctly when there
|
||||
is only one sequence in the batch, which may exercise different
|
||||
code paths in the attention kernel.
|
||||
"""
|
||||
single_prompt = ["Explain quantum computing in simple terms."]
|
||||
fia_outputs = _generate_with_backend(single_prompt)
|
||||
fa3_outputs = _generate_with_backend(single_prompt, attention_backend="FLASH_ATTN")
|
||||
_assert_outputs_match(fia_outputs, fa3_outputs, label="[SinglePrompt] ")
|
||||
|
||||
|
||||
@pytest.mark.skipif(not _fa3_available(), reason="flash_attn_npu_v3 is not installed")
|
||||
def test_fa3_vs_fia_mixed_lengths():
|
||||
"""Compare FA3 and FIA with mixed prompt lengths in the same batch.
|
||||
|
||||
This exercises both prefill and decode paths within a single batch,
|
||||
verifying that FA3 handles variable-length sequences correctly.
|
||||
"""
|
||||
mixed_prompts = [
|
||||
"Hi",
|
||||
"The capital of France is",
|
||||
LONG_PROMPT[:256],
|
||||
"What is 2+2?",
|
||||
LONG_PROMPT[:MAX_MODEL_LEN],
|
||||
]
|
||||
fia_outputs = _generate_with_backend(mixed_prompts)
|
||||
fa3_outputs = _generate_with_backend(mixed_prompts, attention_backend="FLASH_ATTN")
|
||||
_assert_outputs_match(fia_outputs, fa3_outputs, label="[MixedLen] ")
|
||||
|
||||
|
||||
@pytest.mark.skipif(not _fa3_available(), reason="flash_attn_npu_v3 is not installed")
|
||||
def test_fa3_vs_fia_with_chunkprefill():
|
||||
"""Compare FA3 and FIA with single token generation where chunkprefill is used."""
|
||||
fia_outputs = _generate_with_backend(SHORT_PROMPTS, max_tokens=2, max_num_seqs=2, max_num_batched_tokens=5)
|
||||
fa3_outputs = _generate_with_backend(
|
||||
SHORT_PROMPTS, attention_backend="FLASH_ATTN", max_tokens=2, max_num_seqs=2, max_num_batched_tokens=5
|
||||
)
|
||||
_assert_outputs_match(fia_outputs, fa3_outputs, label="[Chunkprefill] ")
|
||||
|
||||
|
||||
@pytest.mark.skipif(not _fa3_available(), reason="flash_attn_npu_v3 is not installed")
|
||||
def test_fa3_vs_fia_logprobs():
|
||||
"""Compare FA3 and FIA logprobs for fine-grained numerical verification."""
|
||||
fia_logprobs = _generate_logprobs_with_backend(SHORT_PROMPTS[:1])
|
||||
fa3_logprobs = _generate_logprobs_with_backend(SHORT_PROMPTS[:1], attention_backend="FLASH_ATTN")
|
||||
|
||||
for i, (fia_out, fa3_out) in enumerate(zip(fia_logprobs, fa3_logprobs)):
|
||||
fia_ids, _, fia_lp = fia_out
|
||||
fa3_ids, _, fa3_lp = fa3_out
|
||||
|
||||
assert fia_ids == fa3_ids, (
|
||||
f"Prompt {i}: FA3 and FIA token ids differ.\n FIA ids: {fia_ids}\n FA3 ids: {fa3_ids}"
|
||||
)
|
||||
|
||||
assert len(fia_lp) == len(fa3_lp), (
|
||||
f"Prompt {i}: Different number of logprob steps: FIA {len(fia_lp)} vs FA3 {len(fa3_lp)}"
|
||||
)
|
||||
|
||||
for t, (fia_token_lp, fa3_token_lp) in enumerate(zip(fia_lp, fa3_lp)):
|
||||
assert set(fia_token_lp.keys()) == set(fa3_token_lp.keys()), (
|
||||
f"Prompt {i}, token {t}: Logprob token sets differ.\n"
|
||||
f" FIA keys: {sorted(fia_token_lp.keys())}\n"
|
||||
f" FA3 keys: {sorted(fa3_token_lp.keys())}"
|
||||
)
|
||||
for token_id in fia_token_lp:
|
||||
fia_logprob = fia_token_lp[token_id].logprob
|
||||
fa3_logprob = fa3_token_lp[token_id].logprob
|
||||
assert abs(fia_logprob - fa3_logprob) < 1e-3, (
|
||||
f"Prompt {i}, token {t}, token_id {token_id} ("
|
||||
f"'{fia_token_lp[token_id].decoded_token}'): "
|
||||
f"logprobs differ: FIA {fia_logprob:.6f} vs FA3 {fa3_logprob:.6f}"
|
||||
)
|
||||
641
tests/e2e/pull_request/one_card/test_batch_invariant.py
Normal file
641
tests/e2e/pull_request/one_card/test_batch_invariant.py
Normal file
@@ -0,0 +1,641 @@
|
||||
# Adapt from https://github.com/vllm-project/vllm/blob/main/tests/v1/determinism/test_batch_invariant.py
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
import gc
|
||||
import os
|
||||
import random
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
from vllm import LLM, SamplingParams
|
||||
|
||||
from tests.e2e.conftest import ModelName, cleanup_dist_env_and_memory, model_cache
|
||||
|
||||
os.environ["VLLM_BATCH_INVARIANT"] = "1"
|
||||
|
||||
DEFAULT_MODEL = ModelName.QWEN3_06B
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def enable_batch_invariant_mode(monkeypatch: pytest.MonkeyPatch):
|
||||
"""Automatically enable batch invariant kernel overrides for all tests."""
|
||||
monkeypatch.setenv("VLLM_BATCH_INVARIANT", "1")
|
||||
|
||||
|
||||
def _random_prompt(min_words: int = 1024, max_words: int = 1024 * 2) -> str:
|
||||
# Generate more realistic prompts that will actually produce varied tokens
|
||||
# Use a mix of common English text patterns
|
||||
|
||||
prompt_templates = [
|
||||
# Question-answer style
|
||||
"Question: What is the capital of France?\nAnswer: The capital of France is",
|
||||
"Q: How does photosynthesis work?\nA: Photosynthesis is the process by which",
|
||||
"User: Can you explain quantum mechanics?\nAssistant: Quantum mechanics is",
|
||||
# Story/narrative style
|
||||
"Once upon a time in a distant galaxy, there lived",
|
||||
"The old man walked slowly down the street, remembering",
|
||||
"In the year 2157, humanity finally discovered",
|
||||
# Technical/code style
|
||||
"To implement a binary search tree in Python, first we need to",
|
||||
"The algorithm works by iterating through the array and",
|
||||
"Here's how to optimize database queries using indexing:",
|
||||
# Factual/informative style
|
||||
"The Renaissance was a period in European history that",
|
||||
"Climate change is caused by several factors including",
|
||||
"The human brain contains approximately 86 billion neurons which",
|
||||
# Conversational style
|
||||
"I've been thinking about getting a new laptop because",
|
||||
"Yesterday I went to the store and bought",
|
||||
"My favorite thing about summer is definitely",
|
||||
]
|
||||
|
||||
# Pick a random template
|
||||
base_prompt = random.choice(prompt_templates)
|
||||
|
||||
if max_words < min_words:
|
||||
max_words = min_words
|
||||
target_words = random.randint(min_words, max_words)
|
||||
|
||||
if target_words > 50:
|
||||
# For longer prompts, repeat context
|
||||
padding_text = " This is an interesting topic that deserves more explanation. " * (target_words // 50)
|
||||
base_prompt = base_prompt + padding_text
|
||||
|
||||
return base_prompt
|
||||
|
||||
|
||||
def _extract_step_logprobs(request_output):
|
||||
if getattr(request_output, "outputs", None):
|
||||
inner = request_output.outputs[0]
|
||||
if hasattr(inner, "logprobs") and inner.logprobs is not None:
|
||||
t = torch.tensor(
|
||||
[inner.logprobs[i][tid].logprob for i, tid in enumerate(inner.token_ids)],
|
||||
dtype=torch.float32,
|
||||
)
|
||||
return t, inner.token_ids
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=DEFAULT_MODEL,
|
||||
max_num_seqs=int(os.getenv("VLLM_NEEDLE_BATCH_SIZE", "64")),
|
||||
gpu_memory_utilization=float(os.getenv("VLLM_GPU_MEMORY_UTILIZATION", "0.95")),
|
||||
max_model_len=int(os.getenv("VLLM_MAX_MODEL_LEN", "8192")),
|
||||
dtype="bfloat16",
|
||||
tensor_parallel_size=int(os.getenv("VLLM_TP_SIZE", "1")),
|
||||
enable_prefix_caching=False,
|
||||
distributed_executor_backend="mp",
|
||||
compilation_config={
|
||||
"cudagraph_mode": "PIECEWISE",
|
||||
"cudagraph_capture_sizes": [1, 32, 64],
|
||||
},
|
||||
)
|
||||
def test_v1_generation_is_deterministic_across_batch_sizes_with_needle(
|
||||
vllm_runner,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
):
|
||||
"""
|
||||
Ensures that the same request (the 'needle' prompt) yields identical output
|
||||
whether run alone (bs=1) or mixed into a larger batch (e.g., bs=64),
|
||||
using the high-level v1 LLM() API only (no manual batching).
|
||||
|
||||
Strategy:
|
||||
- Create two LLM engines with identical config except max_num_seqs: 1 vs N.
|
||||
- Compute a baseline output for the needle prompt with the bs=1 engine.
|
||||
- For many trials, generate a batch (size N) where the needle appears at a
|
||||
random position among random filler prompts using the bs=N engine.
|
||||
- Track how many trials match vs mismatch, and report totals at the end.
|
||||
The test fails if any mismatches occur, but we still dump pass/fail
|
||||
counts.
|
||||
|
||||
Notes:
|
||||
- Use seeded stochastic sampling with a fixed seed to test determinism.
|
||||
- Outputs are intentionally longer and sampled at higher temperature/top_p
|
||||
to produce a more random-sounding phrase, yet remain deterministic by
|
||||
seed.
|
||||
- Keep max_tokens and max_model_len bounded for speed and memory use.
|
||||
"""
|
||||
seed = int(os.getenv("VLLM_TEST_SEED", "12345"))
|
||||
random.seed(seed)
|
||||
|
||||
# Allow overrides from environment (useful for CI tuning)
|
||||
num_trials = int(os.getenv("VLLM_NEEDLE_TRIALS", "5"))
|
||||
max_batch_size = int(os.getenv("VLLM_NEEDLE_BATCH_SIZE", "144"))
|
||||
min_random_prompt = int(os.getenv("VLLM_MIN_PROMPT", "1024"))
|
||||
max_random_prompt = int(os.getenv("VLLM_MAX_PROMPT", "2048"))
|
||||
assert max_batch_size >= 2, "Batch size should be >= 2 to mix needle."
|
||||
|
||||
# Sampling parameters: longer outputs with a more random-sounding
|
||||
# continuation,but still deterministic due to fixed seed.
|
||||
temperature = float(os.getenv("VLLM_NEEDLE_TEMPERATURE", "0.0"))
|
||||
top_p = float(os.getenv("VLLM_NEEDLE_TOP_P", "0.95"))
|
||||
max_tokens = int(os.getenv("VLLM_NEEDLE_MAX_TOKENS", "35"))
|
||||
|
||||
sampling = SamplingParams(
|
||||
temperature=temperature,
|
||||
top_p=top_p,
|
||||
max_tokens=max_tokens,
|
||||
seed=20240919,
|
||||
)
|
||||
|
||||
needle_prompt = "There once was a "
|
||||
|
||||
# Baseline generation for the needle prompt alone.
|
||||
baseline_out = vllm_runner.model.generate([needle_prompt], sampling)
|
||||
assert len(baseline_out) == 1
|
||||
assert len(baseline_out[0].outputs) >= 1
|
||||
baseline_text = baseline_out[0].outputs[0].text
|
||||
|
||||
mismatches = 0
|
||||
|
||||
for trial in range(num_trials):
|
||||
# Create a batch of size `max_batch_size` and insert the needle at
|
||||
# a random index
|
||||
prompts: list[str] = []
|
||||
batch_size = random.randint(max_batch_size // 2, max_batch_size)
|
||||
needle_pos = random.randint(0, batch_size - 1)
|
||||
for i in range(batch_size):
|
||||
if i == needle_pos:
|
||||
prompts.append(needle_prompt)
|
||||
else:
|
||||
prompts.append(_random_prompt(min_random_prompt, max_random_prompt))
|
||||
|
||||
# Generate with the larger-batch engine
|
||||
outputs = vllm_runner.model.generate(prompts, sampling)
|
||||
# Find the needle output by position
|
||||
needle_output = outputs[needle_pos]
|
||||
assert needle_output.prompt == needle_prompt
|
||||
assert len(needle_output.outputs) >= 1
|
||||
text = needle_output.outputs[0].text
|
||||
|
||||
if text != baseline_text:
|
||||
print(f"{text}\n\n== Not the same as ==\n\n{baseline_text}\n\n")
|
||||
mismatches += 1
|
||||
|
||||
passes = num_trials - mismatches
|
||||
# Dump how many passed vs failed
|
||||
print(f"[determinism] total={num_trials}, passed={passes}, failed={mismatches}, max_batch_size={max_batch_size}")
|
||||
|
||||
if mismatches > 0:
|
||||
pytest.fail(
|
||||
f"Nondeterministic outputs detected: {mismatches} failed out "
|
||||
f"of {num_trials} trials (max_batch_size={max_batch_size})."
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.model(
|
||||
model_name=DEFAULT_MODEL,
|
||||
max_num_seqs=144,
|
||||
gpu_memory_utilization=float(os.getenv("VLLM_GPU_MEMORY_UTILIZATION", "0.95")),
|
||||
max_model_len=8192,
|
||||
dtype="bfloat16",
|
||||
tensor_parallel_size=int(os.getenv("VLLM_TEST_TP_SIZE", "1")),
|
||||
enable_prefix_caching=False,
|
||||
distributed_executor_backend="mp",
|
||||
compilation_config={
|
||||
"cudagraph_mode": "PIECEWISE",
|
||||
"cudagraph_capture_sizes": [1, 32, 64],
|
||||
},
|
||||
)
|
||||
def test_logprobs_bitwise_batch_invariance_bs1_vs_bsN(vllm_runner, monkeypatch: pytest.MonkeyPatch):
|
||||
seed = int(os.getenv("VLLM_TEST_SEED", "12345"))
|
||||
random.seed(seed)
|
||||
tp_size = int(os.getenv("VLLM_TEST_TP_SIZE", "1"))
|
||||
|
||||
# For batch invariance, disable custom all-reduce to ensure deterministic
|
||||
# all-reduce operations (custom all-reduce may not be deterministic)
|
||||
import vllm.envs as envs
|
||||
|
||||
disable_custom_ar = envs.VLLM_BATCH_INVARIANT
|
||||
|
||||
if disable_custom_ar:
|
||||
print(f"\n{'=' * 80}")
|
||||
print(f"BATCH INVARIANCE MODE: Disabling custom all-reduce (TP={tp_size})")
|
||||
print(f"{'=' * 80}\n")
|
||||
|
||||
# Use more realistic prompts for better token generation
|
||||
prompts = [_random_prompt(10, 50) for i in range(32)]
|
||||
|
||||
sp = SamplingParams(
|
||||
temperature=0.6,
|
||||
top_p=1.0,
|
||||
max_tokens=8,
|
||||
seed=1234,
|
||||
logprobs=5,
|
||||
)
|
||||
|
||||
# BS=1: run prompts individually and collect logprobs per step.
|
||||
print("\n" + "=" * 80)
|
||||
print("STARTING BS=1 RUNS (each prompt individually)")
|
||||
print("=" * 80 + "\n")
|
||||
|
||||
bs1_logprobs_per_prompt = []
|
||||
bs1_tokens_per_prompt = []
|
||||
for idx, p in enumerate(prompts):
|
||||
print(f"\n[BS=1] Running prompt {idx}/{len(prompts)} - Preview: {p[:80]}...")
|
||||
outs = vllm_runner.model.generate([p], sp, use_tqdm=False)
|
||||
assert len(outs) == 1
|
||||
step_logprobs, token_ids = _extract_step_logprobs(outs[0])
|
||||
if step_logprobs is None:
|
||||
pytest.skip("Logits are not available on RequestOutput; enable logprobs return to run this test.")
|
||||
bs1_logprobs_per_prompt.append(step_logprobs)
|
||||
bs1_tokens_per_prompt.append(token_ids)
|
||||
print(f"[BS=1] Prompt {idx} generated tokens: {token_ids}")
|
||||
|
||||
# BS=N: run prompts in a batch and collect logprobs per step for each
|
||||
# prompt.
|
||||
print("\n" + "=" * 80)
|
||||
print(f"STARTING BS={len(prompts)} RUN (all prompts batched)")
|
||||
print("=" * 80 + "\n")
|
||||
|
||||
outs_batched = vllm_runner.model.generate(prompts, sp, use_tqdm=False)
|
||||
assert len(outs_batched) == len(prompts)
|
||||
bsN_logprobs_per_prompt = []
|
||||
bsN_tokens_per_prompt = []
|
||||
|
||||
print(f"\n[BS={len(prompts)}] Processing batched outputs...")
|
||||
for idx, o in enumerate(outs_batched):
|
||||
tokens = o.outputs[0].token_ids if o.outputs else "N/A"
|
||||
print(f"[BS={len(prompts)}] Prompt {idx} generated tokens: {tokens}")
|
||||
step_logprobs, token_ids = _extract_step_logprobs(o)
|
||||
if step_logprobs is None:
|
||||
pytest.skip("Logits are not available on RequestOutput; enable logprobs return to run this test.")
|
||||
bsN_logprobs_per_prompt.append(step_logprobs)
|
||||
bsN_tokens_per_prompt.append(token_ids)
|
||||
|
||||
# Compare step-by-step logprobs for each prompt between BS=1 and BS=N runs.
|
||||
failed_prompts = []
|
||||
for i, (logprobs_bs1, logprobs_bsN, tokens_bs1, tokens_bsN) in enumerate(
|
||||
zip(
|
||||
bs1_logprobs_per_prompt,
|
||||
bsN_logprobs_per_prompt,
|
||||
bs1_tokens_per_prompt,
|
||||
bsN_tokens_per_prompt,
|
||||
)
|
||||
):
|
||||
if len(logprobs_bs1) != len(logprobs_bsN):
|
||||
reason = f"Different number of steps: {len(logprobs_bs1)} (BS=1) vs {len(logprobs_bsN)} (BS=N)"
|
||||
failed_prompts.append(
|
||||
{
|
||||
"prompt_idx": i,
|
||||
"step": "all",
|
||||
"reason": reason,
|
||||
"prompt_preview": prompts[i][:100],
|
||||
"bs1_tokens": tokens_bs1,
|
||||
"bsN_tokens": tokens_bsN,
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
# Check if tokens match first
|
||||
if tokens_bs1 != tokens_bsN:
|
||||
failed_prompts.append(
|
||||
{
|
||||
"prompt_idx": i,
|
||||
"step": "sampling",
|
||||
"reason": "Different tokens sampled",
|
||||
"prompt_preview": prompts[i][:100],
|
||||
"bs1_tokens": tokens_bs1,
|
||||
"bsN_tokens": tokens_bsN,
|
||||
"bs1_all_logprobs": [logprobs_bs1[s].tolist() for s in range(len(logprobs_bs1))],
|
||||
"bsN_all_logprobs": [logprobs_bsN[s].tolist() for s in range(len(logprobs_bsN))],
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
for t, (a, b) in enumerate(zip(logprobs_bs1, logprobs_bsN)):
|
||||
if a.shape != b.shape:
|
||||
failed_prompts.append(
|
||||
{
|
||||
"prompt_idx": i,
|
||||
"step": t,
|
||||
"reason": f"Shape mismatch: {a.shape} vs {b.shape}",
|
||||
"prompt_preview": prompts[i][:100],
|
||||
"bs1_tokens": tokens_bs1,
|
||||
"bsN_tokens": tokens_bsN,
|
||||
}
|
||||
)
|
||||
break
|
||||
|
||||
if not torch.equal(a, b):
|
||||
max_diff = torch.abs(a - b).max().item()
|
||||
# Print which token failed
|
||||
print(f"\n[DIVERGENCE] Prompt {i}, Token {t}: max_diff={max_diff:.6e}")
|
||||
bs1_tok = tokens_bs1[t] if t < len(tokens_bs1) else "N/A"
|
||||
bsN_tok = tokens_bsN[t] if t < len(tokens_bsN) else "N/A"
|
||||
print(f" Token IDs: bs1={bs1_tok}, bsN={bsN_tok}")
|
||||
print(f" BS=1 logprob: {a.tolist()}")
|
||||
print(f" BS=N logprob: {b.tolist()}")
|
||||
failed_prompts.append(
|
||||
{
|
||||
"prompt_idx": i,
|
||||
"step": t,
|
||||
"reason": f"Bitwise mismatch (max_diff={max_diff:.6e})",
|
||||
"prompt_preview": prompts[i][:100],
|
||||
"bs1_tokens": tokens_bs1,
|
||||
"bsN_tokens": tokens_bsN,
|
||||
"bs1_all_logprobs": [logprobs_bs1[s].tolist() for s in range(len(logprobs_bs1))],
|
||||
"bsN_all_logprobs": [logprobs_bsN[s].tolist() for s in range(len(logprobs_bsN))],
|
||||
}
|
||||
)
|
||||
break
|
||||
|
||||
# Print summary of all failures
|
||||
if failed_prompts:
|
||||
print(f"\n{'=' * 80}")
|
||||
fail_msg = f"BATCH INVARIANCE FAILURES: {len(failed_prompts)}/{len(prompts)} prompts failed"
|
||||
print(fail_msg)
|
||||
print(f"{'=' * 80}")
|
||||
for fail in failed_prompts:
|
||||
print(f"\nPrompt {fail['prompt_idx']} (step {fail['step']}):")
|
||||
print(f" Reason: {fail['reason']}")
|
||||
print(f" Preview: {fail['prompt_preview']}...")
|
||||
|
||||
# Always show the tokens
|
||||
if "bs1_tokens" in fail:
|
||||
print(f" BS=1 tokens: {fail['bs1_tokens']}")
|
||||
if "bsN_tokens" in fail:
|
||||
print(f" BS=N tokens: {fail['bsN_tokens']}")
|
||||
|
||||
if "bs1_all_logprobs" in fail:
|
||||
print(f" BS=1 logprobs for all {len(fail['bs1_all_logprobs'])} steps:")
|
||||
for step_idx, logprobs in enumerate(fail["bs1_all_logprobs"]):
|
||||
print(f" Step {step_idx}: {logprobs}")
|
||||
print(f" BS=N logprobs for all {len(fail['bsN_all_logprobs'])} steps:")
|
||||
for step_idx, logprobs in enumerate(fail["bsN_all_logprobs"]):
|
||||
print(f" Step {step_idx}: {logprobs}")
|
||||
print(f"{'=' * 80}\n")
|
||||
|
||||
# Fail the test with summary
|
||||
msg = (
|
||||
f"Batch invariance violated in {len(failed_prompts)}/{len(prompts)} prompts. See output above for details."
|
||||
)
|
||||
pytest.fail(msg)
|
||||
|
||||
|
||||
@pytest.mark.model(
|
||||
model_name=DEFAULT_MODEL,
|
||||
max_num_seqs=144,
|
||||
gpu_memory_utilization=0.95,
|
||||
max_model_len=8192,
|
||||
dtype="float16",
|
||||
tensor_parallel_size=int(os.getenv("VLLM_TP_SIZE", "1")),
|
||||
enable_prefix_caching=False,
|
||||
distributed_executor_backend="mp",
|
||||
compilation_config={
|
||||
"cudagraph_mode": "PIECEWISE",
|
||||
"cudagraph_capture_sizes": [1, 32, 64],
|
||||
},
|
||||
)
|
||||
def test_simple_generation(vllm_runner, monkeypatch: pytest.MonkeyPatch):
|
||||
"""
|
||||
Simple test that runs the model with a basic prompt and prints the output.
|
||||
Useful for quick smoke testing and debugging.
|
||||
"""
|
||||
prompt = "The capital of France is"
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0.0,
|
||||
max_tokens=20,
|
||||
)
|
||||
|
||||
print(f"\n{'=' * 80}")
|
||||
print("Running simple generation test")
|
||||
print(f"Prompt: '{prompt}'")
|
||||
print(f"{'=' * 80}\n")
|
||||
|
||||
outputs = vllm_runner.model.generate([prompt], sampling_params)
|
||||
|
||||
assert len(outputs) == 1
|
||||
output_text = outputs[0].outputs[0].text
|
||||
|
||||
print(f"Output: '{output_text}'")
|
||||
print(f"\n{'=' * 80}")
|
||||
print(f"Full completion: '{prompt}{output_text}'")
|
||||
print(f"{'=' * 80}\n")
|
||||
|
||||
|
||||
def test_logprobs_without_batch_invariance_should_fail(monkeypatch: pytest.MonkeyPatch):
|
||||
"""
|
||||
This test is the inverse of test_logprobs_bitwise_batch_invariance_bs1_vs_bsN.
|
||||
It DISABLES batch invariance mode and expects to see non-deterministic behavior
|
||||
between BS=1 and BS=N runs. This demonstrates that batch invariance is actually
|
||||
doing something useful.
|
||||
|
||||
The test will PASS if we detect differences (proving batch invariance matters).
|
||||
The test will FAIL if everything matches (suggesting batch invariance isn't needed).
|
||||
"""
|
||||
# CRITICAL: Clear model cache to free up memory before creating a new LLM instance
|
||||
# This test uses different configuration (batch invariance disabled) so it cannot reuse cached models
|
||||
model_cache.clear()
|
||||
|
||||
gc.collect()
|
||||
torch.npu.empty_cache()
|
||||
|
||||
# CRITICAL: Disable batch invariance for this test
|
||||
monkeypatch.setenv("VLLM_BATCH_INVARIANT", "0")
|
||||
seed = int(os.getenv("VLLM_TEST_SEED", "12345"))
|
||||
random.seed(seed)
|
||||
model_name = DEFAULT_MODEL
|
||||
tp_size = int(os.getenv("VLLM_TEST_TP_SIZE", "1"))
|
||||
|
||||
print(f"\n{'=' * 80}")
|
||||
print("BATCH INVARIANCE DISABLED: Expecting non-deterministic behavior")
|
||||
print(f"{'=' * 80}\n")
|
||||
|
||||
llm = LLM(
|
||||
model=model_name,
|
||||
tensor_parallel_size=tp_size,
|
||||
enable_prefix_caching=False,
|
||||
max_num_seqs=32,
|
||||
max_model_len=8192,
|
||||
dtype="bfloat16",
|
||||
compilation_config={
|
||||
"cudagraph_mode": "PIECEWISE",
|
||||
"cudagraph_capture_sizes": [1, 32, 64],
|
||||
},
|
||||
distributed_executor_backend="mp",
|
||||
)
|
||||
|
||||
# build ragged prompts to change shapes significantly across BS=1 vs BS=N
|
||||
long_min = int(os.getenv("VLLM_MIN_PROMPT", "768"))
|
||||
long_max = int(os.getenv("VLLM_MAX_PROMPT", "2048"))
|
||||
prompts: list[str] = []
|
||||
options = [
|
||||
(max(long_min, 1536), max(long_max, 3072)), # very long
|
||||
(max(1024, long_min), max(2048, long_max)), # long
|
||||
(256, 512), # mid
|
||||
(10, 20), # short
|
||||
]
|
||||
|
||||
for _ in range(32):
|
||||
lo, hi = random.choice(options)
|
||||
prompts.append(_random_prompt(lo, hi))
|
||||
|
||||
sp = SamplingParams(
|
||||
temperature=0.6,
|
||||
top_p=1.0,
|
||||
max_tokens=8,
|
||||
seed=1234,
|
||||
logprobs=5,
|
||||
)
|
||||
|
||||
# BS=1: run prompts individually and collect logprobs per step.
|
||||
print("\n" + "=" * 80)
|
||||
print("STARTING BS=1 RUNS (each prompt individually)")
|
||||
print("=" * 80 + "\n")
|
||||
|
||||
bs1_logprobs_per_prompt = []
|
||||
bs1_tokens_per_prompt = []
|
||||
for idx, p in enumerate(prompts):
|
||||
print(f"\n[BS=1] Running prompt {idx}/{len(prompts)} - Preview: {p[:80]}...")
|
||||
outs = llm.generate([p], sp, use_tqdm=False)
|
||||
assert len(outs) == 1
|
||||
step_logprobs, token_ids = _extract_step_logprobs(outs[0])
|
||||
if step_logprobs is None:
|
||||
pytest.skip("Logits are not available on RequestOutput; enable logprobs return to run this test.")
|
||||
bs1_logprobs_per_prompt.append(step_logprobs)
|
||||
bs1_tokens_per_prompt.append(token_ids)
|
||||
print(f"[BS=1] Prompt {idx} generated tokens: {token_ids}")
|
||||
|
||||
# BS=N: run prompts in a batch and collect logprobs per step for each prompt.
|
||||
print("\n" + "=" * 80)
|
||||
print(f"STARTING BS={len(prompts)} RUN (all prompts batched)")
|
||||
print("=" * 80 + "\n")
|
||||
|
||||
outs_batched = llm.generate(prompts, sp, use_tqdm=False)
|
||||
assert len(outs_batched) == len(prompts)
|
||||
bsN_logprobs_per_prompt = []
|
||||
bsN_tokens_per_prompt = []
|
||||
|
||||
print(f"\n[BS={len(prompts)}] Processing batched outputs...")
|
||||
for idx, o in enumerate(outs_batched):
|
||||
tokens = o.outputs[0].token_ids if o.outputs else "N/A"
|
||||
print(f"[BS={len(prompts)}] Prompt {idx} generated tokens: {tokens}")
|
||||
step_logprobs, token_ids = _extract_step_logprobs(o)
|
||||
if step_logprobs is None:
|
||||
pytest.skip("Logits are not available on RequestOutput; enable logprobs return to run this test.")
|
||||
bsN_logprobs_per_prompt.append(step_logprobs)
|
||||
bsN_tokens_per_prompt.append(token_ids)
|
||||
|
||||
# Compare step-by-step logprobs for each prompt between BS=1 and BS=N runs.
|
||||
differences_found = []
|
||||
for i, (logprobs_bs1, logprobs_bsN, tokens_bs1, tokens_bsN) in enumerate(
|
||||
zip(
|
||||
bs1_logprobs_per_prompt,
|
||||
bsN_logprobs_per_prompt,
|
||||
bs1_tokens_per_prompt,
|
||||
bsN_tokens_per_prompt,
|
||||
)
|
||||
):
|
||||
if len(logprobs_bs1) != len(logprobs_bsN):
|
||||
reason = f"Different number of steps: {len(logprobs_bs1)} (BS=1) vs {len(logprobs_bsN)} (BS=N)"
|
||||
differences_found.append(
|
||||
{
|
||||
"prompt_idx": i,
|
||||
"step": "all",
|
||||
"reason": reason,
|
||||
"prompt_preview": prompts[i][:100],
|
||||
"bs1_tokens": tokens_bs1,
|
||||
"bsN_tokens": tokens_bsN,
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
# Check if tokens match first
|
||||
if tokens_bs1 != tokens_bsN:
|
||||
differences_found.append(
|
||||
{
|
||||
"prompt_idx": i,
|
||||
"step": "sampling",
|
||||
"reason": "Different tokens sampled",
|
||||
"prompt_preview": prompts[i][:100],
|
||||
"bs1_tokens": tokens_bs1,
|
||||
"bsN_tokens": tokens_bsN,
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
for t, (a, b) in enumerate(zip(logprobs_bs1, logprobs_bsN)):
|
||||
if a.shape != b.shape:
|
||||
differences_found.append(
|
||||
{
|
||||
"prompt_idx": i,
|
||||
"step": t,
|
||||
"reason": f"Shape mismatch: {a.shape} vs {b.shape}",
|
||||
"prompt_preview": prompts[i][:100],
|
||||
"bs1_tokens": tokens_bs1,
|
||||
"bsN_tokens": tokens_bsN,
|
||||
}
|
||||
)
|
||||
break
|
||||
|
||||
if not torch.equal(a, b):
|
||||
max_diff = torch.abs(a - b).max().item()
|
||||
print(f"\n[EXPECTED DIVERGENCE FOUND] Prompt {i}, Token {t}: max_diff={max_diff:.6e}")
|
||||
bs1_tok = tokens_bs1[t] if t < len(tokens_bs1) else "N/A"
|
||||
bsN_tok = tokens_bsN[t] if t < len(tokens_bsN) else "N/A"
|
||||
print(f" Token IDs: bs1={bs1_tok}, bsN={bsN_tok}")
|
||||
print(f" BS=1 logprob: {a.tolist()}")
|
||||
print(f" BS=N logprob: {b.tolist()}")
|
||||
differences_found.append(
|
||||
{
|
||||
"prompt_idx": i,
|
||||
"step": t,
|
||||
"reason": f"Bitwise mismatch (max_diff={max_diff:.6e})",
|
||||
"prompt_preview": prompts[i][:100],
|
||||
"bs1_tokens": tokens_bs1,
|
||||
"bsN_tokens": tokens_bsN,
|
||||
}
|
||||
)
|
||||
break
|
||||
del llm
|
||||
cleanup_dist_env_and_memory()
|
||||
# Print summary
|
||||
print(f"\n{'=' * 80}")
|
||||
if differences_found:
|
||||
success_msg = (
|
||||
f"✓ SUCCESS: Batch invariance is doing something! "
|
||||
f"Found {len(differences_found)}/{len(prompts)} prompts "
|
||||
f"with differences when batch invariance was DISABLED."
|
||||
)
|
||||
print(success_msg)
|
||||
print(f"{'=' * 80}")
|
||||
for diff in differences_found:
|
||||
print(f"\nPrompt {diff['prompt_idx']} (step {diff['step']}):")
|
||||
print(f" Reason: {diff['reason']}")
|
||||
print(f" Preview: {diff['prompt_preview']}...")
|
||||
if "bs1_tokens" in diff:
|
||||
print(f" BS=1 tokens: {diff['bs1_tokens']}")
|
||||
if "bsN_tokens" in diff:
|
||||
print(f" BS=N tokens: {diff['bsN_tokens']}")
|
||||
print(f"{'=' * 80}\n")
|
||||
# Test PASSES because we found differences (batch invariance matters!)
|
||||
return
|
||||
else:
|
||||
# Test FAILS because everything matched even without batch invariance
|
||||
fail_msg = (
|
||||
f"✗ UNEXPECTED: All {len(prompts)} prompts matched "
|
||||
f"between BS=1 and BS=N even with batch invariance DISABLED. "
|
||||
f"This suggests batch invariance might not be necessary, "
|
||||
f"or the test needs more sensitive prompts."
|
||||
)
|
||||
print(fail_msg)
|
||||
print(f"{'=' * 80}\n")
|
||||
pytest.fail(fail_msg)
|
||||
57
tests/e2e/pull_request/one_card/test_camem.py
Normal file
57
tests/e2e/pull_request/one_card/test_camem.py
Normal file
@@ -0,0 +1,57 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import torch
|
||||
from vllm import SamplingParams
|
||||
from vllm.utils.mem_constants import GiB_bytes
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.utils import fork_new_process_for_each_test
|
||||
|
||||
|
||||
@fork_new_process_for_each_test
|
||||
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_NZ": "0"})
|
||||
def test_end_to_end():
|
||||
free, total = torch.npu.mem_get_info()
|
||||
used_bytes_baseline = total - free # in case other process is running
|
||||
|
||||
prompt = "How are you?"
|
||||
sampling_params = SamplingParams(temperature=0, max_tokens=10)
|
||||
|
||||
with VllmRunner("Qwen/Qwen3-0.6B", enable_sleep_mode=True, cudagraph_capture_sizes=[1, 2, 4, 8]) as runner:
|
||||
output = runner.model.generate(prompt, sampling_params)
|
||||
# the benefit of `llm.sleep(level=2)` is mainly CPU memory usage,
|
||||
# which is difficult to measure in the test. therefore, we only
|
||||
# test sleep level 1 here.
|
||||
runner.model.sleep(level=1)
|
||||
|
||||
free_gpu_bytes_after_sleep, total = torch.npu.mem_get_info()
|
||||
used_bytes = total - free_gpu_bytes_after_sleep - used_bytes_baseline
|
||||
# now the memory usage should be less than the model weights
|
||||
# (0.5B model, 1GiB weights)
|
||||
assert used_bytes < 1 * GiB_bytes
|
||||
|
||||
runner.model.wake_up()
|
||||
output2 = runner.model.generate(prompt, sampling_params)
|
||||
|
||||
# cmp output
|
||||
assert output[0].outputs[0].text == output2[0].outputs[0].text
|
||||
@@ -0,0 +1,74 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
import os
|
||||
|
||||
import pytest
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
|
||||
from tests.e2e.conftest import ModelName
|
||||
|
||||
os.environ["VLLM_USE_MODELSCOPE"] = "True"
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
|
||||
MODELS = ModelName.QWEN3_06B
|
||||
|
||||
|
||||
def get_prompt_embeds(chat, tokenizer, embedding_layer):
|
||||
"""Convert chat messages to prompt embeddings."""
|
||||
token_ids = tokenizer.apply_chat_template(chat, add_generation_prompt=True, return_tensors="pt", return_dict=False)
|
||||
prompt_embeds = embedding_layer(token_ids).squeeze(0)
|
||||
return prompt_embeds
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=MODELS,
|
||||
compilation_config={"cudagraph_capture_sizes": [1, 2, 4, 8]},
|
||||
extra_kwargs={"enable_prompt_embeds": True},
|
||||
)
|
||||
def test_mixed_prompt_embeds_and_text(vllm_runner):
|
||||
"""Test mixed inputs with both prompt embeddings and text prompts."""
|
||||
# Prepare prompt embeddings for first request
|
||||
tokenizer = AutoTokenizer.from_pretrained(MODELS)
|
||||
transformers_model = AutoModelForCausalLM.from_pretrained(MODELS)
|
||||
embedding_layer = transformers_model.get_input_embeddings()
|
||||
|
||||
chat = [{"role": "user", "content": "What is AI?"}]
|
||||
prompt_embeds = get_prompt_embeds(chat, tokenizer, embedding_layer)
|
||||
|
||||
# Prepare text prompt for second request
|
||||
text_prompt = "What is machine learning?"
|
||||
|
||||
# Test prompt embeddings
|
||||
embeds_output = vllm_runner.model.generate(
|
||||
{
|
||||
"prompt_embeds": prompt_embeds,
|
||||
}
|
||||
)
|
||||
|
||||
# Test text prompt
|
||||
text_output = vllm_runner.model.generate(text_prompt)
|
||||
|
||||
# Verify both types of inputs work
|
||||
assert len(embeds_output) == 1
|
||||
assert len(text_output) == 1
|
||||
assert len(embeds_output[0].outputs[0].text) > 0
|
||||
assert len(text_output[0].outputs[0].text) > 0
|
||||
|
||||
print("\n[Prompt Embeds Output]:", embeds_output[0].outputs[0].text)
|
||||
print("[Text Prompt Output]:", text_output[0].outputs[0].text)
|
||||
177
tests/e2e/pull_request/one_card/test_cpu_offloading.py
Normal file
177
tests/e2e/pull_request/one_card/test_cpu_offloading.py
Normal file
@@ -0,0 +1,177 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
import socket
|
||||
import time
|
||||
|
||||
import msgspec
|
||||
import msgspec.msgpack
|
||||
import pytest
|
||||
import zmq
|
||||
from vllm import LLM, SamplingParams, TokensPrompt
|
||||
from vllm.config import KVEventsConfig, KVTransferConfig
|
||||
from vllm.distributed.kv_events import BlockStored, KVEventBatch
|
||||
|
||||
|
||||
class MockSubscriber:
|
||||
"""Helper class to receive and verify published events"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
endpoint: str,
|
||||
topic: str,
|
||||
):
|
||||
self.ctx = zmq.Context.instance() # type: ignore
|
||||
self.topic_bytes = topic.encode("utf-8")
|
||||
|
||||
# Set up subscriber socket
|
||||
self.sub = self.ctx.socket(zmq.SUB) # type: ignore
|
||||
self.sub.setsockopt(zmq.SUBSCRIBE, self.topic_bytes) # type: ignore
|
||||
self.sub.connect(endpoint)
|
||||
|
||||
self.decoder = msgspec.msgpack.Decoder(type=KVEventBatch)
|
||||
|
||||
def get_new_cpu_stored_events(self) -> list[BlockStored]:
|
||||
cpu_stored_events: list[BlockStored] = []
|
||||
|
||||
poller = zmq.Poller() # type: ignore
|
||||
poller.register(self.sub, zmq.POLLIN) # type: ignore
|
||||
|
||||
timeout = 1000 # 1 second
|
||||
while True:
|
||||
events = dict(poller.poll(timeout))
|
||||
|
||||
if events.get(self.sub) != zmq.POLLIN: # type: ignore
|
||||
return cpu_stored_events
|
||||
|
||||
topic_bytes, _, payload = self.sub.recv_multipart()
|
||||
|
||||
assert topic_bytes == self.topic_bytes
|
||||
|
||||
event_batch = self.decoder.decode(payload)
|
||||
assert isinstance(event_batch, KVEventBatch)
|
||||
for event in event_batch.events:
|
||||
if isinstance(event, BlockStored) and event.medium == "CPU":
|
||||
cpu_stored_events.append(event)
|
||||
timeout = 100
|
||||
|
||||
def close(self):
|
||||
"""Clean up resources"""
|
||||
self.sub.close()
|
||||
|
||||
|
||||
def _latency_test(llm: LLM, subscriber: MockSubscriber):
|
||||
sampling_params = SamplingParams(max_tokens=1)
|
||||
|
||||
num_times_cpu_better_than_cold = 0
|
||||
num_tests = 10
|
||||
total_cold_time = 0.0
|
||||
total_gpu_hit_time = 0.0
|
||||
total_cpu_hit_time = 0.0
|
||||
prompt_token_ids = [0] * 10001
|
||||
for i in range(num_tests):
|
||||
prompt_token_ids[0] = i
|
||||
prompts = [TokensPrompt(prompt_token_ids=prompt_token_ids)]
|
||||
|
||||
# run generation - this should trigger saving KV cache
|
||||
start_time = time.time()
|
||||
llm.generate(prompts, sampling_params, use_tqdm=False)
|
||||
cold_time = time.time() - start_time
|
||||
total_cold_time += cold_time
|
||||
|
||||
# run generation again - should hit the GPU prefix cache
|
||||
start_time = time.time()
|
||||
llm.generate(prompts, sampling_params, use_tqdm=False)
|
||||
gpu_hit_time = time.time() - start_time
|
||||
total_gpu_hit_time += gpu_hit_time
|
||||
|
||||
# reset prefix cache to avoid GPU hit.
|
||||
llm.reset_prefix_cache()
|
||||
|
||||
assert subscriber.get_new_cpu_stored_events()
|
||||
|
||||
# run generation again - this should trigger loading from CPU
|
||||
start_time = time.time()
|
||||
llm.generate(prompts, sampling_params, use_tqdm=False)
|
||||
cpu_hit_time = time.time() - start_time
|
||||
total_cpu_hit_time += cpu_hit_time
|
||||
|
||||
if cpu_hit_time < cold_time:
|
||||
num_times_cpu_better_than_cold += 1
|
||||
|
||||
print("Average times:")
|
||||
print(f" Cold: {total_cold_time * 1000 / num_tests:.2f}ms")
|
||||
print(f" GPU hit: {total_gpu_hit_time * 1000 / num_tests:.2f}ms")
|
||||
print(f" CPU hit: {total_cpu_hit_time * 1000 / num_tests:.2f}ms")
|
||||
|
||||
assert num_times_cpu_better_than_cold >= 0.8 * num_tests
|
||||
|
||||
|
||||
def _accuracy_test(llm: LLM, subscriber: MockSubscriber):
|
||||
sampling_params = SamplingParams(max_tokens=1)
|
||||
cpu_block_size = llm.llm_engine.vllm_config.kv_transfer_config.kv_connector_extra_config["block_size"]
|
||||
|
||||
subscriber.get_new_cpu_stored_events()
|
||||
|
||||
# prepend prompt to be cpu block aligned
|
||||
prompt = "Let's count to 10. One, two, three, four,"
|
||||
while len(llm.generate(prompt, use_tqdm=False)[0].prompt_token_ids) % cpu_block_size != 0:
|
||||
prompt = ". " + prompt
|
||||
|
||||
assert subscriber.get_new_cpu_stored_events()
|
||||
|
||||
test_count = 100
|
||||
success_count = 0
|
||||
for i in range(test_count):
|
||||
if llm.generate(prompt, sampling_params, use_tqdm=False)[0].outputs[0].text == " five":
|
||||
success_count += 1
|
||||
|
||||
assert success_count >= 0.5 * test_count
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="cpu offload connector is deprecated.")
|
||||
def test_cpu_offloading() -> None:
|
||||
"""
|
||||
Tests OffloadingConnector with CPUOffloadingSpec.
|
||||
"""
|
||||
|
||||
# configure OffloadingConnector (spec_name=CPUOffloadingSpec by default)
|
||||
kv_transfer_config = KVTransferConfig(
|
||||
kv_connector="OffloadingConnector",
|
||||
kv_role="kv_both",
|
||||
kv_connector_extra_config={
|
||||
"num_cpu_blocks": 1000,
|
||||
"block_size": 128,
|
||||
"spec_name": "NPUOffloadingSpec",
|
||||
"spec_module_path": "vllm_ascend.kv_offload.npu",
|
||||
},
|
||||
)
|
||||
|
||||
port: int
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
s.bind(("0.0.0.0", 0))
|
||||
port = s.getsockname()[1]
|
||||
|
||||
events_endpoint = f"tcp://*:{port}"
|
||||
kv_events_config = KVEventsConfig(
|
||||
enable_kv_cache_events=True,
|
||||
publisher="zmq",
|
||||
endpoint=events_endpoint,
|
||||
topic="test",
|
||||
)
|
||||
|
||||
llm = LLM(
|
||||
model="Qwen/Qwen3-0.6B",
|
||||
gpu_memory_utilization=0.5,
|
||||
kv_events_config=kv_events_config,
|
||||
kv_transfer_config=kv_transfer_config,
|
||||
)
|
||||
|
||||
events_endpoint = events_endpoint.replace("*", "127.0.0.1")
|
||||
subscriber = MockSubscriber(events_endpoint, topic=kv_events_config.topic)
|
||||
|
||||
try:
|
||||
_latency_test(llm, subscriber)
|
||||
_accuracy_test(llm, subscriber)
|
||||
finally:
|
||||
subscriber.close()
|
||||
del llm
|
||||
130
tests/e2e/pull_request/one_card/test_cpu_weight_offload.py
Normal file
130
tests/e2e/pull_request/one_card/test_cpu_weight_offload.py
Normal file
@@ -0,0 +1,130 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
"""End-to-end tests for CPU weight offloading on Ascend NPU.
|
||||
|
||||
Covers both the prefetch backend (NPUPrefetchOffloader) and the UVA
|
||||
backend (functional_call fallback path, since UVA is not available on
|
||||
NPU hardware). Tests verify that offloading produces the same outputs
|
||||
as the baseline (no offloading).
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import wait_until_npu_memory_free
|
||||
from tests.e2e.pull_request.utils import PROMPTS_SHORT, compare_logprobs
|
||||
|
||||
MODEL = "Qwen/Qwen3-0.6B"
|
||||
|
||||
|
||||
# -------------------- Prefetch backend tests --------------------
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_prefetch_offload_eager():
|
||||
"""Test prefetch CPU offloading in eager mode.
|
||||
|
||||
Compares outputs between:
|
||||
1. Baseline (eager, no offloading)
|
||||
2. Prefetch offloading (group_size=4, num_in_group=1)
|
||||
with enforce_eager=True (no ACL graph capture)
|
||||
"""
|
||||
runner_kwargs = {
|
||||
"model_name": MODEL,
|
||||
"max_model_len": 512,
|
||||
"enforce_eager": True,
|
||||
"offload_backend": "prefetch",
|
||||
"offload_group_size": 4,
|
||||
"offload_num_in_group": 1,
|
||||
}
|
||||
compare_logprobs(runner_kwargs=runner_kwargs, prompts=PROMPTS_SHORT)
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_prefetch_offload_aclgraph():
|
||||
"""Test prefetch CPU offloading with ACL graph capture.
|
||||
|
||||
Compares outputs between:
|
||||
1. Baseline (eager, no offloading)
|
||||
2. Prefetch offloading (group_size=4, num_in_group=1)
|
||||
with ACL graph capture enabled (default, non-eager)
|
||||
"""
|
||||
runner_kwargs = {
|
||||
"model_name": MODEL,
|
||||
"max_model_len": 512,
|
||||
"cudagraph_capture_sizes": [1, 2, 4, 8],
|
||||
"offload_backend": "prefetch",
|
||||
"offload_group_size": 4,
|
||||
"offload_num_in_group": 1,
|
||||
}
|
||||
compare_logprobs(runner_kwargs=runner_kwargs, prompts=PROMPTS_SHORT)
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_prefetch_offload_selective_params():
|
||||
"""Test selective parameter offloading (MLP weights only).
|
||||
|
||||
Only offloads gate_up_proj and down_proj parameters, leaving
|
||||
attention weights on NPU.
|
||||
"""
|
||||
runner_kwargs = {
|
||||
"model_name": MODEL,
|
||||
"max_model_len": 512,
|
||||
"enforce_eager": True,
|
||||
"offload_backend": "prefetch",
|
||||
"offload_group_size": 8,
|
||||
"offload_num_in_group": 2,
|
||||
"offload_prefetch_step": 1,
|
||||
"offload_params": {"gate_up_proj", "down_proj"},
|
||||
}
|
||||
compare_logprobs(runner_kwargs=runner_kwargs, prompts=PROMPTS_SHORT)
|
||||
|
||||
|
||||
# -------------------- UVA backend tests --------------------
|
||||
# UVA (Unified Virtual Addressing) is not available on Ascend NPU, so
|
||||
# the UVA offloader falls back to the functional_call path that moves
|
||||
# weights to device on-demand. Tests below mirror the upstream
|
||||
# test_cpu_offload.py parametrization but only exercise the non-UVA
|
||||
# (functional_call) path with enforce_eager, as NPU does not support
|
||||
# UVA zero-copy.
|
||||
|
||||
|
||||
@pytest.mark.parametrize("disable_pin_memory", [False, True])
|
||||
@wait_until_npu_memory_free()
|
||||
def test_uva_offload_functional_call(disable_pin_memory):
|
||||
"""Test UVA offloader's functional_call fallback on NPU.
|
||||
|
||||
With UVA disabled (forced by env var), the UVA offloader falls back
|
||||
to moving weights to device inside a functional_call wrapper.
|
||||
enforce_eager is required because this fallback is incompatible
|
||||
with graph capture.
|
||||
|
||||
Parametrized over pin_memory to cover both pinned and unpinned
|
||||
CPU storage paths.
|
||||
"""
|
||||
old_uva = os.environ.get("VLLM_WEIGHT_OFFLOADING_DISABLE_UVA")
|
||||
old_pin = os.environ.get("VLLM_WEIGHT_OFFLOADING_DISABLE_PIN_MEMORY")
|
||||
try:
|
||||
os.environ["VLLM_WEIGHT_OFFLOADING_DISABLE_UVA"] = "1"
|
||||
os.environ["VLLM_WEIGHT_OFFLOADING_DISABLE_PIN_MEMORY"] = str(int(disable_pin_memory))
|
||||
runner_kwargs = {
|
||||
"model_name": MODEL,
|
||||
"max_model_len": 512,
|
||||
"enforce_eager": True,
|
||||
"cpu_offload_gb": 1,
|
||||
}
|
||||
compare_logprobs(runner_kwargs=runner_kwargs, prompts=PROMPTS_SHORT)
|
||||
finally:
|
||||
if old_uva is None:
|
||||
os.environ.pop("VLLM_WEIGHT_OFFLOADING_DISABLE_UVA", None)
|
||||
else:
|
||||
os.environ["VLLM_WEIGHT_OFFLOADING_DISABLE_UVA"] = old_uva
|
||||
if old_pin is None:
|
||||
os.environ.pop("VLLM_WEIGHT_OFFLOADING_DISABLE_PIN_MEMORY", None)
|
||||
else:
|
||||
os.environ["VLLM_WEIGHT_OFFLOADING_DISABLE_PIN_MEMORY"] = old_pin
|
||||
291
tests/e2e/pull_request/one_card/test_guided_decoding.py
Normal file
291
tests/e2e/pull_request/one_card/test_guided_decoding.py
Normal file
@@ -0,0 +1,291 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/entrypoints/llm/test_guided_generate.py
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
import json
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import jsonschema
|
||||
import pytest
|
||||
import regex as re
|
||||
from vllm.outputs import RequestOutput
|
||||
from vllm.sampling_params import SamplingParams, StructuredOutputsParams
|
||||
|
||||
from tests.e2e.conftest import ModelName
|
||||
from vllm_ascend.utils import vllm_version_is
|
||||
|
||||
os.environ["VLLM_BATCH_INVARIANT"] = "1"
|
||||
|
||||
MODEL_NAME = ModelName.QWEN3_06B
|
||||
|
||||
GuidedDecodingBackend = ["xgrammar", "guidance", "outlines"]
|
||||
REGEX_COMPILATION_TIMEOUT_ENV = {"VLLM_REGEX_COMPILATION_TIMEOUT_S": "30"}
|
||||
|
||||
|
||||
@pytest.fixture(params=[False, True], ids=["v1", "v2"])
|
||||
def model_runner_env(request):
|
||||
use_v2_model_runner = request.param
|
||||
if use_v2_model_runner and vllm_version_is("0.23.0"):
|
||||
pytest.skip("No need to support v2 model runner for vLLM tag version.")
|
||||
|
||||
with patch.dict(os.environ, {"VLLM_USE_V2_MODEL_RUNNER": "1" if use_v2_model_runner else "0"}):
|
||||
yield
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def sample_regex():
|
||||
return (
|
||||
r"((25[0-5]|(2[0-4]|1\d|[1-9]|)\d)\.){3}"
|
||||
r"(25[0-5]|(2[0-4]|1\d|[1-9]|)\d)"
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def sample_json_schema():
|
||||
return {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": {"type": "string"},
|
||||
"age": {"type": "integer"},
|
||||
"skills": {"type": "array", "items": {"type": "string", "maxLength": 10}, "minItems": 3},
|
||||
"work_history": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"company": {"type": "string"},
|
||||
"duration": {"type": "number"},
|
||||
"position": {"type": "string"},
|
||||
},
|
||||
"required": ["company", "position"],
|
||||
},
|
||||
},
|
||||
},
|
||||
"required": ["name", "age", "skills", "work_history"],
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=MODEL_NAME,
|
||||
compilation_config={"cudagraph_capture_sizes": [1, 2, 4, 8]},
|
||||
extra_kwargs={"seed": 0, "structured_outputs_config": {"backend": "xgrammar"}},
|
||||
)
|
||||
def test_guided_json_completion_xgrammar(sample_json_schema, request):
|
||||
sampling_params = SamplingParams(
|
||||
temperature=1.0, max_tokens=500, structured_outputs=StructuredOutputsParams(json=sample_json_schema)
|
||||
)
|
||||
if not vllm_version_is("0.23.0"):
|
||||
model_marker = request.node.get_closest_marker("model")
|
||||
model_marker.kwargs["env_vars"] = REGEX_COMPILATION_TIMEOUT_ENV
|
||||
with patch.dict(os.environ, REGEX_COMPILATION_TIMEOUT_ENV, clear=False):
|
||||
vllm_runner = request.getfixturevalue("vllm_runner")
|
||||
prompts = [f"Give an example JSON for an employee profile that fits this schema: {sample_json_schema}"] * 2
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
outputs = vllm_runner.model.generate(inputs, sampling_params=sampling_params)
|
||||
|
||||
assert outputs is not None
|
||||
for output in outputs:
|
||||
assert output is not None
|
||||
assert isinstance(output, RequestOutput)
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
assert generated_text is not None
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
output_json = json.loads(generated_text)
|
||||
jsonschema.validate(instance=output_json, schema=sample_json_schema)
|
||||
else:
|
||||
vllm_runner = request.getfixturevalue("vllm_runner")
|
||||
prompts = [f"Give an example JSON for an employee profile that fits this schema: {sample_json_schema}"] * 2
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
outputs = vllm_runner.model.generate(inputs, sampling_params=sampling_params)
|
||||
|
||||
assert outputs is not None
|
||||
for output in outputs:
|
||||
assert output is not None
|
||||
assert isinstance(output, RequestOutput)
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
assert generated_text is not None
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
output_json = json.loads(generated_text)
|
||||
jsonschema.validate(instance=output_json, schema=sample_json_schema)
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=MODEL_NAME,
|
||||
compilation_config={"cudagraph_capture_sizes": [1, 2, 4, 8]},
|
||||
extra_kwargs={"seed": 0, "structured_outputs_config": {"backend": "xgrammar"}},
|
||||
)
|
||||
def test_guided_regex_xgrammar(sample_regex, vllm_runner):
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0.8, top_p=0.95, structured_outputs=StructuredOutputsParams(regex=sample_regex)
|
||||
)
|
||||
prompts = [f"Give an example IPv4 address with this regex: {sample_regex}"] * 2
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
outputs = vllm_runner.model.generate(inputs, sampling_params=sampling_params)
|
||||
assert outputs is not None
|
||||
for output in outputs:
|
||||
assert output is not None
|
||||
assert isinstance(output, RequestOutput)
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
print(generated_text)
|
||||
assert generated_text is not None
|
||||
assert re.fullmatch(".*", generated_text) is not None
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=MODEL_NAME,
|
||||
compilation_config={"cudagraph_capture_sizes": [1, 2, 4, 8]},
|
||||
extra_kwargs={"seed": 0, "structured_outputs_config": {"backend": "guidance"}},
|
||||
)
|
||||
def test_guided_json_completion_guidance(sample_json_schema, vllm_runner):
|
||||
sampling_params = SamplingParams(
|
||||
temperature=1.0, max_tokens=500, structured_outputs=StructuredOutputsParams(json=sample_json_schema)
|
||||
)
|
||||
prompts = [f"Give an example JSON for an employee profile that fits this schema: {sample_json_schema}"] * 2
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
outputs = vllm_runner.model.generate(inputs, sampling_params=sampling_params)
|
||||
|
||||
assert outputs is not None
|
||||
for output in outputs:
|
||||
assert output is not None
|
||||
assert isinstance(output, RequestOutput)
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
assert generated_text is not None
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
output_json = json.loads(generated_text)
|
||||
jsonschema.validate(instance=output_json, schema=sample_json_schema)
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=MODEL_NAME,
|
||||
compilation_config={"cudagraph_capture_sizes": [1, 2, 4, 8]},
|
||||
extra_kwargs={"seed": 0, "structured_outputs_config": {"backend": "guidance"}},
|
||||
)
|
||||
def test_guided_regex_guidance(sample_regex, vllm_runner):
|
||||
sampling_params = SamplingParams(
|
||||
temperature=0.8, top_p=0.95, structured_outputs=StructuredOutputsParams(regex=sample_regex)
|
||||
)
|
||||
prompts = [f"Give an example IPv4 address with this regex: {sample_regex}"] * 2
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
outputs = vllm_runner.model.generate(inputs, sampling_params=sampling_params)
|
||||
assert outputs is not None
|
||||
for output in outputs:
|
||||
assert output is not None
|
||||
assert isinstance(output, RequestOutput)
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
print(generated_text)
|
||||
assert generated_text is not None
|
||||
assert re.fullmatch(".*", generated_text) is not None
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=MODEL_NAME,
|
||||
compilation_config={"cudagraph_capture_sizes": [1, 2, 4, 8]},
|
||||
extra_kwargs={"seed": 0, "structured_outputs_config": {"backend": "auto"}},
|
||||
)
|
||||
def test_guided_auto_rejects_mixed_structured_output_backends(vllm_runner):
|
||||
xgrammar_schema = {
|
||||
"type": "object",
|
||||
"properties": {"name": {"type": "string"}},
|
||||
"required": ["name"],
|
||||
}
|
||||
guidance_schema = {
|
||||
"type": "object",
|
||||
"properties": {"count": {"type": "integer", "multipleOf": 2}},
|
||||
"required": ["count"],
|
||||
}
|
||||
|
||||
xgrammar_params = SamplingParams(
|
||||
temperature=0.0,
|
||||
max_tokens=32,
|
||||
structured_outputs=StructuredOutputsParams(json=xgrammar_schema),
|
||||
)
|
||||
prompts = [f"Give an example JSON that fits this schema: {xgrammar_schema}"]
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
outputs = vllm_runner.model.generate(inputs, sampling_params=xgrammar_params)
|
||||
|
||||
assert outputs is not None
|
||||
assert outputs[0] is not None
|
||||
|
||||
guidance_params = SamplingParams(
|
||||
temperature=0.0,
|
||||
max_tokens=32,
|
||||
structured_outputs=StructuredOutputsParams(json=guidance_schema),
|
||||
)
|
||||
prompts = [f"Give an example JSON that fits this schema: {guidance_schema}"]
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
with pytest.raises(ValueError, match="already using 'xgrammar'.*'guidance'"):
|
||||
vllm_runner.model.generate(inputs, sampling_params=guidance_params)
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=MODEL_NAME,
|
||||
compilation_config={"cudagraph_capture_sizes": [1, 2, 4, 8]},
|
||||
extra_kwargs={"seed": 0, "structured_outputs_config": {"backend": "outlines"}},
|
||||
)
|
||||
def test_guided_json_completion_outlines(sample_json_schema, request):
|
||||
sampling_params = SamplingParams(
|
||||
temperature=1.0, max_tokens=500, structured_outputs=StructuredOutputsParams(json=sample_json_schema)
|
||||
)
|
||||
if not vllm_version_is("0.23.0"):
|
||||
model_marker = request.node.get_closest_marker("model")
|
||||
model_marker.kwargs["env_vars"] = REGEX_COMPILATION_TIMEOUT_ENV
|
||||
with patch.dict(os.environ, REGEX_COMPILATION_TIMEOUT_ENV, clear=False):
|
||||
vllm_runner = request.getfixturevalue("vllm_runner")
|
||||
prompts = [f"Give an example JSON for an employee profile that fits this schema: {sample_json_schema}"] * 2
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
outputs = vllm_runner.model.generate(inputs, sampling_params=sampling_params)
|
||||
|
||||
assert outputs is not None
|
||||
for output in outputs:
|
||||
assert output is not None
|
||||
assert isinstance(output, RequestOutput)
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
assert generated_text is not None
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
output_json = json.loads(generated_text)
|
||||
jsonschema.validate(instance=output_json, schema=sample_json_schema)
|
||||
else:
|
||||
vllm_runner = request.getfixturevalue("vllm_runner")
|
||||
prompts = [f"Give an example JSON for an employee profile that fits this schema: {sample_json_schema}"] * 2
|
||||
inputs = vllm_runner.get_inputs(prompts)
|
||||
outputs = vllm_runner.model.generate(inputs, sampling_params=sampling_params)
|
||||
|
||||
assert outputs is not None
|
||||
for output in outputs:
|
||||
assert output is not None
|
||||
assert isinstance(output, RequestOutput)
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text
|
||||
assert generated_text is not None
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
output_json = json.loads(generated_text)
|
||||
jsonschema.validate(instance=output_json, schema=sample_json_schema)
|
||||
42
tests/e2e/pull_request/one_card/test_minicpm.py
Normal file
42
tests/e2e/pull_request/one_card/test_minicpm.py
Normal file
@@ -0,0 +1,42 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/entrypoints/llm/test_guided_generate.py
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
|
||||
MINICPM_MODELS = [
|
||||
"openbmb/MiniCPM-2B-sft-bf16",
|
||||
"OpenBMB/MiniCPM4-0.5B",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MINICPM_MODELS)
|
||||
def test_minicpm(model) -> None:
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
max_tokens = 5
|
||||
|
||||
with VllmRunner(model, max_model_len=512, gpu_memory_utilization=0.7) as runner:
|
||||
runner.generate_greedy(example_prompts, max_tokens)
|
||||
93
tests/e2e/pull_request/one_card/test_multi_instance.py
Normal file
93
tests/e2e/pull_request/one_card/test_multi_instance.py
Normal file
@@ -0,0 +1,93 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
|
||||
"""
|
||||
Two VllmRunner instances are nested so that the first instance's worker
|
||||
process is still holding NPU memory when the second instance's worker process
|
||||
starts. Both instances must:
|
||||
|
||||
1. Initialize without raising any exception (no OOM during
|
||||
determine_available_memory / KV-cache allocation).
|
||||
2. Successfully complete a short generation request.
|
||||
|
||||
The model is Qwen/Qwen3-0.6B (~0.5 GiB weights) and gpu_memory_utilization
|
||||
is set to 0.4 per instance so that two instances comfortably fit on a single
|
||||
64 GiB Ascend 910B card while leaving enough headroom to avoid the
|
||||
pre-fix negative-KV-cache condition.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
|
||||
MODEL = "Qwen/Qwen3-0.6B"
|
||||
_PROMPTS = ["Hello, my name is"]
|
||||
_MAX_TOKENS = 5
|
||||
# Use a low utilization so two instances fit side-by-side on one card:
|
||||
# 2 × 0.4 × card_total ≤ card_total (holds for any card ≥ 1 GiB)
|
||||
_GPU_MEM_UTIL = 0.4
|
||||
_MAX_MODEL_LEN = 512
|
||||
|
||||
|
||||
def test_two_instances_on_single_card() -> None:
|
||||
"""
|
||||
Regression test for PR #7427 (multi-instance OOM on single card).
|
||||
|
||||
Start a first vllm-ascend instance; while it is still running and holding
|
||||
NPU memory, start a second instance with identical settings. Both must
|
||||
initialize correctly and produce non-empty outputs.
|
||||
|
||||
Failure signature (pre-fix):
|
||||
RuntimeError / ValueError during the second instance's init, or
|
||||
"Available KV cache memory: -X.XX GiB" in the logs followed by
|
||||
zero KV blocks being allocated.
|
||||
"""
|
||||
# ── First instance ──────────────────────────────────────────────────
|
||||
with VllmRunner(
|
||||
MODEL,
|
||||
max_model_len=_MAX_MODEL_LEN,
|
||||
gpu_memory_utilization=_GPU_MEM_UTIL,
|
||||
enforce_eager=True,
|
||||
) as runner1:
|
||||
# ── Second instance starts while first is still alive ────────────
|
||||
# This is the exact scenario from PR #7427: the second worker process
|
||||
# sees a reduced init_snapshot.free_memory because the first instance's
|
||||
# worker is still holding NPU memory.
|
||||
with VllmRunner(
|
||||
MODEL,
|
||||
max_model_len=_MAX_MODEL_LEN,
|
||||
gpu_memory_utilization=_GPU_MEM_UTIL,
|
||||
enforce_eager=True,
|
||||
) as runner2:
|
||||
outputs2 = runner2.generate_greedy(_PROMPTS, max_tokens=_MAX_TOKENS)
|
||||
|
||||
outputs1 = runner1.generate_greedy(_PROMPTS, max_tokens=_MAX_TOKENS)
|
||||
|
||||
# ── Assertions ───────────────────────────────────────────────────────
|
||||
assert outputs1, "First instance produced no outputs"
|
||||
assert outputs2, "Second instance produced no outputs"
|
||||
|
||||
_, text1 = outputs1[0]
|
||||
_, text2 = outputs2[0]
|
||||
|
||||
assert text1, "First instance output text is empty — model may have failed to run"
|
||||
assert text2, (
|
||||
"Second instance output text is empty — "
|
||||
"KV cache may have been allocated with zero blocks (pre-fix OOM regression)"
|
||||
)
|
||||
@@ -0,0 +1,105 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
"""
|
||||
Compare the outputs of vLLM with multistream_overlap_shared_expert
|
||||
enabled and disabled.
|
||||
|
||||
Run `pytest tests/e2e/pull_request/one_card/test_multistream_overlap_shared_expert.py`.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
from tests.e2e.model_utils import check_outputs_equal
|
||||
|
||||
MODELS = [
|
||||
"vllm-ascend/DeepSeek-V2-Lite-W8A8",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
@pytest.mark.parametrize("max_tokens", [32])
|
||||
def test_models_with_multistream_overlap_shared_expert(
|
||||
model: str,
|
||||
max_tokens: int,
|
||||
) -> None:
|
||||
prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
|
||||
sampling_params = SamplingParams(max_tokens=max_tokens, temperature=0.0)
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
enforce_eager=True,
|
||||
cudagraph_capture_sizes=[4, 8, 16, 32],
|
||||
additional_config={
|
||||
"multistream_overlap_shared_expert": True,
|
||||
},
|
||||
quantization="ascend",
|
||||
) as runner:
|
||||
vllm_moe_ms_eager_outputs = runner.model.generate(prompts, sampling_params)
|
||||
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
cudagraph_capture_sizes=[4, 8, 16, 32],
|
||||
additional_config={
|
||||
"multistream_overlap_shared_expert": True,
|
||||
},
|
||||
quantization="ascend",
|
||||
) as runner:
|
||||
vllm_moe_ms_aclgraph_outputs = runner.model.generate(prompts, sampling_params)
|
||||
|
||||
with VllmRunner(
|
||||
model,
|
||||
max_model_len=1024,
|
||||
enforce_eager=True,
|
||||
cudagraph_capture_sizes=[4, 8, 16, 32],
|
||||
quantization="ascend",
|
||||
) as runner:
|
||||
vllm_eager_outputs = runner.model.generate(prompts, sampling_params)
|
||||
|
||||
vllm_moe_ms_eager_outputs_list = []
|
||||
for output in vllm_moe_ms_eager_outputs:
|
||||
vllm_moe_ms_eager_outputs_list.append((output.outputs[0].index, output.outputs[0].text))
|
||||
|
||||
vllm_moe_ms_aclgraph_outputs_list = []
|
||||
for output in vllm_moe_ms_aclgraph_outputs:
|
||||
vllm_moe_ms_aclgraph_outputs_list.append((output.outputs[0].index, output.outputs[0].text))
|
||||
|
||||
vllm_eager_outputs_list = []
|
||||
for output in vllm_eager_outputs:
|
||||
vllm_eager_outputs_list.append((output.outputs[0].index, output.outputs[0].text))
|
||||
|
||||
check_outputs_equal(
|
||||
outputs_0_lst=vllm_eager_outputs_list,
|
||||
outputs_1_lst=vllm_moe_ms_eager_outputs_list,
|
||||
name_0="vllm_eager_outputs",
|
||||
name_1="vllm_moe_ms_eager_outputs",
|
||||
)
|
||||
|
||||
check_outputs_equal(
|
||||
outputs_0_lst=vllm_eager_outputs_list,
|
||||
outputs_1_lst=vllm_moe_ms_aclgraph_outputs_list,
|
||||
name_0="vllm_eager_outputs",
|
||||
name_1="vllm_moe_ms_aclgraph_outputs",
|
||||
)
|
||||
176
tests/e2e/pull_request/one_card/test_npu_ipc_weight_transfer.py
Normal file
176
tests/e2e/pull_request/one_card/test_npu_ipc_weight_transfer.py
Normal file
@@ -0,0 +1,176 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
"""End-to-end test for the NPU IPC weight transfer engine.
|
||||
|
||||
Unlike the HCCL engine, NPU IPC requires the trainer and the inference worker
|
||||
to be co-located on the *same* physical NPU chip, so only a single NPU is
|
||||
needed. The trainer model is built from the architecture config with random
|
||||
weights (download-free); set ``WEIGHT_TRANSFER_TEST_MODEL=/path/to/checkpoint``
|
||||
to share real weights instead. See ``examples/rl/rlhf_http_npu_ipc.py`` for the
|
||||
end-user workflow.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
import torch
|
||||
import torch_npu # noqa: F401 # registers the NPU backend
|
||||
from transformers import AutoConfig, AutoModelForCausalLM
|
||||
|
||||
from tests.e2e.conftest import RemoteOpenAIServer
|
||||
|
||||
MODEL_NAME = "Qwen/Qwen3-0.6B"
|
||||
|
||||
INFERENCE_DEVICE_INDEX = 0
|
||||
|
||||
PROMPTS = [
|
||||
"Hello, my name is",
|
||||
"The capital of France is",
|
||||
]
|
||||
|
||||
UPDATE_TIMEOUT = 300
|
||||
CONTROL_TIMEOUT = 60
|
||||
|
||||
|
||||
def _build_trainer_model(device_index: int):
|
||||
device = f"npu:{device_index}"
|
||||
override_path = os.getenv("WEIGHT_TRANSFER_TEST_MODEL")
|
||||
if override_path:
|
||||
model = AutoModelForCausalLM.from_pretrained(override_path, dtype=torch.bfloat16)
|
||||
else:
|
||||
config = AutoConfig.from_pretrained(MODEL_NAME, trust_remote_code=True)
|
||||
model = AutoModelForCausalLM.from_config(config)
|
||||
model = model.to(device=device, dtype=torch.bfloat16)
|
||||
model.eval()
|
||||
return model
|
||||
|
||||
|
||||
def _post(server: RemoteOpenAIServer, route: str, *, json=None, timeout=CONTROL_TIMEOUT):
|
||||
response = requests.post(server.url_for(route), json=json, timeout=timeout)
|
||||
response.raise_for_status()
|
||||
return response
|
||||
|
||||
|
||||
def _generate(client, model, prompts):
|
||||
completions = []
|
||||
for prompt in prompts:
|
||||
response = client.completions.create(model=model, prompt=prompt, max_tokens=16, temperature=0)
|
||||
completions.append(response.choices[0].text)
|
||||
return completions
|
||||
|
||||
|
||||
def _has_lifecycle_endpoints(server: RemoteOpenAIServer) -> bool:
|
||||
"""Probe ``/start_weight_update``; also performs the actual call when present."""
|
||||
try:
|
||||
response = requests.post(
|
||||
server.url_for("start_weight_update"),
|
||||
json={"is_checkpoint_format": True},
|
||||
timeout=CONTROL_TIMEOUT,
|
||||
)
|
||||
except requests.RequestException:
|
||||
return False
|
||||
if response.status_code == 404:
|
||||
return False
|
||||
response.raise_for_status()
|
||||
return True
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
torch.npu.device_count() < 1,
|
||||
reason="NPU IPC weight transfer e2e test requires at least 1 NPU.",
|
||||
)
|
||||
def test_npu_ipc_weight_transfer_updates_server_weights():
|
||||
from vllm.utils.network_utils import get_open_port
|
||||
|
||||
port = get_open_port()
|
||||
server_args = [
|
||||
"--enforce-eager",
|
||||
"--load-format",
|
||||
"dummy",
|
||||
"--weight-transfer-config",
|
||||
'{"backend": "ipc"}',
|
||||
"--max-model-len",
|
||||
"1024",
|
||||
# IPC co-locates the trainer on the same NPU, so leave room for it.
|
||||
"--gpu-memory-utilization",
|
||||
"0.5",
|
||||
"--port",
|
||||
str(port),
|
||||
"--trust-remote-code",
|
||||
]
|
||||
# VLLM_SERVER_DEV_MODE registers the dev endpoints; insecure serialization
|
||||
# lets the server unpickle the IPC handles sent over HTTP. Pin the server to
|
||||
# physical NPU 0 so its IPC UUID matches the trainer below.
|
||||
env_dict = {
|
||||
"VLLM_SERVER_DEV_MODE": "1",
|
||||
"VLLM_ALLOW_INSECURE_SERIALIZATION": "1",
|
||||
"ASCEND_RT_VISIBLE_DEVICES": str(INFERENCE_DEVICE_INDEX),
|
||||
"VLLM_ASCEND_ENABLE_NZ": "0",
|
||||
}
|
||||
|
||||
with RemoteOpenAIServer(
|
||||
MODEL_NAME,
|
||||
vllm_serve_args=server_args,
|
||||
server_host="127.0.0.1",
|
||||
server_port=port,
|
||||
env_dict=env_dict,
|
||||
auto_port=False,
|
||||
) as server:
|
||||
client = server.get_client()
|
||||
|
||||
outputs_before = _generate(client, MODEL_NAME, PROMPTS)
|
||||
|
||||
# Trainer shares physical NPU 0 with the server. It leaves
|
||||
# ASCEND_RT_VISIBLE_DEVICES unset (identity mapping logical 0 ->
|
||||
# physical 0) so both processes resolve to the same IPC UUID.
|
||||
torch.npu.set_device(INFERENCE_DEVICE_INDEX)
|
||||
train_model = _build_trainer_model(INFERENCE_DEVICE_INDEX)
|
||||
os.environ["VLLM_ALLOW_INSECURE_SERIALIZATION"] = "1"
|
||||
|
||||
from vllm_ascend.distributed.weight_transfer.npu_ipc_engine import (
|
||||
NPUIPCTrainerSendWeightsArgs,
|
||||
NPUIPCWeightTransferEngine,
|
||||
)
|
||||
|
||||
_post(server, "init_weight_transfer_engine", json={"init_info": {}})
|
||||
|
||||
_post(server, "pause")
|
||||
# The probe performs /start_weight_update when present, so it must not
|
||||
# be called again below. Older vLLM without the lifecycle endpoints is
|
||||
# out of scope for this IPC test.
|
||||
if not _has_lifecycle_endpoints(server):
|
||||
_post(server, "resume")
|
||||
pytest.skip("vLLM build lacks the /start_weight_update lifecycle endpoints required by NPU IPC.")
|
||||
|
||||
# trainer_send_weights POSTs to /update_weights itself; the server
|
||||
# rebuilds tensors locally and loads them before the POST returns (no
|
||||
# collective back to the trainer, so no background thread is needed).
|
||||
# train_model stays referenced so the shared NPU storage outlives it.
|
||||
trainer_args = NPUIPCTrainerSendWeightsArgs(send_mode="http", url=server.url_root)
|
||||
NPUIPCWeightTransferEngine.trainer_send_weights(
|
||||
iterator=train_model.named_parameters(),
|
||||
trainer_args=trainer_args,
|
||||
)
|
||||
|
||||
_post(server, "finish_weight_update")
|
||||
_post(server, "resume")
|
||||
|
||||
outputs_after = _generate(client, MODEL_NAME, PROMPTS)
|
||||
|
||||
assert outputs_after != outputs_before, "server weights did not change after NPU IPC transfer"
|
||||
30
tests/e2e/pull_request/one_card/test_qwen3_0_6b.py
Normal file
30
tests/e2e/pull_request/one_card/test_qwen3_0_6b.py
Normal file
@@ -0,0 +1,30 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
|
||||
from tests.e2e.conftest import wait_until_npu_memory_free
|
||||
from tests.e2e.pull_request.utils import PROMPTS_SHORT, compare_logprobs
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_dense_default_full_and_piecewise_graph():
|
||||
"""Verify dense generation on the default FULL_AND_PIECEWISE graph path."""
|
||||
runner_kwargs = {
|
||||
"model_name": "Qwen/Qwen3-0.6B",
|
||||
"max_model_len": 1024,
|
||||
"cudagraph_capture_sizes": [1, 2, 4, 8],
|
||||
}
|
||||
compare_logprobs(runner_kwargs=runner_kwargs, prompts=PROMPTS_SHORT)
|
||||
71
tests/e2e/pull_request/one_card/test_qwen3_5_0_8b.py
Normal file
71
tests/e2e/pull_request/one_card/test_qwen3_5_0_8b.py
Normal file
@@ -0,0 +1,71 @@
|
||||
#
|
||||
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
import huggingface_hub
|
||||
from huggingface_hub import snapshot_download as hf_snapshot_download
|
||||
from vllm.assets.image import ImageAsset
|
||||
|
||||
from tests.e2e.conftest import VllmRunner, qwen_prompt, wait_until_npu_memory_free
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_mamba_ssm_multimodal_reasoning_mtp_full_decode_only():
|
||||
"""Verify Mamba/SSM multimodal reasoning with MTP and full decode only."""
|
||||
image = ImageAsset("cherry_blossom").pil_image.convert("RGB")
|
||||
img_questions = [
|
||||
"What is the content of this image?",
|
||||
"Describe the content of this image in detail.",
|
||||
"What's in the image?",
|
||||
"Where is this image taken?",
|
||||
]
|
||||
images = [image] * len(img_questions)
|
||||
prompts = qwen_prompt(img_questions)
|
||||
|
||||
model_path = hf_snapshot_download(
|
||||
"Qwen/Qwen3.5-0.8B",
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
with VllmRunner(
|
||||
model_path,
|
||||
dtype="bfloat16",
|
||||
max_model_len=2048,
|
||||
max_num_batched_tokens=1024,
|
||||
limit_mm_per_prompt={"image": 1},
|
||||
mm_processor_kwargs={
|
||||
"min_pixels": 28 * 28,
|
||||
"max_pixels": 1280 * 28 * 28,
|
||||
"fps": 1,
|
||||
},
|
||||
compilation_config={
|
||||
"cudagraph_mode": "FULL_DECODE_ONLY",
|
||||
"cudagraph_capture_sizes": [2, 4, 6, 8],
|
||||
},
|
||||
speculative_config={
|
||||
"method": "mtp",
|
||||
"num_speculative_tokens": 1,
|
||||
},
|
||||
) as runner:
|
||||
outputs = runner.generate_greedy(
|
||||
prompts=prompts,
|
||||
images=images,
|
||||
max_tokens=64,
|
||||
)
|
||||
|
||||
assert len(outputs) == len(prompts)
|
||||
for _, output_str in outputs:
|
||||
assert output_str, "Generated output should not be empty."
|
||||
87
tests/e2e/pull_request/one_card/test_qwen3_8b_w8a8.py
Normal file
87
tests/e2e/pull_request/one_card/test_qwen3_8b_w8a8.py
Normal file
@@ -0,0 +1,87 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
from vllm import SamplingParams
|
||||
from vllm.config import CompilationConfig
|
||||
|
||||
from tests.e2e.conftest import VllmRunner, cleanup_dist_env_and_memory, wait_until_npu_memory_free
|
||||
from tests.e2e.pull_request.utils import PROMPTS_SHORT
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_dense_w8a8_eagle3_full_graph():
|
||||
"""Verify dense W8A8 inference with Eagle-3 speculative decoding."""
|
||||
example_prompts = PROMPTS_SHORT
|
||||
sampling_params = SamplingParams(
|
||||
max_tokens=300,
|
||||
temperature=0.0,
|
||||
ignore_eos=False,
|
||||
)
|
||||
|
||||
with VllmRunner(
|
||||
"vllm-ascend/Qwen3-8B-W8A8",
|
||||
tensor_parallel_size=1,
|
||||
pipeline_parallel_size=1,
|
||||
data_parallel_size=1,
|
||||
disable_log_stats=False,
|
||||
max_model_len=4096,
|
||||
seed=1024,
|
||||
async_scheduling=False,
|
||||
quantization="ascend",
|
||||
speculative_config={
|
||||
"disable_padded_drafter_batch": False,
|
||||
"method": "eagle3",
|
||||
"model": "RedHatAI/Qwen3-8B-speculator.eagle3",
|
||||
"num_speculative_tokens": 2,
|
||||
"draft_tensor_parallel_size": 1,
|
||||
"max_model_len": 128,
|
||||
},
|
||||
compilation_config=CompilationConfig(cudagraph_mode="FULL", cudagraph_capture_sizes=[5, 12]),
|
||||
) as runner:
|
||||
spec_outputs = runner.generate(example_prompts, sampling_params)
|
||||
cleanup_dist_env_and_memory()
|
||||
del runner
|
||||
|
||||
with VllmRunner(
|
||||
"vllm-ascend/Qwen3-8B-W8A8",
|
||||
tensor_parallel_size=1,
|
||||
pipeline_parallel_size=1,
|
||||
data_parallel_size=1,
|
||||
disable_log_stats=False,
|
||||
max_model_len=4096,
|
||||
seed=1024,
|
||||
async_scheduling=False,
|
||||
quantization="ascend",
|
||||
compilation_config=CompilationConfig(cudagraph_mode="FULL_DECODE_ONLY", cudagraph_capture_sizes=[12]),
|
||||
) as runner:
|
||||
ref_outputs = runner.generate(example_prompts, sampling_params)
|
||||
cleanup_dist_env_and_memory()
|
||||
del runner
|
||||
|
||||
matches = 0
|
||||
threshold = 0.66
|
||||
for ref_output, spec_output in zip(ref_outputs, spec_outputs):
|
||||
ref_token_ids = ref_output[0][0]
|
||||
spec_token_ids = spec_output[0][0]
|
||||
if ref_token_ids == spec_token_ids[: len(ref_token_ids)]:
|
||||
matches += 1
|
||||
else:
|
||||
print(f"ref_output: {ref_output[1][0]}")
|
||||
print(f"spec_output: {spec_output[1][0]}")
|
||||
|
||||
assert matches > int(threshold * len(ref_outputs))
|
||||
55
tests/e2e/pull_request/one_card/test_qwen3_embedding_0_6b.py
Normal file
55
tests/e2e/pull_request/one_card/test_qwen3_embedding_0_6b.py
Normal file
@@ -0,0 +1,55 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
import huggingface_hub
|
||||
from modelscope import snapshot_download as modelscope_snapshot_download # type: ignore[import-untyped]
|
||||
|
||||
from tests.e2e.conftest import HfRunner, VllmRunner, cleanup_dist_env_and_memory, wait_until_npu_memory_free
|
||||
from tests.e2e.utils import check_embeddings_close
|
||||
|
||||
|
||||
@wait_until_npu_memory_free()
|
||||
def test_embedding_full_decode_only():
|
||||
"""Verify embedding outputs with full decode only."""
|
||||
queries = ["What is the capital of China?", "Explain gravity"]
|
||||
model = "Qwen/Qwen3-Embedding-0.6B"
|
||||
model_name = modelscope_snapshot_download(
|
||||
model,
|
||||
local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE,
|
||||
)
|
||||
with VllmRunner(model_name, runner="pooling", max_model_len=None, cudagraph_capture_sizes=[4]) as vllm_runner:
|
||||
vllm_outputs = vllm_runner.embed(queries)
|
||||
cleanup_dist_env_and_memory()
|
||||
del vllm_runner
|
||||
|
||||
with HfRunner(
|
||||
model_name,
|
||||
dtype="float32",
|
||||
is_sentence_transformer=True,
|
||||
) as hf_runner:
|
||||
hf_outputs = hf_runner.encode(queries)
|
||||
cleanup_dist_env_and_memory()
|
||||
del hf_runner
|
||||
|
||||
check_embeddings_close(
|
||||
embeddings_0_lst=hf_outputs,
|
||||
embeddings_1_lst=vllm_outputs,
|
||||
name_0="hf",
|
||||
name_1="vllm",
|
||||
tol=1e-2,
|
||||
)
|
||||
105
tests/e2e/pull_request/one_card/test_sampler.py
Normal file
105
tests/e2e/pull_request/one_card/test_sampler.py
Normal file
@@ -0,0 +1,105 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/entrypoints/llm/test_guided_generate.py
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
import gc
|
||||
import os
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
from vllm import LLM, SamplingParams
|
||||
|
||||
from tests.e2e.conftest import ModelName, cleanup_dist_env_and_memory, model_cache
|
||||
|
||||
os.environ["VLLM_BATCH_INVARIANT"] = "1"
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=ModelName.QWEN3_06B,
|
||||
quantization=None,
|
||||
max_model_len=8192,
|
||||
dtype="bfloat16",
|
||||
gpu_memory_utilization=0.9,
|
||||
enable_prefix_caching=False,
|
||||
max_num_seqs=32,
|
||||
tensor_parallel_size=1,
|
||||
distributed_executor_backend="mp",
|
||||
compilation_config={"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1, 32, 64]},
|
||||
)
|
||||
def test_qwen3_topk(vllm_runner) -> None:
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
sampling_params = SamplingParams(max_tokens=5, temperature=0.0, top_k=50, top_p=0.9)
|
||||
vllm_runner.generate(example_prompts, sampling_params)
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
@pytest.mark.model(
|
||||
model_name=ModelName.QWEN3_06B,
|
||||
quantization=None,
|
||||
max_model_len=8192,
|
||||
dtype="bfloat16",
|
||||
gpu_memory_utilization=0.9,
|
||||
enable_prefix_caching=False,
|
||||
max_num_seqs=32,
|
||||
tensor_parallel_size=1,
|
||||
distributed_executor_backend="mp",
|
||||
compilation_config={"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1, 32, 64]},
|
||||
)
|
||||
def test_qwen3_prompt_logprobs(vllm_runner) -> None:
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
vllm_runner.generate_greedy_logprobs(example_prompts, max_tokens=5, num_logprobs=1)
|
||||
|
||||
|
||||
@pytest.mark.timeout(1000)
|
||||
def test_qwen3_exponential_overlap(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# enable_async_exponential is mutually exclusive with VLLM_BATCH_INVARIANT
|
||||
# (see vllm_ascend/ascend_config.py). The module-level os.environ setting
|
||||
# would silently disable async_exponential, so this test creates its own
|
||||
# LLM instance with batch invariant mode turned off.
|
||||
model_cache.clear()
|
||||
gc.collect()
|
||||
torch.npu.empty_cache()
|
||||
|
||||
monkeypatch.setenv("VLLM_BATCH_INVARIANT", "0")
|
||||
|
||||
llm = LLM(
|
||||
model=ModelName.QWEN3_06B,
|
||||
quantization=None,
|
||||
max_model_len=8192,
|
||||
dtype="bfloat16",
|
||||
gpu_memory_utilization=0.9,
|
||||
enable_prefix_caching=False,
|
||||
max_num_seqs=32,
|
||||
tensor_parallel_size=1,
|
||||
distributed_executor_backend="mp",
|
||||
compilation_config={"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1, 2, 4, 8]},
|
||||
additional_config={"enable_async_exponential": True},
|
||||
)
|
||||
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
]
|
||||
sampling_params = SamplingParams(max_tokens=5, temperature=1.0, top_k=50, top_p=0.9)
|
||||
llm.generate(example_prompts, sampling_params)
|
||||
|
||||
del llm
|
||||
cleanup_dist_env_and_memory()
|
||||
103
tests/e2e/pull_request/one_card/test_simple_cpu_offload.py
Normal file
103
tests/e2e/pull_request/one_card/test_simple_cpu_offload.py
Normal file
@@ -0,0 +1,103 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
"""End-to-end tests for the Ascend ``SimpleCPUOffloadConnector``.
|
||||
|
||||
The simple CPU offloading scheduler/worker pair is reused from upstream
|
||||
vLLM; here we only exercise the NPU-native worker path
|
||||
(``aclrtMemcpyBatchAsync`` + ``torch.npu`` streams) to confirm that
|
||||
KV blocks are stored to and reloaded from CPU correctly on Ascend.
|
||||
"""
|
||||
|
||||
import os
|
||||
import time
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams, TokensPrompt
|
||||
from vllm.config import KVTransferConfig
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"
|
||||
|
||||
|
||||
def _build_kv_transfer_config(
|
||||
cpu_bytes_to_use: int,
|
||||
lazy_offload: bool = False,
|
||||
) -> KVTransferConfig:
|
||||
return KVTransferConfig(
|
||||
kv_connector="SimpleCPUOffloadConnector",
|
||||
kv_role="kv_both",
|
||||
kv_connector_extra_config={
|
||||
"cpu_bytes_to_use": cpu_bytes_to_use,
|
||||
"lazy_offload": lazy_offload,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def test_simple_cpu_offload_accuracy() -> None:
|
||||
"""Reset GPU prefix cache after a cold run; verify the CPU-loaded KV
|
||||
cache reproduces the cold-run output deterministically."""
|
||||
sampling_params = SamplingParams(max_tokens=1, temperature=0)
|
||||
|
||||
# Long enough prompt to occupy multiple full KV blocks.
|
||||
prompt = "hi " * 500 + "Let's count to ten. One, two, three, "
|
||||
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3-0.6B",
|
||||
max_model_len=4096,
|
||||
gpu_memory_utilization=0.5,
|
||||
enable_prefix_caching=True,
|
||||
kv_transfer_config=_build_kv_transfer_config(1 << 30), # 1 GiB
|
||||
enforce_eager=True,
|
||||
) as runner:
|
||||
llm = runner.model
|
||||
|
||||
# Cold run — populates GPU cache and triggers CPU offload.
|
||||
cold_output = llm.generate(prompt, sampling_params, use_tqdm=False)[0]
|
||||
expected = cold_output.outputs[0].text
|
||||
|
||||
success = 0
|
||||
attempts = 5
|
||||
for _ in range(attempts):
|
||||
# Let the engine core drain pending store transfers.
|
||||
time.sleep(2)
|
||||
# Reset GPU prefix cache so the next run must reload from CPU.
|
||||
if not llm.reset_prefix_cache():
|
||||
continue
|
||||
output = llm.generate(prompt, sampling_params, use_tqdm=False)[0]
|
||||
if output.outputs[0].text == expected:
|
||||
success += 1
|
||||
|
||||
assert success >= int(0.5 * attempts), (
|
||||
f"CPU-load accuracy too low: {success}/{attempts} matched baseline output {expected!r}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("lazy", [False, True])
|
||||
def test_simple_cpu_offload_no_crash_on_repeat(lazy: bool) -> None:
|
||||
"""Smoke test: many short generations exercise both eager and lazy
|
||||
offload paths without errors and yield non-empty outputs."""
|
||||
sampling_params = SamplingParams(max_tokens=4, temperature=0)
|
||||
prompt_token_ids = [0] * 257
|
||||
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3-0.6B",
|
||||
max_model_len=2048,
|
||||
gpu_memory_utilization=0.5,
|
||||
enable_prefix_caching=True,
|
||||
kv_transfer_config=_build_kv_transfer_config(
|
||||
cpu_bytes_to_use=512 * (1 << 20), # 512 MiB
|
||||
lazy_offload=lazy,
|
||||
),
|
||||
enforce_eager=True,
|
||||
) as runner:
|
||||
llm = runner.model
|
||||
for i in range(8):
|
||||
prompt_token_ids[0] = i
|
||||
prompts = [TokensPrompt(prompt_token_ids=prompt_token_ids)]
|
||||
outs = llm.generate(prompts, sampling_params, use_tqdm=False)
|
||||
assert outs and len(outs[0].outputs[0].token_ids) > 0
|
||||
137
tests/e2e/pull_request/one_card/test_vlm.py
Normal file
137
tests/e2e/pull_request/one_card/test_vlm.py
Normal file
@@ -0,0 +1,137 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
# Adapted from vllm/tests/basic_correctness/test_basic_correctness.py
|
||||
#
|
||||
"""Compare the short outputs of HF and vLLM when using greedy sampling.
|
||||
|
||||
Run `pytest tests/e2e/pull_request/one_card/test_vlm.py`.
|
||||
"""
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from vllm import SamplingParams
|
||||
from vllm.assets.audio import AudioAsset
|
||||
from vllm.assets.image import ImageAsset
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
WHISPER_MODELS = [
|
||||
"openai-mirror/whisper-large-v3-turbo",
|
||||
]
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"VLLM_WORKER_MULTIPROC_METHOD": "spawn"})
|
||||
def test_multimodal_vl(vl_config):
|
||||
image = ImageAsset("cherry_blossom").pil_image.convert("RGB")
|
||||
|
||||
img_questions = [
|
||||
"What is the content of this image?",
|
||||
"Describe the content of this image in detail.",
|
||||
"What's in the image?",
|
||||
"Where is this image taken?",
|
||||
]
|
||||
|
||||
images = [image] * len(img_questions)
|
||||
prompts = vl_config["prompt_fn"](img_questions)
|
||||
|
||||
with VllmRunner(
|
||||
vl_config["model"],
|
||||
mm_processor_kwargs=vl_config["mm_processor_kwargs"],
|
||||
max_model_len=8192,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
limit_mm_per_prompt={"image": 1},
|
||||
) as vllm_model:
|
||||
outputs = vllm_model.generate_greedy(
|
||||
prompts=prompts,
|
||||
images=images,
|
||||
max_tokens=64,
|
||||
)
|
||||
|
||||
assert len(outputs) == len(prompts)
|
||||
|
||||
for _, output_str in outputs:
|
||||
assert output_str, "Generated output should not be empty."
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"VLLM_WORKER_MULTIPROC_METHOD": "spawn"})
|
||||
def test_multimodal_vl_language_model_only():
|
||||
example_prompts = [
|
||||
"Hello, my name is",
|
||||
"The president of the United States is",
|
||||
"The capital of France is",
|
||||
"The future of AI is",
|
||||
]
|
||||
max_tokens = 5
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3-VL-8B-Instruct",
|
||||
max_model_len=4096,
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
gpu_memory_utilization=0.90,
|
||||
language_model_only=True,
|
||||
) as vllm_model:
|
||||
vllm_model.generate_greedy(example_prompts, max_tokens)
|
||||
|
||||
|
||||
@patch.dict(os.environ, {"VLLM_WORKER_MULTIPROC_METHOD": "spawn"})
|
||||
def test_multimodal_audio():
|
||||
audio_prompt = "".join([f"Audio {idx + 1}: <|audio_bos|><|AUDIO|><|audio_eos|>\n" for idx in range(2)])
|
||||
question = "What sport and what nursery rhyme are referenced?"
|
||||
prompt = (
|
||||
"<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n"
|
||||
"<|im_start|>user\n"
|
||||
f"{audio_prompt}{question}<|im_end|>\n"
|
||||
"<|im_start|>assistant\n"
|
||||
)
|
||||
mm_data = {
|
||||
"audio": [asset.audio_and_sample_rate for asset in [AudioAsset("mary_had_lamb"), AudioAsset("winning_call")]]
|
||||
}
|
||||
inputs = {"prompt": prompt, "multi_modal_data": mm_data}
|
||||
|
||||
sampling_params = SamplingParams(temperature=0.2, max_tokens=10, stop_token_ids=None)
|
||||
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen2-Audio-7B-Instruct",
|
||||
max_model_len=4096,
|
||||
max_num_seqs=5,
|
||||
dtype="bfloat16",
|
||||
limit_mm_per_prompt={"audio": 2},
|
||||
cudagraph_capture_sizes=[1, 2, 4, 8],
|
||||
gpu_memory_utilization=0.9,
|
||||
) as runner:
|
||||
outputs = runner.generate(inputs, sampling_params=sampling_params)
|
||||
|
||||
assert outputs is not None, "Generated outputs should not be None."
|
||||
assert len(outputs) > 0, "Generated outputs should not be empty."
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", WHISPER_MODELS)
|
||||
@patch.dict(os.environ, {"VLLM_WORKER_MULTIPROC_METHOD": "spawn"})
|
||||
def test_whisper(model) -> None:
|
||||
prompts = ["<|startoftranscript|><|en|><|transcribe|><|notimestamps|>"]
|
||||
audios = [AudioAsset("mary_had_lamb").audio_and_sample_rate]
|
||||
|
||||
sampling_params = SamplingParams(temperature=0.2, max_tokens=10, stop_token_ids=None)
|
||||
|
||||
with VllmRunner(
|
||||
model, max_model_len=448, max_num_seqs=5, dtype="bfloat16", block_size=128, gpu_memory_utilization=0.9
|
||||
) as runner:
|
||||
outputs = runner.generate(prompts=prompts, audios=audios, sampling_params=sampling_params)
|
||||
|
||||
assert outputs is not None, "Generated outputs should not be None."
|
||||
assert len(outputs) > 0, "Generated outputs should not be empty."
|
||||
65
tests/e2e/pull_request/one_card/test_xlite.py
Normal file
65
tests/e2e/pull_request/one_card/test_xlite.py
Normal file
@@ -0,0 +1,65 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# Copyright 2023 The vLLM team.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
"""
|
||||
Compare the outputs of vLLM with and without xlite via logprob-based accuracy
|
||||
check (3 tokens: 1 prefill + 2 decode).
|
||||
|
||||
Run `pytest tests/e2e/pull_request/one_card/test_xlite.py`.
|
||||
"""
|
||||
|
||||
# ruff: noqa: E501
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.e2e.pull_request.utils import PROMPTS_SHORT, LLMTestCase, compare_logprobs
|
||||
|
||||
os.environ["VLLM_ASCEND_ENABLE_NZ"] = "2"
|
||||
|
||||
CASE_DECODE_ONLY = LLMTestCase(
|
||||
model="Qwen/Qwen3-0.6B",
|
||||
prompts=PROMPTS_SHORT,
|
||||
)
|
||||
|
||||
CASE_FULL = LLMTestCase(
|
||||
model="Qwen/Qwen3-0.6B",
|
||||
prompts=PROMPTS_SHORT,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="TODO: Re-enable xlite_decode_only e2e test when stable.")
|
||||
@pytest.mark.parametrize("cur_case", [CASE_DECODE_ONLY])
|
||||
def test_models_with_xlite_decode_only(cur_case: LLMTestCase):
|
||||
runner_kwargs = {
|
||||
"model_name": cur_case.model,
|
||||
"max_model_len": 1024,
|
||||
"block_size": 128,
|
||||
"additional_config": {"xlite_graph_config": {"enabled": True}},
|
||||
}
|
||||
compare_logprobs(runner_kwargs=runner_kwargs, prompts=cur_case.prompts)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cur_case", [CASE_FULL])
|
||||
def test_models_with_xlite_full_mode(cur_case: LLMTestCase):
|
||||
runner_kwargs = {
|
||||
"model_name": cur_case.model,
|
||||
"max_model_len": 1024,
|
||||
"block_size": 128,
|
||||
"additional_config": {"xlite_graph_config": {"enabled": True, "full_mode": True}},
|
||||
}
|
||||
compare_logprobs(runner_kwargs=runner_kwargs, prompts=cur_case.prompts)
|
||||
Reference in New Issue
Block a user