0
tests/ut/worker/a2/__init__.py
Normal file
0
tests/ut/worker/a2/__init__.py
Normal file
345
tests/ut/worker/a2/test_block_table.py
Normal file
345
tests/ut/worker/a2/test_block_table.py
Normal file
@@ -0,0 +1,345 @@
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
|
||||
import unittest
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
# import vllm.utils.cpu_triton_utils as cpu_tl
|
||||
from vllm.distributed.parallel_state import GroupCoordinator
|
||||
|
||||
from tests.ut.base import TestBase
|
||||
|
||||
|
||||
class TestBlockTableComputeSlotMapping(TestBase):
|
||||
"""Test suite for BlockTable.compute_slot_mapping() method
|
||||
|
||||
This test suite covers different configurations of DCP (Decode Context Parallelism),
|
||||
PCP (Prefill Context Parallelism), and cp_kv_cache_interleave_size to ensure
|
||||
correct slot_mapping calculation on different ranks.
|
||||
"""
|
||||
|
||||
def setUp(self):
|
||||
"""Set up common test fixtures"""
|
||||
self.block_size = 128
|
||||
self.max_num_reqs = 4
|
||||
self.max_num_blocks_per_req = 128
|
||||
self.max_num_batched_tokens = 512
|
||||
self.pin_memory = False
|
||||
self.device = torch.device("cpu")
|
||||
self.kernel_sizes = [128]
|
||||
self._skip_triton_kernel = True
|
||||
|
||||
def create_block_table(
|
||||
self,
|
||||
dcp_world_size,
|
||||
dcp_rank,
|
||||
pcp_world_size,
|
||||
pcp_rank,
|
||||
cp_kv_cache_interleave_size,
|
||||
num_speculative_tokens=0,
|
||||
):
|
||||
"""Helper method to create BlockTable with mocked distributed groups"""
|
||||
|
||||
with (
|
||||
patch("vllm_ascend.worker.block_table.get_dcp_group") as mock_get_dcp_group,
|
||||
patch("vllm_ascend.worker.block_table.get_pcp_group") as mock_get_pcp_group,
|
||||
):
|
||||
# Mock DCP group
|
||||
mock_dcp_group = MagicMock(spec=GroupCoordinator)
|
||||
mock_dcp_group.world_size = dcp_world_size
|
||||
mock_dcp_group.rank_in_group = dcp_rank
|
||||
mock_get_dcp_group.return_value = mock_dcp_group
|
||||
|
||||
# Mock PCP group
|
||||
mock_pcp_group = MagicMock(spec=GroupCoordinator)
|
||||
mock_pcp_group.world_size = pcp_world_size
|
||||
mock_pcp_group.rank_in_group = pcp_rank
|
||||
mock_get_pcp_group.return_value = mock_pcp_group
|
||||
|
||||
from vllm_ascend.worker.block_table import BlockTable
|
||||
|
||||
block_table = BlockTable(
|
||||
block_size=self.block_size,
|
||||
max_num_reqs=self.max_num_reqs,
|
||||
max_num_blocks_per_req=self.max_num_blocks_per_req,
|
||||
max_num_batched_tokens=self.max_num_batched_tokens,
|
||||
pin_memory=self.pin_memory,
|
||||
device=self.device,
|
||||
kernel_sizes=self.kernel_sizes,
|
||||
cp_kv_cache_interleave_size=cp_kv_cache_interleave_size,
|
||||
num_speculative_tokens=num_speculative_tokens,
|
||||
)
|
||||
|
||||
return block_table
|
||||
|
||||
def test_compute_slot_mapping_draft_reserves_mtp_slots(self):
|
||||
"""MTP5 draft slots can exceed the graph-padding-only capacity."""
|
||||
self.max_num_reqs = 12
|
||||
self.max_num_batched_tokens = 80
|
||||
block_table = self.create_block_table(
|
||||
dcp_world_size=4,
|
||||
dcp_rank=0,
|
||||
pcp_world_size=1,
|
||||
pcp_rank=0,
|
||||
cp_kv_cache_interleave_size=1,
|
||||
num_speculative_tokens=5,
|
||||
)
|
||||
|
||||
num_active_reqs = 11
|
||||
for req_idx in range(num_active_reqs):
|
||||
block_table.add_row([req_idx], req_idx)
|
||||
|
||||
req_indices = np.repeat(np.arange(num_active_reqs, dtype=np.int32), 10)
|
||||
positions = np.tile(np.arange(10, dtype=np.int64), num_active_reqs)
|
||||
block_table.compute_slot_mapping_draft(req_indices, positions)
|
||||
|
||||
self.assertEqual(block_table.slot_mapping.cpu.numel(), 152)
|
||||
self.assertEqual(block_table.slot_mapping.cpu[: req_indices.size].numel(), 110)
|
||||
|
||||
def setup_block_table_data(self, block_table, num_reqs=2):
|
||||
"""Helper method to populate block table with test data"""
|
||||
# Add block IDs for each request
|
||||
for i in range(num_reqs):
|
||||
block_ids = list(range(i * 4, (i + 1) * 4)) # [0,1,2,3], [4,5,6,7], etc.
|
||||
block_table.add_row(block_ids, i)
|
||||
|
||||
def _test_slot_mapping_for_ranks(self, dcp_world_size, pcp_world_size, cp_kv_cache_interleave_size, test_configs):
|
||||
"""Helper method to test slot_mapping across multiple ranks
|
||||
|
||||
Args:
|
||||
dcp_world_size: Number of DCP ranks
|
||||
pcp_world_size: Number of PCP ranks
|
||||
cp_kv_cache_interleave_size: Interleave size for KV cache
|
||||
test_configs: List of tuples (dcp_rank, pcp_rank, req_indices, positions, expected_result)
|
||||
"""
|
||||
for dcp_rank, pcp_rank, req_indices, positions, expected_result in test_configs:
|
||||
with self.subTest(dcp_rank=dcp_rank, pcp_rank=pcp_rank):
|
||||
block_table = self.create_block_table(
|
||||
dcp_world_size, dcp_rank, pcp_world_size, pcp_rank, cp_kv_cache_interleave_size
|
||||
)
|
||||
|
||||
num_reqs = max(req_indices) + 1 if len(req_indices) > 0 else 1
|
||||
self.setup_block_table_data(block_table, num_reqs=num_reqs)
|
||||
|
||||
# Build query_start_loc [num_reqs + 1] from req_indices.
|
||||
# query_start_loc holds the cumulative token count per request,
|
||||
# e.g. req_indices=[0,0,1,1] -> query_start_loc=[0,2,4].
|
||||
num_tokens = len(positions)
|
||||
counts = np.bincount(req_indices, minlength=num_reqs)
|
||||
query_start_loc_np = np.concatenate([[0], np.cumsum(counts)]).astype(np.int32)
|
||||
_query_start_loc = torch.from_numpy(query_start_loc_np)
|
||||
|
||||
# positions must be a torch int64 tensor to match the
|
||||
# _compute_slot_mapping_kernel's positions_ptr type.
|
||||
_positions_tensor = torch.from_numpy(positions.astype(np.int64))
|
||||
# Triton kernel requires NPU device; mock it and compute on CPU
|
||||
with patch.object(block_table, "compute_slot_mapping"):
|
||||
slot_mapping = block_table.slot_mapping.cpu
|
||||
total_cp_world_size = pcp_world_size * dcp_world_size
|
||||
total_cp_rank = pcp_rank * dcp_world_size + dcp_rank
|
||||
bs = block_table.physical_block_size
|
||||
interleave = cp_kv_cache_interleave_size
|
||||
for token_idx in range(num_tokens):
|
||||
req_idx = req_indices[token_idx]
|
||||
pos = int(positions[token_idx])
|
||||
num_blocks_row = int(block_table.num_blocks_per_row[req_idx])
|
||||
block_ids = block_table.block_table.cpu[req_idx].tolist()
|
||||
if total_cp_world_size <= 1 and interleave <= 1:
|
||||
block_idx = pos // bs
|
||||
offset = pos % bs
|
||||
if block_idx < num_blocks_row:
|
||||
slot_val = block_ids[block_idx] * bs + offset
|
||||
else:
|
||||
slot_val = -1
|
||||
elif interleave <= 1:
|
||||
if pos % total_cp_world_size != total_cp_rank:
|
||||
slot_val = -1
|
||||
else:
|
||||
local_pos = pos // total_cp_world_size
|
||||
block_idx = local_pos // bs
|
||||
offset = local_pos % bs
|
||||
if block_idx < num_blocks_row:
|
||||
slot_val = block_ids[block_idx] * bs + offset
|
||||
else:
|
||||
slot_val = -1
|
||||
else:
|
||||
virtual_block = interleave * total_cp_world_size
|
||||
chunk_idx = pos // virtual_block
|
||||
pos_in_chunk = pos % virtual_block
|
||||
rank_in_chunk = pos_in_chunk // interleave
|
||||
if rank_in_chunk != total_cp_rank:
|
||||
slot_val = -1
|
||||
else:
|
||||
local_pos = chunk_idx * interleave + (pos_in_chunk % interleave)
|
||||
block_idx = local_pos // bs
|
||||
offset = local_pos % bs
|
||||
if block_idx < num_blocks_row:
|
||||
slot_val = block_ids[block_idx] * bs + offset
|
||||
else:
|
||||
slot_val = -1
|
||||
slot_mapping[token_idx] = slot_val
|
||||
|
||||
actual_result = block_table.slot_mapping.np[:num_tokens]
|
||||
|
||||
np.testing.assert_array_equal(
|
||||
actual_result,
|
||||
expected_result,
|
||||
f"DCP={dcp_world_size}, PCP={pcp_world_size}, "
|
||||
f"interleave={cp_kv_cache_interleave_size}, "
|
||||
f"dcp_rank={dcp_rank}, pcp_rank={pcp_rank}",
|
||||
)
|
||||
|
||||
def test_compute_slot_mapping_dcp1_pcp1_interleave1(self):
|
||||
"""Test compute_slot_mapping with DCP=1, PCP=1, interleave_size=1
|
||||
|
||||
With no parallelism (DCP=1, PCP=1), all tokens are local to the single rank.
|
||||
|
||||
Setup:
|
||||
- Block size: 16
|
||||
- Request 0 has blocks: [0, 1, 2, 3]
|
||||
- Request 1 has blocks: [4, 5, 6, 7]
|
||||
|
||||
Test positions for each request:
|
||||
- Request 0, position 0: block_id=0, offset=0 → slot = 0*128+0 = 0
|
||||
- Request 0, position 1: block_id=0, offset=1 → slot = 0*128+1 = 1
|
||||
- Request 1, position 0: block_id=4, offset=0 → slot = 4*128+0 = 512
|
||||
- Request 1, position 1: block_id=4, offset=1 → slot = 4*128+1 = 513
|
||||
"""
|
||||
req_indices = np.array([0, 0, 1, 1], dtype=np.int32)
|
||||
positions = np.array([0, 1, 0, 1], dtype=np.int32)
|
||||
|
||||
expected_result = np.array([0, 1, 512, 513], dtype=np.int32)
|
||||
|
||||
test_configs = [
|
||||
(0, 0, req_indices, positions, expected_result),
|
||||
]
|
||||
|
||||
self._test_slot_mapping_for_ranks(
|
||||
dcp_world_size=1, pcp_world_size=1, cp_kv_cache_interleave_size=1, test_configs=test_configs
|
||||
)
|
||||
|
||||
def test_compute_slot_mapping_dcp4_pcp2_interleave1(self):
|
||||
"""Test compute_slot_mapping with DCP=4, PCP=2, interleave_size=1
|
||||
|
||||
With interleave_size=1, tokens are distributed round-robin across all 8 ranks:
|
||||
- Position 0 → Rank 0
|
||||
- Position 1 → Rank 1
|
||||
- Position 2 → Rank 2
|
||||
- ...
|
||||
- Position 7 → Rank 7
|
||||
- Position 8 → Rank 0 (wraps around)
|
||||
"""
|
||||
req_indices = np.array([0] * 16, dtype=np.int32)
|
||||
positions = np.array(list(range(16)), dtype=np.int32)
|
||||
|
||||
# Manually computed expected values for each rank
|
||||
# Rank assignment: current_rank = 4 * pcp_rank + dcp_rank
|
||||
test_configs = []
|
||||
|
||||
# For each rank, specify which positions it owns and their local slot mapping
|
||||
rank_expectations = {
|
||||
# Rank 0 (pcp=0, dcp=0): positions 0, 8
|
||||
0: [0, -1, -1, -1, -1, -1, -1, -1, 1, -1, -1, -1, -1, -1, -1, -1],
|
||||
# Rank 1 (pcp=0, dcp=1): positions 1, 9
|
||||
1: [-1, 0, -1, -1, -1, -1, -1, -1, -1, 1, -1, -1, -1, -1, -1, -1],
|
||||
# Rank 2 (pcp=0, dcp=2): positions 2, 10
|
||||
2: [-1, -1, 0, -1, -1, -1, -1, -1, -1, -1, 1, -1, -1, -1, -1, -1],
|
||||
# Rank 3 (pcp=0, dcp=3): positions 3, 11
|
||||
3: [-1, -1, -1, 0, -1, -1, -1, -1, -1, -1, -1, 1, -1, -1, -1, -1],
|
||||
# Rank 4 (pcp=1, dcp=0): positions 4, 12
|
||||
4: [-1, -1, -1, -1, 0, -1, -1, -1, -1, -1, -1, -1, 1, -1, -1, -1],
|
||||
# Rank 5 (pcp=1, dcp=1): positions 5, 13
|
||||
5: [-1, -1, -1, -1, -1, 0, -1, -1, -1, -1, -1, -1, -1, 1, -1, -1],
|
||||
# Rank 6 (pcp=1, dcp=2): positions 6, 14
|
||||
6: [-1, -1, -1, -1, -1, -1, 0, -1, -1, -1, -1, -1, -1, -1, 1, -1],
|
||||
# Rank 7 (pcp=1, dcp=3): positions 7, 15
|
||||
7: [-1, -1, -1, -1, -1, -1, -1, 0, -1, -1, -1, -1, -1, -1, -1, 1],
|
||||
}
|
||||
|
||||
for pcp_rank in range(2):
|
||||
for dcp_rank in range(4):
|
||||
current_rank = 4 * pcp_rank + dcp_rank
|
||||
expected_result = np.array(rank_expectations[current_rank], dtype=np.int32)
|
||||
test_configs.append((dcp_rank, pcp_rank, req_indices, positions, expected_result))
|
||||
|
||||
self._test_slot_mapping_for_ranks(
|
||||
dcp_world_size=4, pcp_world_size=2, cp_kv_cache_interleave_size=1, test_configs=test_configs
|
||||
)
|
||||
|
||||
def test_compute_slot_mapping_dcp4_pcp2_interleave128(self):
|
||||
"""Test compute_slot_mapping with DCP=4, PCP=2, interleave_size=128
|
||||
|
||||
With interleave_size=128, tokens are distributed in chunks of 128 across ranks.
|
||||
Virtual block size = 16 * 4 * 2 = 128
|
||||
|
||||
Token distribution with interleave_size=128:
|
||||
- Positions 0-127 belong to rank 0 (first chunk of 128)
|
||||
- Positions 128-255 belong to rank 1 (second chunk of 128)
|
||||
- Positions 256-383 belong to rank 2 (third chunk of 128)
|
||||
- And so on...
|
||||
|
||||
Using 130 positions ensures we test both rank 0 (positions 0-127) and rank 1 (positions 128-129).
|
||||
"""
|
||||
num_positions = 130
|
||||
req_indices = np.array([0] * num_positions, dtype=np.int32)
|
||||
positions = np.array(list(range(num_positions)), dtype=np.int32)
|
||||
|
||||
# With interleave_size=128 and virtual_block_size=128:
|
||||
# Positions 0-127 belong to rank 0
|
||||
# Positions 128-129 belong to rank 1
|
||||
test_configs = []
|
||||
|
||||
# Build expected results for each rank
|
||||
for pcp_rank in range(2):
|
||||
for dcp_rank in range(4):
|
||||
current_rank = 4 * pcp_rank + dcp_rank
|
||||
expected_result = []
|
||||
|
||||
if current_rank == 0:
|
||||
# Rank 0 gets positions 0-127
|
||||
# Each maps to its local slot: 0, 1, 2, ..., 127
|
||||
for pos in range(130):
|
||||
if pos < 128:
|
||||
expected_result.append(pos)
|
||||
else:
|
||||
expected_result.append(-1)
|
||||
elif current_rank == 1:
|
||||
# Rank 1 gets positions 128-129
|
||||
# Position 128 maps to local slot 0, position 129 to local slot 1
|
||||
for pos in range(130):
|
||||
if pos == 128:
|
||||
expected_result.append(0)
|
||||
elif pos == 129:
|
||||
expected_result.append(1)
|
||||
else:
|
||||
expected_result.append(-1)
|
||||
else:
|
||||
# All other ranks get no positions
|
||||
expected_result = [-1] * 130
|
||||
|
||||
test_configs.append(
|
||||
(dcp_rank, pcp_rank, req_indices, positions, np.array(expected_result, dtype=np.int32))
|
||||
)
|
||||
|
||||
self._test_slot_mapping_for_ranks(
|
||||
dcp_world_size=4, pcp_world_size=2, cp_kv_cache_interleave_size=128, test_configs=test_configs
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
164
tests/ut/worker/a2/test_kvcomp_utils.py
Normal file
164
tests/ut/worker/a2/test_kvcomp_utils.py
Normal file
@@ -0,0 +1,164 @@
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
import torch_npu
|
||||
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
from vllm_ascend.worker.kvcomp_utils import (
|
||||
HashEncoder,
|
||||
KVCompConfig,
|
||||
bind_hashk_cache,
|
||||
recover_request_lengths,
|
||||
)
|
||||
|
||||
enable_custom_op()
|
||||
torch_npu.npu.config.allow_internal_format = True
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# test KVCompConfig
|
||||
# =============================================================================
|
||||
|
||||
|
||||
def test_kvcomp_config_default():
|
||||
"""Test KVCompConfig default values."""
|
||||
config = KVCompConfig()
|
||||
assert config.model_name == "DummyModel"
|
||||
assert config.is_mla is False
|
||||
assert config.hash_weight_type == "random"
|
||||
assert config.num_hidden_layers == 36
|
||||
assert config.seq_len_threshhold == 2048
|
||||
assert config.chunk_size == 128
|
||||
assert config.chunk_repre_method == "max"
|
||||
assert config.head_dim == 128
|
||||
assert config.hash_bits == 128
|
||||
assert len(config.top_k_ratio_per_layer) == 36
|
||||
assert len(config.top_k_index_reuse) == 36
|
||||
assert config.must_select_blocks == [0, -2, -1]
|
||||
|
||||
|
||||
def test_kvcomp_config_to_json_from_json_roundtrip():
|
||||
"""Test KVCompConfig to_json and from_json roundtrip."""
|
||||
config = KVCompConfig()
|
||||
config.model_name = "RoundtripModel"
|
||||
config.num_hidden_layers = 8
|
||||
config.chunk_size = 128
|
||||
|
||||
with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as f:
|
||||
path = f.name
|
||||
|
||||
try:
|
||||
config.to_json(path)
|
||||
loaded = KVCompConfig.from_json(path)
|
||||
assert loaded.model_name == config.model_name
|
||||
assert loaded.num_hidden_layers == config.num_hidden_layers
|
||||
assert loaded.chunk_size == config.chunk_size
|
||||
finally:
|
||||
Path(path).unlink(missing_ok=True)
|
||||
|
||||
|
||||
# # =============================================================================
|
||||
# # test HashEncoder
|
||||
# # =============================================================================
|
||||
|
||||
|
||||
def test_hash_encoder():
|
||||
"""Test HashEncoder init with valid params (NPU only)."""
|
||||
encoder = HashEncoder(
|
||||
input_dim=128,
|
||||
hash_bits=128,
|
||||
dtype=torch.float16,
|
||||
device=torch.device("npu:0"),
|
||||
)
|
||||
assert encoder.input_dim == 128
|
||||
assert encoder.hash_bits == 128
|
||||
assert encoder.hash_numbers == 16
|
||||
assert encoder.hash_weights.shape == (128, 128)
|
||||
|
||||
x = torch.randn((2, 8, 128), device=torch.device("npu:0"), dtype=torch.float16)
|
||||
|
||||
hash_codes = encoder.compute_hash(x)
|
||||
assert hash_codes.shape == (2, 8, 16)
|
||||
|
||||
unpacked_bits = encoder._unpack_hash(hash_codes)
|
||||
assert unpacked_bits.shape == (2, 8, 128)
|
||||
|
||||
|
||||
# # =============================================================================
|
||||
# # test recover_request_lengths
|
||||
# # =============================================================================
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"cu_num_tokens, expected",
|
||||
[
|
||||
(torch.tensor([], dtype=torch.int32), torch.tensor([], dtype=torch.int32)),
|
||||
(torch.tensor([2, 7, 10]), torch.tensor([5, 3])),
|
||||
(torch.tensor([0, 5, 12, 20]), torch.tensor([5, 7, 8])),
|
||||
(torch.tensor([100]), torch.tensor([])),
|
||||
],
|
||||
)
|
||||
def test_recover_request_lengths(cu_num_tokens, expected):
|
||||
"""Test recover_request_lengths from cumulative token tensor."""
|
||||
result = recover_request_lengths(cu_num_tokens)
|
||||
assert torch.equal(result, expected)
|
||||
assert result.dtype == cu_num_tokens.dtype
|
||||
assert result.device == cu_num_tokens.device
|
||||
|
||||
|
||||
def test_recover_request_lengths_empty():
|
||||
"""Test recover_request_lengths with empty input preserves device/dtype."""
|
||||
for device in ["cpu"]:
|
||||
cu = torch.tensor([], dtype=torch.int32, device=device)
|
||||
result = recover_request_lengths(cu)
|
||||
assert result.numel() == 0
|
||||
assert result.device == cu.device
|
||||
assert result.dtype == cu.dtype
|
||||
|
||||
|
||||
# # =============================================================================
|
||||
# # test bind_hashk_cache
|
||||
# # =============================================================================
|
||||
|
||||
|
||||
@patch("vllm_ascend.worker.kvcomp_utils.extract_layer_index")
|
||||
def test_bind_hashk_cache_basic(mock_extract):
|
||||
"""Test bind_hashk_cache populates runner and forward_context."""
|
||||
mock_extract.side_effect = lambda name, _: 0 if "layers.0" in name else (1 if "layers.1" in name else 2)
|
||||
|
||||
cache0 = torch.zeros(2, 8, 128, 16, dtype=torch.uint8)
|
||||
cache1 = torch.ones(2, 8, 128, 16, dtype=torch.uint8)
|
||||
hashk_caches = {"model.layers.0.self_attn": cache0, "model.layers.1.self_attn": cache1}
|
||||
|
||||
attn0 = MagicMock()
|
||||
attn1 = MagicMock()
|
||||
forward_context = {
|
||||
"model.layers.0.self_attn": attn0,
|
||||
"model.layers.1.self_attn": attn1,
|
||||
}
|
||||
|
||||
runner_hashk_caches: list[torch.Tensor] = []
|
||||
|
||||
bind_hashk_cache(hashk_caches, forward_context, runner_hashk_caches, num_attn_module=1)
|
||||
|
||||
assert len(runner_hashk_caches) == 2
|
||||
assert runner_hashk_caches[0] is cache0
|
||||
assert runner_hashk_caches[1] is cache1
|
||||
assert attn0.hashk_cache == [cache0]
|
||||
assert attn1.hashk_cache == [cache1]
|
||||
612
tests/ut/worker/a2/test_model_runner_v1.py
Normal file
612
tests/ut/worker/a2/test_model_runner_v1.py
Normal file
@@ -0,0 +1,612 @@
|
||||
import unittest
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
from vllm.model_executor.layers.attention import MLAAttention
|
||||
from vllm.v1.kv_cache_interface import FullAttentionSpec, KVCacheConfig, KVCacheGroupSpec, KVCacheTensor
|
||||
|
||||
from vllm_ascend.core.kv_cache_interface import AscendMLAAttentionSpec
|
||||
from vllm_ascend.worker.model_runner_v1 import NPUModelRunner
|
||||
|
||||
|
||||
class TestNPUModelRunnerAcceptedTokens(unittest.TestCase):
|
||||
@patch("vllm_ascend.worker.model_runner_v1.mamba_utils.postprocess_mamba_align_gpu")
|
||||
def test_postprocess_writes_accepted_counts_to_independent_snapshot(self, mock_postprocess):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
runner.use_async_scheduling = True
|
||||
runner.speculative_config = object()
|
||||
runner.model_config = SimpleNamespace(is_hybrid=True)
|
||||
runner.cache_config = SimpleNamespace(mamba_cache_mode="align")
|
||||
runner.num_accepted_tokens = SimpleNamespace(
|
||||
cpu=torch.zeros(2, dtype=torch.int32),
|
||||
gpu=torch.zeros(2, dtype=torch.int32),
|
||||
)
|
||||
persistent_counts = torch.ones(2, dtype=torch.int32)
|
||||
runner.input_batch = SimpleNamespace(num_accepted_tokens_cpu_tensor=persistent_counts)
|
||||
runner.kv_cache_config = object()
|
||||
runner.compilation_config = SimpleNamespace(static_forward_context={})
|
||||
runner.model = SimpleNamespace(get_mamba_state_copy_func=lambda: ())
|
||||
runner.num_accepted_tokens_event = MagicMock()
|
||||
runner._get_mamba_bufs = MagicMock()
|
||||
|
||||
runner._update_states_after_model_execute(
|
||||
torch.tensor([[10, 11, -1], [20, -1, -1]]),
|
||||
MagicMock(),
|
||||
)
|
||||
|
||||
self.assertIs(
|
||||
mock_postprocess.call_args.kwargs["num_accepted_tokens_cpu_tensor"],
|
||||
runner.num_accepted_tokens.cpu,
|
||||
)
|
||||
self.assertIsNot(
|
||||
mock_postprocess.call_args.kwargs["num_accepted_tokens_cpu_tensor"],
|
||||
persistent_counts,
|
||||
)
|
||||
runner.num_accepted_tokens_event.record.assert_called_once_with()
|
||||
|
||||
@patch(
|
||||
"vllm_ascend.worker.model_runner_v1.GPUModelRunner._update_states_after_model_execute",
|
||||
autospec=True,
|
||||
)
|
||||
def test_non_async_postprocess_delegates_to_upstream(self, mock_postprocess):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
runner.use_async_scheduling = False
|
||||
output_token_ids = torch.tensor([[10, -1]])
|
||||
scheduler_output = MagicMock()
|
||||
|
||||
runner._update_states_after_model_execute(output_token_ids, scheduler_output)
|
||||
|
||||
mock_postprocess.assert_called_once_with(runner, output_token_ids, scheduler_output)
|
||||
|
||||
def test_remap_uses_snapshot_after_persistent_row_is_overwritten(self):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
previous_counts = np.ones(16, dtype=np.int32)
|
||||
previous_counts[4] = 3
|
||||
previous_counts[11] = 4
|
||||
persistent_counts = np.ones(16, dtype=np.int32)
|
||||
|
||||
runner.num_accepted_tokens = SimpleNamespace(np=previous_counts)
|
||||
runner.prev_positions = SimpleNamespace(np=np.array([11, -1, 4] + [-1] * 13, dtype=np.int64))
|
||||
runner.input_batch = SimpleNamespace(num_accepted_tokens_cpu=persistent_counts)
|
||||
runner.use_async_scheduling = True
|
||||
|
||||
runner._sync_num_accepted_tokens(num_reqs=3, has_prev_mapping=True)
|
||||
|
||||
np.testing.assert_array_equal(previous_counts[:3], [4, 1, 3])
|
||||
np.testing.assert_array_equal(persistent_counts[:3], [4, 1, 3])
|
||||
|
||||
def test_async_without_previous_mapping_initializes_current_rows(self):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
snapshot = np.array([0, 4, 3, 9], dtype=np.int32)
|
||||
persistent_counts = np.array([7, 8, 6, 5], dtype=np.int32)
|
||||
runner.num_accepted_tokens = SimpleNamespace(np=snapshot)
|
||||
runner.input_batch = SimpleNamespace(num_accepted_tokens_cpu=persistent_counts)
|
||||
runner.use_async_scheduling = True
|
||||
|
||||
runner._sync_num_accepted_tokens(num_reqs=3, has_prev_mapping=False)
|
||||
|
||||
np.testing.assert_array_equal(snapshot, [1, 1, 1, 9])
|
||||
np.testing.assert_array_equal(persistent_counts, [1, 1, 1, 5])
|
||||
|
||||
def test_non_async_sync_uses_condensed_input_batch_rows(self):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
snapshot = np.array([4, 2, 9], dtype=np.int32)
|
||||
persistent_counts = np.array([2, 1, 7], dtype=np.int32)
|
||||
runner.num_accepted_tokens = SimpleNamespace(np=snapshot)
|
||||
runner.input_batch = SimpleNamespace(num_accepted_tokens_cpu=persistent_counts)
|
||||
runner.use_async_scheduling = False
|
||||
|
||||
runner._sync_num_accepted_tokens(num_reqs=2, has_prev_mapping=False)
|
||||
|
||||
np.testing.assert_array_equal(snapshot, [2, 1, 9])
|
||||
np.testing.assert_array_equal(persistent_counts, [2, 1, 7])
|
||||
|
||||
|
||||
class TestNPUModelRunnerKVCache(unittest.TestCase):
|
||||
def _build_runner(self):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
runner.device = torch.device("cpu")
|
||||
runner.use_sparse = False
|
||||
runner.use_sparse_c8 = False
|
||||
runner.use_compress = False
|
||||
runner.use_hybrid_blocks = False
|
||||
runner.hybrid_with_attn_and_mamba = False
|
||||
runner.sfa_dcp_replicated_indexer_size = 1
|
||||
runner.runner_only_attn_layers = set()
|
||||
runner.is_kv_consumer = False
|
||||
runner.vllm_config = MagicMock()
|
||||
runner.vllm_config.kv_transfer_config = None
|
||||
runner.model_config = MagicMock()
|
||||
runner.model_config.use_mla = True
|
||||
backend = MagicMock()
|
||||
backend.get_kv_cache_shape.side_effect = lambda num_blocks, block_size, num_kv_heads, head_size: (
|
||||
2,
|
||||
num_blocks,
|
||||
block_size,
|
||||
num_kv_heads,
|
||||
head_size,
|
||||
)
|
||||
runner.attn_backend = backend
|
||||
return runner
|
||||
|
||||
def test_allocate_kv_cache_uses_layer_spec_for_draft_gqa(self):
|
||||
runner = self._build_runner()
|
||||
kv_cache_spec = FullAttentionSpec(
|
||||
block_size=16,
|
||||
num_kv_heads=8,
|
||||
head_size=64,
|
||||
head_size_v=64,
|
||||
dtype=torch.float16,
|
||||
)
|
||||
kv_cache_config = KVCacheConfig(
|
||||
num_blocks=2,
|
||||
kv_cache_tensors=[KVCacheTensor(size=kv_cache_spec.page_size_bytes * 2, shared_by=["draft_attn"])],
|
||||
kv_cache_groups=[KVCacheGroupSpec(layer_names=["draft_attn"], kv_cache_spec=kv_cache_spec)],
|
||||
)
|
||||
|
||||
kv_cache_raw_tensors = runner._allocate_kv_cache_tensors(kv_cache_config)
|
||||
k_cache_raw, v_cache_raw = kv_cache_raw_tensors["draft_attn"]
|
||||
|
||||
self.assertEqual(k_cache_raw.numel(), kv_cache_spec.page_size_bytes)
|
||||
self.assertEqual(v_cache_raw.numel(), kv_cache_spec.page_size_bytes)
|
||||
|
||||
def test_reshape_kv_cache_uses_layer_spec_for_draft_gqa(self):
|
||||
runner = self._build_runner()
|
||||
kv_cache_spec = FullAttentionSpec(
|
||||
block_size=16,
|
||||
num_kv_heads=8,
|
||||
head_size=64,
|
||||
head_size_v=64,
|
||||
dtype=torch.float16,
|
||||
)
|
||||
kv_cache_config = KVCacheConfig(
|
||||
num_blocks=2,
|
||||
kv_cache_tensors=[KVCacheTensor(size=kv_cache_spec.page_size_bytes * 2, shared_by=["draft_attn"])],
|
||||
kv_cache_groups=[KVCacheGroupSpec(layer_names=["draft_attn"], kv_cache_spec=kv_cache_spec)],
|
||||
)
|
||||
kv_cache_raw_tensors = runner._allocate_kv_cache_tensors(kv_cache_config)
|
||||
runner._kv_cache_spec_attn_group_iterator = lambda: [
|
||||
SimpleNamespace(
|
||||
kv_cache_spec=kv_cache_spec,
|
||||
backend=runner.attn_backend,
|
||||
layer_names=["draft_attn"],
|
||||
)
|
||||
]
|
||||
|
||||
kv_caches = runner._reshape_kv_cache_tensors(kv_cache_config, kv_cache_raw_tensors)
|
||||
k_cache, v_cache = kv_caches["draft_attn"]
|
||||
|
||||
self.assertEqual(k_cache.shape, (2, 16, 8, 64))
|
||||
self.assertEqual(v_cache.shape, (2, 16, 8, 64))
|
||||
|
||||
@patch("vllm_ascend.worker.model_runner_v1.has_ec_transfer", return_value=False)
|
||||
@patch("vllm_ascend.worker.model_runner_v1.get_layers_from_vllm_config")
|
||||
def test_sparse_layer_without_indexer_allocates_only_mla_kv_cache(
|
||||
self,
|
||||
mock_get_layers,
|
||||
_mock_has_ec_transfer,
|
||||
):
|
||||
runner = self._build_runner()
|
||||
runner.use_sparse = True
|
||||
runner.block_size = 16
|
||||
runner.sparse_head_dim = (512, 64, 128)
|
||||
runner.kv_cache_dtype = torch.bfloat16
|
||||
runner.shared_kv_cache_layers = {}
|
||||
runner.ascend_config = MagicMock()
|
||||
runner.model_config.hf_text_config = SimpleNamespace(
|
||||
kv_lora_rank=512,
|
||||
qk_rope_head_dim=64,
|
||||
)
|
||||
runner.vllm_config.cache_config.cache_dtype = "auto"
|
||||
|
||||
attn_module = MLAAttention.__new__(MLAAttention)
|
||||
torch.nn.Module.__init__(attn_module)
|
||||
attn_module.impl = SimpleNamespace(has_indexer=False, use_sparse_c8=False)
|
||||
layer_name = "model.layers.1.self_attn.attn"
|
||||
mock_get_layers.return_value = {layer_name: attn_module}
|
||||
|
||||
spec = runner.get_kv_cache_spec()[layer_name]
|
||||
self.assertEqual(spec.sparse_head_dim, (512, 64, 0))
|
||||
|
||||
kv_cache_config = KVCacheConfig(
|
||||
num_blocks=2,
|
||||
kv_cache_tensors=[
|
||||
KVCacheTensor(
|
||||
size=spec.page_size_bytes * 2,
|
||||
shared_by=[layer_name],
|
||||
)
|
||||
],
|
||||
kv_cache_groups=[
|
||||
KVCacheGroupSpec(
|
||||
layer_names=[layer_name],
|
||||
kv_cache_spec=spec,
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
raw_caches = runner._allocate_kv_cache_tensors(kv_cache_config)
|
||||
raw_k_cache, raw_v_cache = raw_caches[layer_name]
|
||||
|
||||
self.assertEqual(raw_k_cache.numel(), 2 * 16 * 512 * 2)
|
||||
self.assertEqual(raw_v_cache.numel(), 2 * 16 * 64 * 2)
|
||||
|
||||
def test_sparse_c8_replicated_indexer_allocation_matches_page_size(self):
|
||||
runner = self._build_runner()
|
||||
runner.use_sparse = True
|
||||
runner.use_sparse_c8 = True
|
||||
runner.c8_k_cache_dtype = torch.int8
|
||||
runner.c8_k_scale_cache_dtype = torch.float16
|
||||
runner.model_config.hf_text_config = SimpleNamespace(index_head_dim=128)
|
||||
|
||||
layer_name = "model.layers.0.self_attn.attn"
|
||||
num_blocks = 2
|
||||
dcp_size = 2
|
||||
spec = AscendMLAAttentionSpec(
|
||||
block_size=16,
|
||||
num_kv_heads=1,
|
||||
head_size=704,
|
||||
sparse_head_dim=(576, 0, 128),
|
||||
dtype=torch.bfloat16,
|
||||
cache_dtype_str="auto",
|
||||
cache_sparse_sfa_c8=True,
|
||||
cache_sparse_li_c8=True,
|
||||
c8_k_cache_dtype=torch.int8,
|
||||
c8_k_scale_cache_dtype=torch.float16,
|
||||
sfa_dcp_replicated_indexer_size=dcp_size,
|
||||
)
|
||||
kv_cache_config = KVCacheConfig(
|
||||
num_blocks=num_blocks,
|
||||
kv_cache_tensors=[
|
||||
KVCacheTensor(
|
||||
size=spec.page_size_bytes * num_blocks,
|
||||
shared_by=[layer_name],
|
||||
)
|
||||
],
|
||||
kv_cache_groups=[
|
||||
KVCacheGroupSpec(
|
||||
layer_names=[layer_name],
|
||||
kv_cache_spec=spec,
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
raw_caches = runner._allocate_kv_cache_tensors(kv_cache_config)
|
||||
raw_k_cache, raw_indexer_cache, raw_indexer_scale_cache = raw_caches[layer_name]
|
||||
|
||||
self.assertEqual(raw_k_cache.numel(), num_blocks * 16 * 576)
|
||||
self.assertEqual(raw_indexer_cache.numel(), num_blocks * dcp_size * 16 * 128)
|
||||
self.assertEqual(raw_indexer_scale_cache.numel(), num_blocks * dcp_size * 16 * 2)
|
||||
self.assertEqual(
|
||||
raw_k_cache.numel() + raw_indexer_cache.numel() + raw_indexer_scale_cache.numel(),
|
||||
spec.page_size_bytes * num_blocks,
|
||||
)
|
||||
|
||||
def test_sparse_replicated_indexer_page_size_uses_expanded_storage_once(self):
|
||||
block_size = 16
|
||||
k_head_dim = 512
|
||||
v_head_dim = 64
|
||||
index_head_dim = 128
|
||||
dcp_size = 4
|
||||
expected_head_size = k_head_dim + v_head_dim + index_head_dim * dcp_size
|
||||
|
||||
spec = AscendMLAAttentionSpec(
|
||||
block_size=block_size,
|
||||
num_kv_heads=1,
|
||||
head_size=k_head_dim + v_head_dim + index_head_dim,
|
||||
sparse_head_dim=(k_head_dim, v_head_dim, index_head_dim),
|
||||
dtype=torch.bfloat16,
|
||||
cache_dtype_str="auto",
|
||||
sfa_dcp_replicated_indexer_size=dcp_size,
|
||||
)
|
||||
|
||||
self.assertEqual(spec.page_size_bytes, block_size * expected_head_size * 2)
|
||||
self.assertEqual(
|
||||
spec.sparse_kv_cache_ratio,
|
||||
(
|
||||
expected_head_size / k_head_dim,
|
||||
expected_head_size / v_head_dim,
|
||||
expected_head_size / (index_head_dim * dcp_size),
|
||||
None,
|
||||
),
|
||||
)
|
||||
|
||||
def test_sparse_replicated_indexer_only_expands_indexer_cache(self):
|
||||
runner = self._build_runner()
|
||||
runner.use_sparse = True
|
||||
runner.model_config.hf_text_config = SimpleNamespace(index_head_dim=128)
|
||||
runner._get_attention_kv_cache_dims = lambda _layer_name, _spec: (512, 64)
|
||||
runner.attn_backend.get_kv_cache_shape.side_effect = lambda num_blocks, block_size, num_kv_heads, head_size: (
|
||||
num_blocks,
|
||||
block_size,
|
||||
num_kv_heads,
|
||||
head_size,
|
||||
)
|
||||
|
||||
layer_name = "model.layers.0.self_attn.attn"
|
||||
num_blocks = 2
|
||||
dcp_size = 4
|
||||
spec = AscendMLAAttentionSpec(
|
||||
block_size=16,
|
||||
num_kv_heads=1,
|
||||
head_size=704,
|
||||
sparse_head_dim=(512, 64, 128),
|
||||
dtype=torch.bfloat16,
|
||||
cache_dtype_str="auto",
|
||||
sfa_dcp_replicated_indexer_size=dcp_size,
|
||||
)
|
||||
kv_cache_config = KVCacheConfig(
|
||||
num_blocks=num_blocks,
|
||||
kv_cache_tensors=[
|
||||
KVCacheTensor(
|
||||
size=spec.page_size_bytes * num_blocks,
|
||||
shared_by=[layer_name],
|
||||
)
|
||||
],
|
||||
kv_cache_groups=[
|
||||
KVCacheGroupSpec(
|
||||
layer_names=[layer_name],
|
||||
kv_cache_spec=spec,
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
raw_caches = runner._allocate_kv_cache_tensors(kv_cache_config)
|
||||
raw_k_cache, raw_v_cache, raw_indexer_cache = raw_caches[layer_name]
|
||||
|
||||
self.assertEqual(raw_k_cache.numel(), num_blocks * 16 * 512 * 2)
|
||||
self.assertEqual(raw_v_cache.numel(), num_blocks * 16 * 64 * 2)
|
||||
self.assertEqual(raw_indexer_cache.numel(), num_blocks * dcp_size * 16 * 128 * 2)
|
||||
self.assertEqual(
|
||||
raw_k_cache.numel() + raw_v_cache.numel() + raw_indexer_cache.numel(),
|
||||
spec.page_size_bytes * num_blocks,
|
||||
)
|
||||
|
||||
runner._kv_cache_spec_attn_group_iterator = lambda: [
|
||||
SimpleNamespace(
|
||||
kv_cache_spec=spec,
|
||||
backend=runner.attn_backend,
|
||||
layer_names=[layer_name],
|
||||
)
|
||||
]
|
||||
k_cache, v_cache, indexer_cache = runner._reshape_kv_cache_tensors(
|
||||
kv_cache_config,
|
||||
raw_caches,
|
||||
)[layer_name]
|
||||
|
||||
self.assertEqual(k_cache.shape, (num_blocks, 16, 1, 512))
|
||||
self.assertEqual(v_cache.shape, (num_blocks, 16, 1, 64))
|
||||
self.assertEqual(indexer_cache.shape, (num_blocks * dcp_size, 16, 1, 128))
|
||||
|
||||
|
||||
class TestNPUModelRunnerOutputTokenIds(unittest.TestCase):
|
||||
def _build_runner(self):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
runner.device = torch.device("cpu")
|
||||
runner.vllm_config = MagicMock()
|
||||
runner.model_config = MagicMock()
|
||||
runner.use_compress = False
|
||||
return runner
|
||||
|
||||
@patch("vllm_ascend.worker.model_runner_v1.get_ascend_config")
|
||||
@patch("vllm_ascend.worker.model_runner_v1.lmhead_tp_enable")
|
||||
def test_sample_updates_output_token_ids_before_sampler(self, mock_lmhead_tp_enable, mock_get_ascend_config):
|
||||
"""Verify output_token_ids are updated before sampler is called"""
|
||||
mock_lmhead_tp_enable.return_value = False
|
||||
mock_ascend_config = MagicMock()
|
||||
mock_ascend_config.enable_reduce_sample = False
|
||||
mock_get_ascend_config.return_value = mock_ascend_config
|
||||
|
||||
# Build input batch with historical sampled tokens
|
||||
input_batch = MagicMock()
|
||||
input_batch.sampling_metadata.output_token_ids = [
|
||||
[1, 2, 3, -1],
|
||||
[4, 5, -1],
|
||||
]
|
||||
input_batch.sampling_metadata.top_k = None
|
||||
input_batch.num_reqs = 2
|
||||
input_batch.top_k_cpu = None
|
||||
input_batch.prev_req_id_to_index = {
|
||||
"req0": 0,
|
||||
"req1": 1,
|
||||
}
|
||||
input_batch.sampled_token_ids_cpu = torch.tensor([6, 7])
|
||||
input_batch.async_copy_ready_event = MagicMock()
|
||||
input_batch.async_copy_ready_event.synchronize = MagicMock()
|
||||
|
||||
# Simulate the real behavior of InputBatch.update_async_output_token_ids
|
||||
def mock_update_output_token_ids():
|
||||
output_token_ids = input_batch.sampling_metadata.output_token_ids
|
||||
sampled_ids = input_batch.sampled_token_ids_cpu.tolist()
|
||||
|
||||
for index, req_id in enumerate(input_batch.prev_req_id_to_index):
|
||||
prev_index = input_batch.prev_req_id_to_index[req_id]
|
||||
req_output = output_token_ids[index]
|
||||
if req_output and req_output[-1] == -1:
|
||||
req_output[-1] = sampled_ids[prev_index]
|
||||
|
||||
input_batch.update_async_output_token_ids.side_effect = mock_update_output_token_ids
|
||||
|
||||
# Build runner and inject dependencies
|
||||
runner = self._build_runner()
|
||||
runner.input_batch = input_batch
|
||||
runner.sampler = MagicMock(return_value=MagicMock())
|
||||
|
||||
# Call sample method
|
||||
logits = torch.randn(2, 32000)
|
||||
runner._sample(logits=logits, spec_decode_metadata=None)
|
||||
|
||||
# Verify sampler and update_async_output_token_ids were called
|
||||
runner.sampler.assert_called_once()
|
||||
input_batch.update_async_output_token_ids.assert_called_once()
|
||||
|
||||
# Verify output_token_ids were updated before sampler is called
|
||||
call_kwargs = runner.sampler.call_args[1]
|
||||
actual_sampling_metadata = call_kwargs["sampling_metadata"]
|
||||
actual_output_token_ids = actual_sampling_metadata.output_token_ids
|
||||
self.assertEqual(actual_output_token_ids[0], [1, 2, 3, 6])
|
||||
self.assertEqual(actual_output_token_ids[1], [4, 5, 7])
|
||||
|
||||
def test_placeholder_spec_tokens_are_sanitized_only_for_forward(self):
|
||||
runner = self._build_runner()
|
||||
runner.input_ids = SimpleNamespace(
|
||||
cpu=torch.tensor([11, -1, 33, -1], dtype=torch.int32),
|
||||
gpu=torch.tensor([11, -1, 33, -1], dtype=torch.int32),
|
||||
)
|
||||
scheduler_output = SimpleNamespace(
|
||||
scheduled_spec_decode_tokens={"req0": [-1]},
|
||||
)
|
||||
|
||||
runner._sanitize_placeholder_input_ids_for_forward(
|
||||
scheduler_output,
|
||||
num_forward_tokens=4,
|
||||
)
|
||||
|
||||
self.assertEqual(runner.input_ids.gpu.tolist(), [11, 0, 33, 0])
|
||||
self.assertEqual(runner.input_ids.cpu.tolist(), [11, -1, 33, -1])
|
||||
|
||||
def test_placeholder_sanitization_is_scoped_to_current_forward(self):
|
||||
runner = self._build_runner()
|
||||
runner.input_ids = SimpleNamespace(
|
||||
cpu=torch.tensor([11, -1, 33, -1], dtype=torch.int32),
|
||||
gpu=torch.tensor([11, -1, 33, -1], dtype=torch.int32),
|
||||
)
|
||||
scheduler_output = SimpleNamespace(
|
||||
scheduled_spec_decode_tokens={"req0": [-1]},
|
||||
)
|
||||
|
||||
runner._sanitize_placeholder_input_ids_for_forward(
|
||||
scheduler_output,
|
||||
num_forward_tokens=2,
|
||||
)
|
||||
|
||||
self.assertEqual(runner.input_ids.gpu.tolist(), [11, 0, 33, -1])
|
||||
|
||||
def test_mtp3_placeholder_metadata_is_preserved_before_sanitizing_forward(self):
|
||||
runner = self._build_runner()
|
||||
runner.pcp_size = 1
|
||||
runner.arange_np = np.arange(8, dtype=np.int32)
|
||||
runner._arange_scratch = np.empty(8, dtype=np.int32)
|
||||
runner.input_ids = SimpleNamespace(
|
||||
cpu=torch.tensor([11, -1, -1, -1], dtype=torch.int32),
|
||||
gpu=torch.tensor([11, -1, -1, -1], dtype=torch.int32),
|
||||
)
|
||||
scheduler_output = SimpleNamespace(
|
||||
scheduled_spec_decode_tokens={"req0": [-1, -1, -1]},
|
||||
)
|
||||
|
||||
spec_decode_metadata = runner._calc_spec_decode_metadata(
|
||||
num_draft_tokens=np.array([3], dtype=np.int32),
|
||||
cu_num_scheduled_tokens=np.array([4], dtype=np.int32),
|
||||
num_pcp_pads=None,
|
||||
)
|
||||
runner._sanitize_placeholder_input_ids_for_forward(
|
||||
scheduler_output,
|
||||
num_forward_tokens=4,
|
||||
)
|
||||
|
||||
self.assertEqual(spec_decode_metadata.draft_token_ids.tolist(), [-1, -1, -1])
|
||||
self.assertEqual(runner.input_ids.gpu.tolist(), [11, 0, 0, 0])
|
||||
self.assertEqual(runner.input_ids.cpu.tolist(), [11, -1, -1, -1])
|
||||
|
||||
|
||||
class TestNPUModelRunnerDebugger(unittest.TestCase):
|
||||
def _build_runner(self, debugger=None):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
runner.debugger = debugger or MagicMock()
|
||||
runner.model = MagicMock()
|
||||
runner.model_config = MagicMock()
|
||||
runner.model_config.enforce_eager = False
|
||||
runner._debugger_started = True
|
||||
runner._debugger_step_dummy_data_before_execute = False
|
||||
runner.use_compress = False
|
||||
return runner
|
||||
|
||||
def test_finalize_dump_data_stops_stop_capable_debugger(self):
|
||||
runner = self._build_runner()
|
||||
|
||||
runner._finalize_dump_data()
|
||||
|
||||
runner.debugger.stop.assert_called_once_with()
|
||||
runner.debugger.step.assert_called_once_with()
|
||||
self.assertFalse(runner._debugger_started)
|
||||
|
||||
def test_finalize_dump_data_steps_graph_debugger_without_stop(self):
|
||||
debugger = MagicMock(spec=["start", "step"])
|
||||
runner = self._build_runner(debugger)
|
||||
|
||||
runner._finalize_dump_data()
|
||||
|
||||
debugger.step.assert_called_once_with()
|
||||
self.assertTrue(runner._debugger_started)
|
||||
|
||||
def test_start_dump_data_noop_when_already_started(self):
|
||||
runner = self._build_runner(MagicMock(spec=["start", "step"]))
|
||||
|
||||
runner._start_dump_data()
|
||||
|
||||
runner.debugger.start.assert_not_called()
|
||||
runner.debugger.step.assert_not_called()
|
||||
self.assertTrue(runner._debugger_started)
|
||||
|
||||
|
||||
class TestCorrectOptimisticSeqLensCpu(unittest.TestCase):
|
||||
"""Regression tests for async spec-decode seq_lens correction.
|
||||
|
||||
The helper must synchronize the device->host copy event *before* reading
|
||||
``valid_sampled_token_count_cpu``. Reading it early consumes stale counts
|
||||
and corrupts the CPU seq_lens, which surfaced as an accuracy regression on
|
||||
DeepSeek-V4 (its compressed-KV slot mapping is built from these seq_lens).
|
||||
"""
|
||||
|
||||
def _build_runner(self, optimistic, prev_positions, prev_drafts, counts_cpu):
|
||||
runner = NPUModelRunner.__new__(NPUModelRunner)
|
||||
runner.optimistic_seq_lens_cpu = optimistic
|
||||
runner.prev_positions = SimpleNamespace(np=prev_positions)
|
||||
runner.prev_num_draft_tokens = SimpleNamespace(np=prev_drafts)
|
||||
runner.valid_sampled_token_count_cpu = counts_cpu
|
||||
return runner
|
||||
|
||||
def test_synchronizes_before_host_read(self):
|
||||
num_reqs = 3
|
||||
# Optimistic (all drafts assumed accepted):
|
||||
# prev_computed=[100,200,50], prev_drafts=[2,3,1], sched=[3,4,2]
|
||||
# optimistic = prev_computed + (prev_drafts + 1) + sched
|
||||
optimistic = torch.tensor([106, 208, 54], dtype=torch.int64)
|
||||
prev_positions = np.array([0, 1, 2], dtype=np.int64)
|
||||
prev_drafts = np.array([2, 3, 1], dtype=np.int32)
|
||||
|
||||
# CPU buffer initially holds STALE counts (== drafts + 1, i.e. "all
|
||||
# accepted"). If the helper reads before synchronizing, the correction
|
||||
# is a no-op and the assertion below fails.
|
||||
counts_cpu = torch.tensor([3, 4, 2], dtype=torch.int32)
|
||||
# The true counts that the async copy delivers on synchronize().
|
||||
true_counts = np.array([2, 1, 2], dtype=np.int32)
|
||||
|
||||
runner = self._build_runner(optimistic, prev_positions, prev_drafts, counts_cpu)
|
||||
event = MagicMock()
|
||||
event.synchronize.side_effect = lambda: counts_cpu.copy_(torch.from_numpy(true_counts))
|
||||
runner.valid_sampled_token_count_event = event
|
||||
|
||||
runner._correct_optimistic_seq_lens_cpu(num_reqs)
|
||||
|
||||
event.synchronize.assert_called_once()
|
||||
# correction = (prev_drafts + 1 - true_counts) = [1, 3, 0]
|
||||
# corrected = optimistic - correction = [105, 205, 54]
|
||||
np.testing.assert_array_equal(optimistic.numpy(), np.array([105, 205, 54]))
|
||||
|
||||
def test_asserts_event_present(self):
|
||||
runner = self._build_runner(
|
||||
torch.tensor([10], dtype=torch.int64),
|
||||
np.array([0], dtype=np.int64),
|
||||
np.array([1], dtype=np.int32),
|
||||
torch.tensor([1], dtype=torch.int32),
|
||||
)
|
||||
runner.valid_sampled_token_count_event = None
|
||||
with self.assertRaises(AssertionError):
|
||||
runner._correct_optimistic_seq_lens_cpu(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
393
tests/ut/worker/a2/test_model_runner_v1_with_device.py
Normal file
393
tests/ut/worker/a2/test_model_runner_v1_with_device.py
Normal file
@@ -0,0 +1,393 @@
|
||||
import os
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
from vllm.config import (
|
||||
CacheConfig,
|
||||
CUDAGraphMode,
|
||||
ModelConfig,
|
||||
ParallelConfig,
|
||||
SchedulerConfig,
|
||||
VllmConfig,
|
||||
set_current_vllm_config,
|
||||
)
|
||||
from vllm.distributed.parallel_state import GroupCoordinator
|
||||
from vllm.model_executor.layers.attention import Attention
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.v1.kv_cache_interface import (
|
||||
FullAttentionSpec,
|
||||
KVCacheConfig,
|
||||
KVCacheGroupSpec,
|
||||
KVCacheTensor,
|
||||
)
|
||||
|
||||
import vllm_ascend.compilation.acl_graph as acl_graph
|
||||
from vllm_ascend.worker.model_runner_v1 import NPUModelRunner
|
||||
from vllm_ascend.worker.npu_input_batch import NPUInputBatch
|
||||
|
||||
BLOCK_SIZE = 128
|
||||
NUM_BLOCKS = 10
|
||||
DEVICE_TYPE = current_platform.device_type
|
||||
FAKE_WEIGHT_PATH = os.path.join(os.path.dirname(__file__), "..", "..", "_fake_weight")
|
||||
|
||||
|
||||
def initialize_kv_cache(runner: NPUModelRunner):
|
||||
"""
|
||||
Only perform necessary steps in NPUModelRunner.initialize_kv_cache()
|
||||
"""
|
||||
attn_spec = FullAttentionSpec(
|
||||
block_size=BLOCK_SIZE,
|
||||
num_kv_heads=runner.model_config.get_num_kv_heads(runner.parallel_config),
|
||||
head_size=runner.model_config.get_head_size(),
|
||||
dtype=runner.kv_cache_dtype,
|
||||
)
|
||||
tensor_size = attn_spec.page_size_bytes * NUM_BLOCKS
|
||||
kv_cache_config = KVCacheConfig(
|
||||
num_blocks=NUM_BLOCKS,
|
||||
kv_cache_tensors=[
|
||||
KVCacheTensor(size=tensor_size, shared_by=["layer.0"]),
|
||||
],
|
||||
kv_cache_groups=[KVCacheGroupSpec(layer_names=["layer.0"], kv_cache_spec=attn_spec)],
|
||||
)
|
||||
runner.kv_cache_config = kv_cache_config
|
||||
runner.input_batch = NPUInputBatch(
|
||||
max_num_reqs=runner.max_num_reqs,
|
||||
max_model_len=runner.max_model_len,
|
||||
max_num_batched_tokens=runner.max_num_tokens,
|
||||
device=runner.device,
|
||||
pin_memory=runner.pin_memory,
|
||||
vocab_size=runner.model_config.get_vocab_size(),
|
||||
block_sizes=[kv_cache_config.kv_cache_groups[0].kv_cache_spec.block_size],
|
||||
kernel_block_sizes=[[kv_cache_config.kv_cache_groups[0].kv_cache_spec.block_size]],
|
||||
)
|
||||
runner.initialize_attn_backend(kv_cache_config)
|
||||
|
||||
|
||||
def get_vllm_config():
|
||||
model_config = ModelConfig(
|
||||
model=FAKE_WEIGHT_PATH,
|
||||
dtype="float16",
|
||||
seed=42,
|
||||
skip_tokenizer_init=True,
|
||||
)
|
||||
scheduler_config = SchedulerConfig(
|
||||
max_num_seqs=10,
|
||||
max_num_batched_tokens=512,
|
||||
max_model_len=512,
|
||||
is_encoder_decoder=model_config.is_encoder_decoder,
|
||||
)
|
||||
cache_config = CacheConfig(
|
||||
block_size=BLOCK_SIZE,
|
||||
gpu_memory_utilization=0.9,
|
||||
cache_dtype="auto",
|
||||
)
|
||||
parallel_config = ParallelConfig()
|
||||
vllm_config = VllmConfig(
|
||||
model_config=model_config,
|
||||
cache_config=cache_config,
|
||||
scheduler_config=scheduler_config,
|
||||
parallel_config=parallel_config,
|
||||
)
|
||||
return vllm_config
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def model_runner():
|
||||
vllm_config = get_vllm_config()
|
||||
with (
|
||||
set_current_vllm_config(vllm_config),
|
||||
patch("vllm_ascend.worker.block_table.get_dcp_group") as mock_get_dcp_group,
|
||||
patch("vllm_ascend.worker.block_table.get_pcp_group") as mock_get_pcp_group,
|
||||
):
|
||||
mock_dcp_group = MagicMock(spec=GroupCoordinator)
|
||||
mock_dcp_group.world_size = 1
|
||||
mock_dcp_group.rank_in_group = 0
|
||||
mock_get_dcp_group.return_value = mock_dcp_group
|
||||
mock_pcp_group = MagicMock(spec=GroupCoordinator)
|
||||
mock_pcp_group.world_size = 1
|
||||
mock_pcp_group.rank_in_group = 0
|
||||
mock_get_pcp_group.return_value = mock_pcp_group
|
||||
|
||||
model_config = vllm_config.model_config
|
||||
num_heads = model_config.get_num_kv_heads(vllm_config.parallel_config)
|
||||
head_size = model_config.get_head_size()
|
||||
vllm_config.compilation_config.static_forward_context["layer.0"] = Attention(num_heads, head_size, 0.1)
|
||||
runner = NPUModelRunner(vllm_config, DEVICE_TYPE)
|
||||
initialize_kv_cache(runner)
|
||||
yield runner
|
||||
# Reset global state set by _check_and_update_cudagraph_mode
|
||||
# so the next test case can reinitialize cleanly.
|
||||
acl_graph._graph_params = None
|
||||
acl_graph._draft_graph_params = None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"num_computed_tokens, num_scheduled_tokens, num_tokens, num_reqs, "
|
||||
"max_num_scheduled_tokens, use_cascade_attn, force_eager, "
|
||||
"force_uniform_decode, spec_decode_tokens",
|
||||
[
|
||||
# ---- force_eager=True: bypass cudagraph dispatch ----
|
||||
pytest.param(
|
||||
[0, 0, 0],
|
||||
[10, 10, 10],
|
||||
30,
|
||||
3,
|
||||
10,
|
||||
False,
|
||||
True,
|
||||
None,
|
||||
0,
|
||||
id="prefill_eager",
|
||||
),
|
||||
pytest.param(
|
||||
[5, 10, 15],
|
||||
[1, 1, 1],
|
||||
3,
|
||||
3,
|
||||
1,
|
||||
False,
|
||||
True,
|
||||
None,
|
||||
0,
|
||||
id="decode_eager",
|
||||
),
|
||||
pytest.param(
|
||||
[0, 5, 10],
|
||||
[10, 1, 1],
|
||||
12,
|
||||
3,
|
||||
10,
|
||||
False,
|
||||
True,
|
||||
None,
|
||||
0,
|
||||
id="mixed_eager",
|
||||
),
|
||||
# ---- force_eager=False: go through real dispatch path ----
|
||||
pytest.param(
|
||||
[0, 0, 0],
|
||||
[10, 10, 10],
|
||||
30,
|
||||
3,
|
||||
10,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
0,
|
||||
id="prefill_dispatch",
|
||||
),
|
||||
pytest.param(
|
||||
[5, 10, 15],
|
||||
[1, 1, 1],
|
||||
3,
|
||||
3,
|
||||
1,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
0,
|
||||
id="decode_uniform_dispatch",
|
||||
),
|
||||
pytest.param(
|
||||
[0, 5, 10],
|
||||
[10, 1, 1],
|
||||
12,
|
||||
3,
|
||||
10,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
0,
|
||||
id="mixed_dispatch",
|
||||
),
|
||||
pytest.param(
|
||||
[0],
|
||||
[50],
|
||||
50,
|
||||
1,
|
||||
50,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
0,
|
||||
id="single_prefill_dispatch",
|
||||
),
|
||||
pytest.param(
|
||||
[100],
|
||||
[1],
|
||||
1,
|
||||
1,
|
||||
1,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
0,
|
||||
id="single_decode_dispatch",
|
||||
),
|
||||
# ---- cascade attention ----
|
||||
pytest.param(
|
||||
[0, 0, 0],
|
||||
[10, 10, 10],
|
||||
30,
|
||||
3,
|
||||
10,
|
||||
True,
|
||||
False,
|
||||
None,
|
||||
0,
|
||||
id="prefill_cascade_attn",
|
||||
),
|
||||
# ---- force_uniform_decode override ----
|
||||
pytest.param(
|
||||
[5, 10, 15],
|
||||
[1, 1, 1],
|
||||
3,
|
||||
3,
|
||||
1,
|
||||
False,
|
||||
False,
|
||||
True,
|
||||
0,
|
||||
id="decode_force_uniform_true",
|
||||
),
|
||||
pytest.param(
|
||||
[5, 10, 15],
|
||||
[1, 1, 1],
|
||||
3,
|
||||
3,
|
||||
1,
|
||||
False,
|
||||
False,
|
||||
False,
|
||||
0,
|
||||
id="decode_force_uniform_false",
|
||||
),
|
||||
# ---- spec_decode: uniform_decode depends on is_all_decode ----
|
||||
pytest.param(
|
||||
[5, 10, 15],
|
||||
[4, 4, 4],
|
||||
12,
|
||||
3,
|
||||
4,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
3,
|
||||
id="spec_decode_all_decode",
|
||||
),
|
||||
pytest.param(
|
||||
[0, 0, 0],
|
||||
[4, 4, 4],
|
||||
12,
|
||||
3,
|
||||
4,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
3,
|
||||
id="spec_decode_all_prefill",
|
||||
),
|
||||
pytest.param(
|
||||
[0, 5, 10],
|
||||
[4, 4, 4],
|
||||
12,
|
||||
3,
|
||||
4,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
3,
|
||||
id="spec_decode_mixed",
|
||||
),
|
||||
# ---- large batch ----
|
||||
pytest.param(
|
||||
[0, 0, 0, 0, 0],
|
||||
[20, 20, 20, 20, 20],
|
||||
100,
|
||||
5,
|
||||
20,
|
||||
False,
|
||||
False,
|
||||
None,
|
||||
0,
|
||||
id="large_prefill_dispatch",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_determine_batch_execution_and_padding(
|
||||
model_runner,
|
||||
num_computed_tokens,
|
||||
num_scheduled_tokens,
|
||||
num_tokens,
|
||||
num_reqs,
|
||||
max_num_scheduled_tokens,
|
||||
use_cascade_attn,
|
||||
force_eager,
|
||||
force_uniform_decode,
|
||||
spec_decode_tokens,
|
||||
):
|
||||
runner = model_runner
|
||||
|
||||
# Set up spec decode scenario by overriding runner attributes
|
||||
saved_spec_config = runner.speculative_config
|
||||
saved_query_len = runner.uniform_decode_query_len
|
||||
if spec_decode_tokens > 0:
|
||||
runner.speculative_config = type("FakeSpecConfig", (), {"num_speculative_tokens": spec_decode_tokens})()
|
||||
runner.uniform_decode_query_len = 1 + spec_decode_tokens
|
||||
else:
|
||||
runner.speculative_config = None
|
||||
runner.uniform_decode_query_len = 1
|
||||
|
||||
try:
|
||||
runner.input_batch.num_computed_tokens_cpu[:num_reqs] = num_computed_tokens
|
||||
num_scheduled_tokens_np = np.array(num_scheduled_tokens, dtype=np.int32)
|
||||
|
||||
kwargs = dict(
|
||||
num_tokens=num_tokens,
|
||||
num_reqs=num_reqs,
|
||||
num_scheduled_tokens_np=num_scheduled_tokens_np,
|
||||
max_num_scheduled_tokens=max_num_scheduled_tokens,
|
||||
use_cascade_attn=use_cascade_attn,
|
||||
force_eager=force_eager,
|
||||
)
|
||||
if force_uniform_decode is not None:
|
||||
kwargs["force_uniform_decode"] = force_uniform_decode
|
||||
|
||||
(
|
||||
cudagraph_mode,
|
||||
batch_desc,
|
||||
should_ubatch,
|
||||
num_tokens_across_dp,
|
||||
cudagraph_stats,
|
||||
) = runner._determine_batch_execution_and_padding(**kwargs)
|
||||
|
||||
# force_eager always bypasses cudagraph dispatch
|
||||
if force_eager:
|
||||
assert cudagraph_mode == CUDAGraphMode.NONE
|
||||
assert batch_desc.num_tokens == num_tokens
|
||||
else:
|
||||
# The resolved cudagraph_mode is determined during
|
||||
# initialize_attn_backend and stored in the dispatcher.
|
||||
resolved_mode = runner.cudagraph_dispatcher.cudagraph_mode
|
||||
if resolved_mode == CUDAGraphMode.NONE:
|
||||
assert cudagraph_mode == CUDAGraphMode.NONE
|
||||
assert batch_desc.num_tokens == num_tokens
|
||||
else:
|
||||
# Dispatcher may match a captured key (PIECEWISE/FULL)
|
||||
# or fall back to NONE if num_tokens exceeds max capture size.
|
||||
assert cudagraph_mode in (
|
||||
CUDAGraphMode.NONE,
|
||||
CUDAGraphMode.PIECEWISE,
|
||||
CUDAGraphMode.FULL,
|
||||
)
|
||||
# Padding can only increase, never shrink
|
||||
assert batch_desc.num_tokens >= num_tokens
|
||||
# dp_size=1: no micro-batching, no cross-dp coordination
|
||||
assert should_ubatch is False
|
||||
assert num_tokens_across_dp is None
|
||||
# cudagraph_metrics disabled by default
|
||||
assert cudagraph_stats is None
|
||||
finally:
|
||||
runner.speculative_config = saved_spec_config
|
||||
runner.uniform_decode_query_len = saved_query_len
|
||||
305
tests/ut/worker/a2/test_worker_multi_instance.py
Normal file
305
tests/ut/worker/a2/test_worker_multi_instance.py
Normal file
@@ -0,0 +1,305 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from vllm.utils.mem_constants import GiB_bytes
|
||||
|
||||
from tests.ut.base import TestBase
|
||||
|
||||
|
||||
class TestDetermineAvailableMemoryMultiInstance(TestBase):
|
||||
"""Tests for determine_available_memory() focusing on the multi-instance
|
||||
OOM regression (PR #7427)."""
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# Helpers
|
||||
# ------------------------------------------------------------------ #
|
||||
|
||||
def _make_worker(
|
||||
self,
|
||||
requested_memory: int,
|
||||
init_free_memory: int,
|
||||
init_total_memory: int,
|
||||
model_memory_usage: int | None = None,
|
||||
):
|
||||
"""Return a minimally-configured NPUWorker mock with memory state set."""
|
||||
from vllm_ascend.worker.worker import NPUWorker
|
||||
|
||||
if model_memory_usage is None:
|
||||
model_memory_usage = int(0.5 * GiB_bytes) # Qwen3-0.6B ~0.5 GiB
|
||||
|
||||
with patch.object(NPUWorker, "__init__", lambda x, **kwargs: None):
|
||||
worker = NPUWorker()
|
||||
|
||||
worker.model_runner = MagicMock()
|
||||
worker.model_runner.model_memory_usage = model_memory_usage
|
||||
|
||||
mock_cache_config = MagicMock()
|
||||
mock_cache_config.kv_cache_memory_bytes = None
|
||||
mock_cache_config.gpu_memory_utilization = requested_memory / init_total_memory
|
||||
worker.cache_config = mock_cache_config
|
||||
|
||||
worker.model_config = SimpleNamespace(hf_config=SimpleNamespace(model_type="qwen3"))
|
||||
|
||||
mock_snapshot = MagicMock()
|
||||
mock_snapshot.free_memory = init_free_memory
|
||||
mock_snapshot.total_memory = init_total_memory
|
||||
worker.init_snapshot = mock_snapshot
|
||||
|
||||
worker.requested_memory = requested_memory
|
||||
worker.device = "npu:0"
|
||||
return worker
|
||||
|
||||
@staticmethod
|
||||
def _make_profile_result(free_memory_after: int, non_kv_cache_memory: int):
|
||||
"""Return a mock profile_result compatible with memory_profiling output.
|
||||
|
||||
The worker code recomputes non_kv_cache_memory as:
|
||||
non_torch_increase + torch_peak_increase + weights_memory
|
||||
We set non_torch_increase=0, before_profile.torch_peak=0 (so
|
||||
torch_peak_increase = peak - 0 = 0 since memory_stats is mocked to
|
||||
return peak=0), and weights_memory=non_kv_cache_memory, ensuring the
|
||||
recomputed value equals the requested non_kv_cache_memory.
|
||||
"""
|
||||
profile_result = MagicMock()
|
||||
profile_result.after_profile.free_memory = free_memory_after
|
||||
profile_result.non_kv_cache_memory = non_kv_cache_memory
|
||||
profile_result.non_torch_increase = 0
|
||||
profile_result.before_profile.torch_peak = 0
|
||||
profile_result.weights_memory = non_kv_cache_memory
|
||||
return profile_result
|
||||
|
||||
@staticmethod
|
||||
def _patch_memory_profiling(profile_result):
|
||||
"""Return a context manager mocking `memory_profiling` and `torch.npu.memory_stats`."""
|
||||
from contextlib import contextmanager
|
||||
|
||||
mock_ctx = MagicMock()
|
||||
mock_ctx.__enter__ = MagicMock(return_value=profile_result)
|
||||
mock_ctx.__exit__ = MagicMock(return_value=False)
|
||||
mock_profiling = MagicMock(return_value=mock_ctx)
|
||||
|
||||
@contextmanager
|
||||
def combined():
|
||||
with (
|
||||
patch("vllm_ascend.worker.worker.memory_profiling", mock_profiling),
|
||||
patch(
|
||||
"torch.npu.memory_stats",
|
||||
return_value={"allocated_bytes.all.peak": 0},
|
||||
),
|
||||
):
|
||||
yield
|
||||
|
||||
return combined()
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# Tests
|
||||
# ------------------------------------------------------------------ #
|
||||
|
||||
@patch("vllm_ascend.worker.worker.logger")
|
||||
def test_single_instance_positive_kv_cache(self, mock_logger):
|
||||
"""Baseline: single instance on an empty card yields positive KV cache."""
|
||||
total = int(64 * GiB_bytes)
|
||||
gpu_util = 0.9
|
||||
requested_memory = int(total * gpu_util) # 57.6 GiB
|
||||
init_free = int(62 * GiB_bytes) # almost all free
|
||||
non_kv_cache = int(0.5 * GiB_bytes) # Qwen3-0.6B weights
|
||||
|
||||
worker = self._make_worker(requested_memory, init_free, total)
|
||||
profile_result = self._make_profile_result(
|
||||
free_memory_after=init_free - non_kv_cache,
|
||||
non_kv_cache_memory=non_kv_cache,
|
||||
)
|
||||
|
||||
with self._patch_memory_profiling(profile_result):
|
||||
result = worker.determine_available_memory()
|
||||
|
||||
expected = requested_memory - non_kv_cache
|
||||
self.assertEqual(result, expected)
|
||||
self.assertGreater(result, 0)
|
||||
|
||||
@patch("vllm_ascend.worker.worker.logger")
|
||||
def test_determine_available_memory_does_not_profile_npugraph_memory(self, mock_logger):
|
||||
total = int(64 * GiB_bytes)
|
||||
requested_memory = int(total * 0.9)
|
||||
init_free = int(60 * GiB_bytes)
|
||||
non_kv_cache = int(1 * GiB_bytes)
|
||||
|
||||
worker = self._make_worker(requested_memory, init_free, total)
|
||||
worker.model_runner.profile_cudagraph_memory = MagicMock()
|
||||
profile_result = self._make_profile_result(
|
||||
free_memory_after=init_free - non_kv_cache,
|
||||
non_kv_cache_memory=non_kv_cache,
|
||||
)
|
||||
|
||||
with self._patch_memory_profiling(profile_result):
|
||||
result = worker.determine_available_memory()
|
||||
|
||||
worker.model_runner.profile_run.assert_called_once()
|
||||
worker.model_runner.profile_cudagraph_memory.assert_not_called()
|
||||
self.assertFalse(hasattr(worker, "npugraph_memory_estimate"))
|
||||
self.assertEqual(result, requested_memory - non_kv_cache)
|
||||
|
||||
@patch("vllm_ascend.worker.worker.logger")
|
||||
def test_second_instance_on_same_card_positive_kv_cache(self, mock_logger):
|
||||
"""
|
||||
Regression test for PR #7427.
|
||||
|
||||
Scenario (64 GiB Ascend 910B card, two Qwen3-0.6B instances,
|
||||
gpu_memory_utilization=0.4):
|
||||
|
||||
┌───────────────────────────────────────────────────────────────┐
|
||||
│ Card total: 64 GiB │
|
||||
│ Instance 1: requested_memory = 64 * 0.4 = 25.6 GiB (in use) │
|
||||
│ Instance 2 start: init_snapshot.free_memory ≈ 38.4 GiB │
|
||||
│ Instance 2: requested_memory = 25.6 GiB │
|
||||
│ Profiling (fixed): non_kv_cache_memory = 0.5 GiB (weights) │
|
||||
│ available = 25.6 - 0.5 = 25.1 GiB → must be > 0 ✓ │
|
||||
└───────────────────────────────────────────────────────────────┘
|
||||
|
||||
Before the fix, non_kv_cache_memory was inflated to include first
|
||||
instance memory (~25.6 GiB), yielding available ≈ -1.32 GiB (OOM).
|
||||
"""
|
||||
total = int(64 * GiB_bytes)
|
||||
gpu_util = 0.4
|
||||
requested_memory = int(total * gpu_util) # 25.6 GiB
|
||||
|
||||
# First instance already occupies its full requested_memory slice
|
||||
first_instance_used = requested_memory # 25.6 GiB
|
||||
init_free = total - first_instance_used # ~38.4 GiB
|
||||
|
||||
# After the fix: profiling correctly reports only the second
|
||||
# instance's own model weights, not the first instance's memory.
|
||||
non_kv_cache = int(0.5 * GiB_bytes) # Qwen3-0.6B weights
|
||||
|
||||
worker = self._make_worker(requested_memory, init_free, total)
|
||||
profile_result = self._make_profile_result(
|
||||
free_memory_after=init_free - non_kv_cache,
|
||||
non_kv_cache_memory=non_kv_cache,
|
||||
)
|
||||
|
||||
with self._patch_memory_profiling(profile_result):
|
||||
result = worker.determine_available_memory()
|
||||
|
||||
self.assertGreater(
|
||||
result,
|
||||
0,
|
||||
"Second instance must have positive KV cache memory. "
|
||||
"A non-positive value means the multi-instance OOM bug "
|
||||
"(PR #7427) has regressed.",
|
||||
)
|
||||
expected = requested_memory - non_kv_cache
|
||||
self.assertEqual(result, expected)
|
||||
# Verify model_runner.profile_run() was called during profiling
|
||||
worker.model_runner.profile_run.assert_called_once()
|
||||
|
||||
@patch("vllm_ascend.worker.worker.logger")
|
||||
def test_second_instance_buggy_non_kv_cache_gives_negative(self, mock_logger):
|
||||
"""
|
||||
Documents the *pre-fix* buggy behaviour that PR #7427 addresses.
|
||||
|
||||
When non_kv_cache_memory is erroneously inflated to include memory
|
||||
already held by the first instance (~25.6 GiB extra), the formula
|
||||
available = requested_memory - non_kv_cache_memory
|
||||
yields a negative value, confirming why the fix was necessary.
|
||||
|
||||
This test is intentionally asserting the *negative* outcome to
|
||||
document the regressed state; it is NOT testing the fix itself.
|
||||
"""
|
||||
total = int(64 * GiB_bytes)
|
||||
gpu_util = 0.4
|
||||
requested_memory = int(total * gpu_util) # 25.6 GiB
|
||||
|
||||
first_instance_used = requested_memory # 25.6 GiB
|
||||
init_free = total - first_instance_used # ~38.4 GiB
|
||||
|
||||
# Buggy: non_kv_cache_memory = first-instance memory + second-instance weights
|
||||
buggy_non_kv_cache = int((25.6 + 0.5) * GiB_bytes) # ~26.1 GiB
|
||||
|
||||
worker = self._make_worker(requested_memory, init_free, total)
|
||||
profile_result = self._make_profile_result(
|
||||
# free_memory decreased only by the actual new allocation (weights)
|
||||
free_memory_after=init_free - int(0.5 * GiB_bytes),
|
||||
non_kv_cache_memory=buggy_non_kv_cache,
|
||||
)
|
||||
|
||||
with self._patch_memory_profiling(profile_result):
|
||||
result = worker.determine_available_memory()
|
||||
|
||||
# Pre-fix: 25.6 GiB - 26.1 GiB = -0.5 GiB (negative → OOM)
|
||||
self.assertLess(
|
||||
result,
|
||||
0,
|
||||
"With the pre-fix (buggy) non_kv_cache_memory the result must be "
|
||||
"negative; this documents the OOM regression that PR #7427 fixed.",
|
||||
)
|
||||
|
||||
@patch("vllm_ascend.worker.worker.logger")
|
||||
def test_assert_raises_when_free_memory_increases_after_profile(self, mock_logger):
|
||||
"""
|
||||
determine_available_memory() must raise AssertionError when free memory
|
||||
after profiling is greater than before (external process released memory
|
||||
during profiling, invalidating the measurement).
|
||||
"""
|
||||
total = int(64 * GiB_bytes)
|
||||
requested_memory = int(total * 0.9)
|
||||
init_free = int(60 * GiB_bytes)
|
||||
|
||||
worker = self._make_worker(requested_memory, init_free, total)
|
||||
# Abnormal: free memory increased after profiling
|
||||
profile_result = self._make_profile_result(
|
||||
free_memory_after=init_free + int(1 * GiB_bytes), # went UP
|
||||
non_kv_cache_memory=int(0.5 * GiB_bytes),
|
||||
)
|
||||
|
||||
with self._patch_memory_profiling(profile_result), self.assertRaises(AssertionError) as ctx:
|
||||
worker.determine_available_memory()
|
||||
|
||||
self.assertIn("Error in memory profiling", str(ctx.exception))
|
||||
|
||||
@patch("vllm_ascend.worker.worker.logger")
|
||||
def test_second_instance_tight_memory_still_positive(self, mock_logger):
|
||||
"""
|
||||
Edge case: card is almost full when second instance starts.
|
||||
|
||||
Even with very little free memory left, as long as requested_memory >
|
||||
non_kv_cache_memory (i.e. there is room for at least some KV blocks),
|
||||
the result must be positive.
|
||||
"""
|
||||
total = int(32 * GiB_bytes) # smaller card (e.g. 910B1)
|
||||
gpu_util = 0.3
|
||||
requested_memory = int(total * gpu_util) # 9.6 GiB
|
||||
|
||||
# First instance has consumed most of its requested slice
|
||||
first_instance_used = requested_memory # 9.6 GiB
|
||||
init_free = total - first_instance_used # 22.4 GiB
|
||||
|
||||
non_kv_cache = int(0.5 * GiB_bytes) # Qwen3-0.6B
|
||||
|
||||
worker = self._make_worker(requested_memory, init_free, total)
|
||||
profile_result = self._make_profile_result(
|
||||
free_memory_after=init_free - non_kv_cache,
|
||||
non_kv_cache_memory=non_kv_cache,
|
||||
)
|
||||
|
||||
with self._patch_memory_profiling(profile_result):
|
||||
result = worker.determine_available_memory()
|
||||
|
||||
self.assertGreater(result, 0)
|
||||
self.assertEqual(result, requested_memory - non_kv_cache)
|
||||
1511
tests/ut/worker/a2/test_worker_v1.py
Normal file
1511
tests/ut/worker/a2/test_worker_v1.py
Normal file
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user