xc-llm-ascend/tests/ut/model_loader/netloader/test_netloader_load.py

#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#

from unittest.mock import MagicMock, patch

import pytest

from vllm_ascend.model_loader.netloader.load import elastic_load


@pytest.fixture
def mock_sources():
    return [
        {
            "device_id": 0,
            "sources": ["a", "b"]
        },
        {
            "device_id": 1,
            "sources": ["c"]
        },
    ]


@patch("vllm_ascend.model_loader.netloader.interaction.elastic.ElasticClient")
@patch("vllm_ascend.model_loader.netloader.executor.elastic_load.P2PLoad")
def test_sources_this_device_empty(mock_p2p, mock_client):
    sources = [{"device_id": 1, "sources": ["c"]}]
    result = elastic_load("model", 0, "model_path", sources, 1, 1)
    assert result is None
    mock_client.assert_not_called()
    mock_p2p.assert_not_called()


@patch("vllm_ascend.model_loader.netloader.interaction.elastic.ElasticClient")
@patch("vllm_ascend.model_loader.netloader.executor.elastic_load.P2PLoad")
def test_client_s_none(mock_p2p, mock_client, mock_sources):
    # Simulate ElasticClient.s as None
    mock_instance = MagicMock()
    mock_instance.s = None
    mock_client.return_value = mock_instance
    result = elastic_load("model", 0, "model_path", mock_sources, 1, 1)
    assert result is None


@patch("vllm_ascend.model_loader.netloader.interaction.elastic.ElasticClient")
@patch("vllm_ascend.model_loader.netloader.executor.elastic_load.P2PLoad")
def test_client_ack_none(mock_p2p, mock_client, mock_sources):
    # Simulate ElasticClient.ack as None
    mock_instance = MagicMock()
    mock_instance.s = True
    mock_instance.ack = None
    mock_client.return_value = mock_instance
    result = elastic_load("model", 0, "model_path", mock_sources, 1, 1)
    assert result is None


@patch("vllm_ascend.model_loader.netloader.load.P2PLoad")
@patch("vllm_ascend.model_loader.netloader.load.logger")
def test_model_load_fail(mock_logger, mock_p2p):
    mock_client = MagicMock()
    mock_client.s = True
    mock_client.ack = ["foo", "bar"]
    mock_client.server_addr = "addr"

    with patch("vllm_ascend.model_loader.netloader.load.ElasticClient",
               return_value=mock_client):
        # P2PLoad.load returns None
        mock_p2p_instance = MagicMock()
        mock_p2p_instance.load.return_value = None
        mock_p2p.return_value = mock_p2p_instance

        sources = [{"device_id": 0, "sources": ["whatever"]}]
        result = elastic_load("model", 0, "model_path", sources, 1, 1)
        assert result is None
        mock_logger.error.assert_called_once()


@patch("vllm_ascend.model_loader.netloader.load.P2PLoad")
@patch("vllm_ascend.model_loader.netloader.load.logger")
def test_model_load_success(mock_logger, mock_p2p):
    mock_client = MagicMock()
    mock_client.s = True
    mock_client.ack = ["foo", "bar"]
    mock_client.server_addr = "addr"

    with patch("vllm_ascend.model_loader.netloader.load.ElasticClient",
               return_value=mock_client):
        expected_model = object()
        mock_p2p_instance = MagicMock()
        mock_p2p_instance.load.return_value = expected_model
        mock_p2p.return_value = mock_p2p_instance

        sources = [{"device_id": 0, "sources": ["whatever"]}]
        result = elastic_load("model", 0, "model_path", sources, 1, 1)
        assert result is expected_model
        mock_logger.info.assert_called_once()


if __name__ == "__main__":
    pytest.main()
[Misc] Add a model loader that utilizes HCCL for weight loading (#2888) ### What this PR does / why we need it? This PR introduces a new model loader called Netloader, which leverages high-bandwidth P2P direct transfer between NPU cards to achieve weight loading. Netloader is implemented as a plugin through the newly added 'register_model_loader' function in vLLM 0.10. It facilitates the process of weight loading by sending weights from a pre-loaded model (server) to an empty model of a newly started instance (client). The server operates concurrently with normal inference tasks through sub-threads and the 'stateless_init_torch_distributed_process_group' in vLLM. The client initiates a transfer request after verifying that the model and partitioning method are the same as the server's, and uses HCCL's collective communication (send/recv) to load the weights in the order they are stored in the model. Application Scenarios: 1. Significantly Reduces Inference Instance Startup Time By reusing the weights of already loaded instances and performing high-speed transfers directly between computing cards, this method reduces model loading latency compared to traditional remote/local pull methods. 2. Reduces Network and Storage Pressure Avoids the need to repeatedly download weight files from remote repositories, reducing the impact on centralized storage and network traffic, thereby enhancing overall system stability and service quality. 3. Improves Resource Utilization and Reduces Costs Accelerating the loading process reduces reliance on redundant computing pools, allowing computing resources to be elastically scaled and reclaimed as needed. 4. Enhances Business Continuity and High Availability In fault recovery scenarios, new instances can quickly take over existing services, avoiding prolonged business interruptions and improving the system's high availability and user experience. ### Does this PR introduce _any_ user-facing change? Netloader utilizes the existing --load-format=netloader and --model-loader-extra-config to be activated. The model-loader-extra-config needs to be input as a JSON string (as it is now) Afterwards, you can check whether the outputs for the same sentence are consistent when the temperature is set to 0. Signed-off-by: destinysky <kangrui10@126.com> - vLLM version: v0.11.0rc3 - vLLM main: https://github.com/vllm-project/vllm/commit/v0.11.0 --------- Signed-off-by: destinysky <kangrui10@126.com> 2025-10-23 15:56:07 +08:00			`#`
			`# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.`
			`#`
			`# Licensed under the Apache License, Version 2.0 (the "License");`
			`# you may not use this file except in compliance with the License.`
			`# You may obtain a copy of the License at`
			`#`
			`# http://www.apache.org/licenses/LICENSE-2.0`
			`#`
			`# Unless required by applicable law or agreed to in writing, software`
			`# distributed under the License is distributed on an "AS IS" BASIS,`
			`# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.`
			`# See the License for the specific language governing permissions and`
			`# limitations under the License.`
			`#`

			`from unittest.mock import MagicMock, patch`

			`import pytest`

			`from vllm_ascend.model_loader.netloader.load import elastic_load`


			`@pytest.fixture`
			`def mock_sources():`
			`return [`
			`{`
			`"device_id": 0,`
			`"sources": ["a", "b"]`
			`},`
			`{`
			`"device_id": 1,`
			`"sources": ["c"]`
			`},`
			`]`


			`@patch("vllm_ascend.model_loader.netloader.interaction.elastic.ElasticClient")`
			`@patch("vllm_ascend.model_loader.netloader.executor.elastic_load.P2PLoad")`
			`def test_sources_this_device_empty(mock_p2p, mock_client):`
			`sources = [{"device_id": 1, "sources": ["c"]}]`
			`result = elastic_load("model", 0, "model_path", sources, 1, 1)`
			`assert result is None`
			`mock_client.assert_not_called()`
			`mock_p2p.assert_not_called()`


			`@patch("vllm_ascend.model_loader.netloader.interaction.elastic.ElasticClient")`
			`@patch("vllm_ascend.model_loader.netloader.executor.elastic_load.P2PLoad")`
			`def test_client_s_none(mock_p2p, mock_client, mock_sources):`
			`# Simulate ElasticClient.s as None`
			`mock_instance = MagicMock()`
			`mock_instance.s = None`
			`mock_client.return_value = mock_instance`
			`result = elastic_load("model", 0, "model_path", mock_sources, 1, 1)`
			`assert result is None`


			`@patch("vllm_ascend.model_loader.netloader.interaction.elastic.ElasticClient")`
			`@patch("vllm_ascend.model_loader.netloader.executor.elastic_load.P2PLoad")`
			`def test_client_ack_none(mock_p2p, mock_client, mock_sources):`
			`# Simulate ElasticClient.ack as None`
			`mock_instance = MagicMock()`
			`mock_instance.s = True`
			`mock_instance.ack = None`
			`mock_client.return_value = mock_instance`
			`result = elastic_load("model", 0, "model_path", mock_sources, 1, 1)`
			`assert result is None`


			`@patch("vllm_ascend.model_loader.netloader.load.P2PLoad")`
			`@patch("vllm_ascend.model_loader.netloader.load.logger")`
			`def test_model_load_fail(mock_logger, mock_p2p):`
			`mock_client = MagicMock()`
			`mock_client.s = True`
			`mock_client.ack = ["foo", "bar"]`
			`mock_client.server_addr = "addr"`

			`with patch("vllm_ascend.model_loader.netloader.load.ElasticClient",`
			`return_value=mock_client):`
			`# P2PLoad.load returns None`
			`mock_p2p_instance = MagicMock()`
			`mock_p2p_instance.load.return_value = None`
			`mock_p2p.return_value = mock_p2p_instance`

			`sources = [{"device_id": 0, "sources": ["whatever"]}]`
			`result = elastic_load("model", 0, "model_path", sources, 1, 1)`
			`assert result is None`
			`mock_logger.error.assert_called_once()`


			`@patch("vllm_ascend.model_loader.netloader.load.P2PLoad")`
			`@patch("vllm_ascend.model_loader.netloader.load.logger")`
			`def test_model_load_success(mock_logger, mock_p2p):`
			`mock_client = MagicMock()`
			`mock_client.s = True`
			`mock_client.ack = ["foo", "bar"]`
			`mock_client.server_addr = "addr"`

			`with patch("vllm_ascend.model_loader.netloader.load.ElasticClient",`
			`return_value=mock_client):`
			`expected_model = object()`
			`mock_p2p_instance = MagicMock()`
			`mock_p2p_instance.load.return_value = expected_model`
			`mock_p2p.return_value = mock_p2p_instance`

			`sources = [{"device_id": 0, "sources": ["whatever"]}]`
			`result = elastic_load("model", 0, "model_path", sources, 1, 1)`
			`assert result is expected_model`
			`mock_logger.info.assert_called_once()`


			`if __name__ == "__main__":`
			`pytest.main()`