init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -25,12 +25,9 @@ def test_models_topk() -> None:
example_prompts = [
"The capital of France is",
]
sampling_params = SamplingParams(max_tokens=10,
temperature=0.0,
top_k=10,
top_p=0.9)
sampling_params = SamplingParams(max_tokens=10, temperature=0.0, top_k=10, top_p=0.9)
with VllmRunner("Qwen/Qwen3-0.6B",
max_model_len=4096,
gpu_memory_utilization=0.7) as runner:
with VllmRunner(
"Qwen/Qwen3-0.6B", max_model_len=4096, cudagraph_capture_sizes=[1, 2, 4, 8], gpu_memory_utilization=0.7
) as runner:
runner.generate(example_prompts, sampling_params)

View File

@@ -1,2 +1,2 @@
# Base docker image used to build the vllm-ascend e2e test image, which is built in the vLLM repository
BASE_IMAGE_NAME="quay.io/ascend/cann:8.2.rc1-910b-ubuntu22.04-py3.11"
BASE_IMAGE_NAME="quay.io/ascend/cann:9.1.0-910b-ubuntu22.04-py3.12"