@@ -25,12 +25,9 @@ def test_models_topk() -> None:
|
||||
example_prompts = [
|
||||
"The capital of France is",
|
||||
]
|
||||
sampling_params = SamplingParams(max_tokens=10,
|
||||
temperature=0.0,
|
||||
top_k=10,
|
||||
top_p=0.9)
|
||||
sampling_params = SamplingParams(max_tokens=10, temperature=0.0, top_k=10, top_p=0.9)
|
||||
|
||||
with VllmRunner("Qwen/Qwen3-0.6B",
|
||||
max_model_len=4096,
|
||||
gpu_memory_utilization=0.7) as runner:
|
||||
with VllmRunner(
|
||||
"Qwen/Qwen3-0.6B", max_model_len=4096, cudagraph_capture_sizes=[1, 2, 4, 8], gpu_memory_utilization=0.7
|
||||
) as runner:
|
||||
runner.generate(example_prompts, sampling_params)
|
||||
|
||||
@@ -1,2 +1,2 @@
|
||||
# Base docker image used to build the vllm-ascend e2e test image, which is built in the vLLM repository
|
||||
BASE_IMAGE_NAME="quay.io/ascend/cann:8.2.rc1-910b-ubuntu22.04-py3.11"
|
||||
BASE_IMAGE_NAME="quay.io/ascend/cann:9.1.0-910b-ubuntu22.04-py3.12"
|
||||
|
||||
Reference in New Issue
Block a user