chore: remove unnecessary limits on quantization methods in test script (#7997)

2025-07-14 07:11:49 +08:00
parent 9379da77de
commit b5dd5e8741
1 changed files with 0 additions and 4 deletions
--- a/test/srt/test_vllm_dependency.py
+++ b/test/srt/test_vllm_dependency.py
@@ -42,10 +42,6 @@ def popen_launch_server_wrapper(base_url, model, is_fp8, is_tp2):
        other_args.extend(["--tp", "2"])
    if "DeepSeek" in model:
        other_args.extend(["--mem-frac", "0.85"])
-    if "AWQ" in model:
-        other_args.extend(["--quantization", "awq"])
-    elif "GPTQ" in model:
-        other_args.extend(["--quantization", "gptq"])

    process = popen_launch_server(
        model,