Drop torchair (#4814)

aclgraph is stable and fast now. Let's drop torchair graph mode now. TODO: some logic to adapt torchair should be cleaned up as well. We'll do it in the following PR. - vLLM version: v0.12.0 - vLLM main: ad32e3e19c Signed-off-by: wangxiyuan <wangxiyuan1007@gmail.com> Co-authored-by: Mengqing Cao <cmq0113@163.com>
2025-12-10 09:20:40 +08:00
parent ba9cda9dfd
commit 835b4c8f1d
84 changed files with 77 additions and 16881 deletions
--- a/docs/source/tutorials/DeepSeek-V3.1.md
+++ b/docs/source/tutorials/DeepSeek-V3.1.md
@@ -128,9 +128,8 @@ vllm serve /weights/DeepSeek-V3.1_w8a8mix_mtp \
 --trust-remote-code \
 --no-enable-prefix-caching \
 --gpu-memory-utilization 0.92 \
--speculative-config '{"num_speculative_tokens": 1, "method": "deepseek_mtp"}' \
+--speculative-config '{"num_speculative_tokens": 1, "method": "mtp"}' \
 --compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}' \
--additional-config '{"torchair_graph_config":{"enabled":false}}'
 ```

 ### Multi-node Deployment
@@ -190,9 +189,8 @@ vllm serve /weights/DeepSeek-V3.1_w8a8mix_mtp \
 --trust-remote-code \
 --no-enable-prefix-caching \
 --gpu-memory-utilization 0.94 \
--speculative-config '{"num_speculative_tokens": 1, "method": "deepseek_mtp"}' \
+--speculative-config '{"num_speculative_tokens": 1, "method": "mtp"}' \
 --compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}' \
--additional-config '{"torchair_graph_config":{"enabled":false}}'
 ```

 **Node 1**
@@ -247,9 +245,8 @@ vllm serve /weights/DeepSeek-V3.1_w8a8mix_mtp \
 --trust-remote-code \
 --no-enable-prefix-caching \
 --gpu-memory-utilization 0.94 \
--speculative-config '{"num_speculative_tokens": 1, "method": "deepseek_mtp"}' \
+--speculative-config '{"num_speculative_tokens": 1, "method": "mtp"}' \
 --compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}' \
--additional-config '{"torchair_graph_config":{"enabled":false}}'
 ```

 ### Prefill-Decode Disaggregation
@@ -421,7 +418,7 @@ vllm serve /weights/DeepSeek-V3.1_w8a8mix_mtp \
  --gpu-memory-utilization 0.9 \
  --quantization ascend \
  --no-enable-prefix-caching \
-  --speculative-config '{"num_speculative_tokens": 1, "method": "deepseek_mtp"}' \
+  --speculative-config '{"num_speculative_tokens": 1, "method": "mtp"}' \
  --additional-config '{"recompute_scheduler_enable":true,"enable_shared_expert_dp": true}' \
  --kv-transfer-config \
  '{"kv_connector": "MooncakeConnector",