[Doc]clean up ascend scheduler config from doc (#4612)
clean up ascend scheduler config from doc - vLLM version: v0.11.2 Signed-off-by: wangxiyuan <wangxiyuan1007@gmail.com>
This commit is contained in:
@@ -174,7 +174,7 @@ vllm serve vllm-ascend/DeepSeek-V3.2-Exp-W8A8 \
|
|||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.92 \
|
--gpu-memory-utilization 0.92 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
### Multi-node Deployment
|
### Multi-node Deployment
|
||||||
@@ -226,7 +226,7 @@ vllm serve /root/.cache/Modelers_Park/DeepSeek-V3.2-Exp \
|
|||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.9 \
|
--gpu-memory-utilization 0.9 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
**Node 1**
|
**Node 1**
|
||||||
@@ -270,7 +270,7 @@ vllm serve /root/.cache/Modelers_Park/DeepSeek-V3.2-Exp \
|
|||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.92 \
|
--gpu-memory-utilization 0.92 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
::::
|
::::
|
||||||
@@ -318,7 +318,7 @@ vllm serve vllm-ascend/DeepSeek-V3.2-Exp-W8A8 \
|
|||||||
--quantization ascend \
|
--quantization ascend \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.9 \
|
--gpu-memory-utilization 0.9 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
**Node 1**
|
**Node 1**
|
||||||
@@ -365,7 +365,7 @@ vllm serve vllm-ascend/DeepSeek-V3.2-Exp-W8A8 \
|
|||||||
--quantization ascend \
|
--quantization ascend \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.92 \
|
--gpu-memory-utilization 0.92 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true,"graph_batch_sizes":[16]}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
::::
|
::::
|
||||||
|
|||||||
@@ -137,7 +137,7 @@ vllm serve vllm-ascend/DeepSeek-V3.1-W8A8 \
|
|||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.9 \
|
--gpu-memory-utilization 0.9 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
**Node 1**
|
**Node 1**
|
||||||
@@ -182,7 +182,7 @@ vllm serve vllm-ascend/DeepSeek-V3.1-W8A8 \
|
|||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.92 \
|
--gpu-memory-utilization 0.92 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
The deployment view looks like:
|
The deployment view looks like:
|
||||||
|
|||||||
@@ -93,7 +93,7 @@ vllm serve /home/cache/weights/Kimi-K2-Instruct-W8A8 \
|
|||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.9 \
|
--gpu-memory-utilization 0.9 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
**Node 1**
|
**Node 1**
|
||||||
@@ -137,7 +137,7 @@ vllm serve /home/cache/weights/Kimi-K2-Instruct-W8A8 \
|
|||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--no-enable-prefix-caching \
|
--no-enable-prefix-caching \
|
||||||
--gpu-memory-utilization 0.92 \
|
--gpu-memory-utilization 0.92 \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":true}}'
|
--additional-config '{"torchair_graph_config":{"enabled":true}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
The deployment view looks like:
|
The deployment view looks like:
|
||||||
|
|||||||
@@ -157,12 +157,7 @@ if __name__ == "__main__":
|
|||||||
additional_config={
|
additional_config={
|
||||||
'torchair_graph_config': {
|
'torchair_graph_config': {
|
||||||
'enabled': True,
|
'enabled': True,
|
||||||
},
|
}
|
||||||
'ascend_scheduler_config':{
|
|
||||||
'enabled': True,
|
|
||||||
'enable_chunked_prefill' : False,
|
|
||||||
'chunked_prefill_enabled': False
|
|
||||||
},
|
|
||||||
})
|
})
|
||||||
|
|
||||||
outputs = llm.generate(prompts, sampling_params)
|
outputs = llm.generate(prompts, sampling_params)
|
||||||
|
|||||||
@@ -27,7 +27,6 @@ The following table lists additional configuration options available in vLLM Asc
|
|||||||
| Name | Type | Default | Description |
|
| Name | Type | Default | Description |
|
||||||
|-------------------------------------|------|---------|-----------------------------------------------------------------------------------------------------------------------------------------------|
|
|-------------------------------------|------|---------|-----------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
| `torchair_graph_config` | dict | `{}` | Configuration options for torchair graph mode |
|
| `torchair_graph_config` | dict | `{}` | Configuration options for torchair graph mode |
|
||||||
| `ascend_scheduler_config` | dict | `{}` | Configuration options for ascend scheduler |
|
|
||||||
| `weight_prefetch_config` | dict | `{}` | Configuration options for weight prefetch |
|
| `weight_prefetch_config` | dict | `{}` | Configuration options for weight prefetch |
|
||||||
| `refresh` | bool | `false` | Whether to refresh global Ascend configuration content. This is usually used by rlhf or ut/e2e test case. |
|
| `refresh` | bool | `false` | Whether to refresh global Ascend configuration content. This is usually used by rlhf or ut/e2e test case. |
|
||||||
| `expert_map_path` | str | `None` | When using expert load balancing for an MoE model, an expert map path needs to be passed in. |
|
| `expert_map_path` | str | `None` | When using expert load balancing for an MoE model, an expert map path needs to be passed in. |
|
||||||
@@ -61,18 +60,6 @@ The details of each configuration option are as follows:
|
|||||||
| `enable_kv_nz`| bool | `False` | Whether to enable KV Cache NZ layout. This option only takes effect on models using MLA (for example, DeepSeek). |
|
| `enable_kv_nz`| bool | `False` | Whether to enable KV Cache NZ layout. This option only takes effect on models using MLA (for example, DeepSeek). |
|
||||||
| `enable_super_kernel` | bool | `False` | Whether to enable super kernel to fuse operators in deepseek moe layers. This option only takes effects on moe models using dynamic w8a8 quantization.|
|
| `enable_super_kernel` | bool | `False` | Whether to enable super kernel to fuse operators in deepseek moe layers. This option only takes effects on moe models using dynamic w8a8 quantization.|
|
||||||
|
|
||||||
**ascend_scheduler_config**
|
|
||||||
|
|
||||||
| Name | Type | Default | Description |
|
|
||||||
| ---- | ---- | ------- | ----------- |
|
|
||||||
| `enabled` | bool | `False` | Whether to enable ascend scheduler for V1 engine.|
|
|
||||||
| `enable_pd_transfer` | bool | `False` | Whether to enable P-D transfer. When it is enabled, decode is started only when prefill of all requests is done. This option only takes effect on offline inference. |
|
|
||||||
| `decode_max_num_seqs` | int | `0` | Whether to change max_num_seqs of decode phase when P-D transfer is enabled. This option only takes effect when enable_pd_transfer is True. |
|
|
||||||
| `max_long_partial_prefills` | Union[int, float] | `float('inf')` | The maximum number of prompts longer than long_prefill_token_threshold that will be prefilled concurrently. |
|
|
||||||
| `long_prefill_token_threshold` | Union[int, float] | `float('inf')` | a request is considered long if the prompt is longer than this number of tokens. |
|
|
||||||
|
|
||||||
ascend_scheduler_config also supports the options from [vllm scheduler config](https://docs.vllm.ai/en/stable/api/vllm/config.html#vllm.config.SchedulerConfig). For example, you can add `enable_chunked_prefill: True` to ascend_scheduler_config as well.
|
|
||||||
|
|
||||||
**weight_prefetch_config**
|
**weight_prefetch_config**
|
||||||
|
|
||||||
| Name | Type | Default | Description |
|
| Name | Type | Default | Description |
|
||||||
@@ -93,12 +80,6 @@ An example of additional configuration is as follows:
|
|||||||
"graph_batch_sizes_init": False,
|
"graph_batch_sizes_init": False,
|
||||||
"enable_kv_nz": False
|
"enable_kv_nz": False
|
||||||
},
|
},
|
||||||
"ascend_scheduler_config": {
|
|
||||||
"enabled": True,
|
|
||||||
"enable_chunked_prefill": True,
|
|
||||||
"max_long_partial_prefills": 1,
|
|
||||||
"long_prefill_token_threshold": 4096,
|
|
||||||
},
|
|
||||||
"weight_prefetch_config": {
|
"weight_prefetch_config": {
|
||||||
"enabled": True,
|
"enabled": True,
|
||||||
"prefetch_ratio": {
|
"prefetch_ratio": {
|
||||||
|
|||||||
@@ -45,14 +45,14 @@ import os
|
|||||||
from vllm import LLM
|
from vllm import LLM
|
||||||
|
|
||||||
# TorchAirGraph only works without chunked-prefill now
|
# TorchAirGraph only works without chunked-prefill now
|
||||||
model = LLM(model="path/to/DeepSeek-R1-0528", additional_config={"torchair_graph_config": {"enabled": True},"ascend_scheduler_config": {"enabled": True}})
|
model = LLM(model="path/to/DeepSeek-R1-0528", additional_config={"torchair_graph_config": {"enabled": True}})
|
||||||
outputs = model.generate("Hello, how are you?")
|
outputs = model.generate("Hello, how are you?")
|
||||||
```
|
```
|
||||||
|
|
||||||
Online example:
|
Online example:
|
||||||
|
|
||||||
```shell
|
```shell
|
||||||
vllm serve path/to/DeepSeek-R1-0528 --additional-config='{"torchair_graph_config": {"enabled": true},"ascend_scheduler_config": {"enabled": true}}'
|
vllm serve path/to/DeepSeek-R1-0528 --additional-config='{"torchair_graph_config": {"enabled": true}}'
|
||||||
```
|
```
|
||||||
|
|
||||||
You can find more details about additional configuration [here](../configuration/additional_config.md).
|
You can find more details about additional configuration [here](../configuration/additional_config.md).
|
||||||
|
|||||||
@@ -42,7 +42,6 @@ if __name__ == "__main__":
|
|||||||
enable_chunked_prefill=False,
|
enable_chunked_prefill=False,
|
||||||
max_num_batched_tokens=2048,
|
max_num_batched_tokens=2048,
|
||||||
max_model_len=1024,
|
max_model_len=1024,
|
||||||
additional_config={"ascend_scheduler_config": {"enabled": False}},
|
|
||||||
max_num_seqs=1,
|
max_num_seqs=1,
|
||||||
block_size=128,
|
block_size=128,
|
||||||
gpu_memory_utilization=0.9
|
gpu_memory_utilization=0.9
|
||||||
|
|||||||
@@ -28,4 +28,4 @@ vllm serve Qwen/Qwen1.5-MoE-A2.7B \
|
|||||||
--gpu-memory-utilization 0.9 \
|
--gpu-memory-utilization 0.9 \
|
||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--enforce-eager \
|
--enforce-eager \
|
||||||
--additional-config '{"ascend_scheduler_config":{"enabled":true},"torchair_graph_config":{"enabled":false, "use_cached_graph":false}}'
|
--additional-config '{"torchair_graph_config":{"enabled":false, "use_cached_graph":false}}'
|
||||||
|
|||||||
Reference in New Issue
Block a user