[Feature] implement eagle spec decoding for model runner v2 (#5840)

### What this PR does / why we need it? this pr implement eagle spec decoding for model runner v2, please see RFC https://github.com/vllm-project/vllm-ascend/issues/5208 ### Does this PR introduce _any_ user-facing change? No ### How was this patch tested? vLLM version: v0.13.0 --------- Signed-off-by: Ronald1995 <ronaldautomobile@163.com>
2026-01-14 09:18:05 +08:00
parent 0415e694cd
commit e20813f441
9 changed files with 468 additions and 82 deletions
--- a/vllm_ascend/patch/worker/patch_triton.py
+++ b/vllm_ascend/patch/worker/patch_triton.py
@@ -1,4 +1,5 @@
 import vllm.model_executor.layers.mamba.ops.causal_conv1d
+import vllm.v1.worker.gpu.sample.gumbel

 from vllm_ascend.ops.triton.fla.chunk import chunk_gated_delta_rule
 from vllm_ascend.ops.triton.fla.layernorm_guard import LayerNormFn
@@ -6,9 +7,12 @@ from vllm_ascend.ops.triton.fla.sigmoid_gating import \
    fused_recurrent_gated_delta_rule_fwd_kernel
 from vllm_ascend.ops.triton.mamba.causal_conv1d import (
    causal_conv1d_fn, causal_conv1d_update_npu)
+from vllm_ascend.worker.v2.sample.gumbel import \
+    gumbel_sample as ascend_gumbel_sample

 vllm.model_executor.layers.mamba.ops.causal_conv1d.causal_conv1d_update = causal_conv1d_update_npu
 vllm.model_executor.layers.mamba.ops.causal_conv1d.causal_conv1d_fn = causal_conv1d_fn
 vllm.model_executor.layers.fla.ops.fused_recurrent.fused_recurrent_gated_delta_rule_fwd_kernel = fused_recurrent_gated_delta_rule_fwd_kernel
 vllm.model_executor.layers.fla.ops.layernorm_guard.LayerNormFn = LayerNormFn
 vllm.model_executor.layers.fla.ops.chunk_gated_delta_rule = chunk_gated_delta_rule
+vllm.v1.worker.gpu.sample.gumbel.gumbel_sample = ascend_gumbel_sample