adapt to main2main for model runner v2 (#7578)

### What this PR does / why we need it? This PR aims to adapt to newest commit of vllm main branch for model runner v2. please refer to https://github.com/vllm-project/vllm-ascend/issues/5208 ### Does this PR introduce _any_ user-facing change? no ### How was this patch tested? - vLLM version: v0.18.0 - vLLM main: ed359c497a --------- Signed-off-by: Ronald1995 <ronaldautomobile@163.com>
2026-03-25 09:08:44 +08:00
parent fc3ec100bc
commit d96440924a
16 changed files with 239 additions and 264 deletions
--- a/vllm_ascend/patch/worker/patch_v2/patch_triton.py
+++ b/vllm_ascend/patch/worker/patch_v2/patch_triton.py
@@ -0,0 +1,12 @@
+from vllm.v1.worker.gpu import input_batch
+from vllm.v1.worker.gpu.sample import gumbel, logprob, penalties
+
+from vllm_ascend.worker.v2.input_batch import post_update
+from vllm_ascend.worker.v2.sample.gumbel import gumbel_sample
+from vllm_ascend.worker.v2.sample.logprob import compute_token_logprobs
+from vllm_ascend.worker.v2.sample.penalties import apply_penalties
+
+logprob.compute_token_logprobs = compute_token_logprobs
+penalties.apply_penalties = apply_penalties
+gumbel.gumbel_sample = gumbel_sample
+input_batch.post_update = post_update