There is a lot hack code for v0.11.0, which makes the code hard to
upgrade to newer vLLM version. Since v0.11.0 will release soon. Let's
drop v0.11.0 support first. Then we'll upgrade to v0.11.2 soon.
- vLLM version: v0.11.0
- vLLM main:
2918c1b49c
Signed-off-by: wangxiyuan <wangxiyuan1007@gmail.com>
17 lines
1020 B
Python
17 lines
1020 B
Python
import vllm.model_executor.layers.fla.ops.chunk
|
|
import vllm.model_executor.layers.fla.ops.fused_recurrent
|
|
import vllm.model_executor.layers.fla.ops.layernorm_guard
|
|
import vllm.model_executor.layers.mamba.ops.causal_conv1d
|
|
|
|
from vllm_ascend.ops.casual_conv1d import (causal_conv1d_fn,
|
|
causal_conv1d_update_npu)
|
|
from vllm_ascend.ops.fla import LayerNormFn, torch_chunk_gated_delta_rule
|
|
from vllm_ascend.ops.sigmoid_gating import \
|
|
fused_recurrent_gated_delta_rule_fwd_kernel
|
|
|
|
vllm.model_executor.layers.mamba.ops.causal_conv1d.causal_conv1d_update = causal_conv1d_update_npu
|
|
vllm.model_executor.layers.mamba.ops.causal_conv1d.causal_conv1d_fn = causal_conv1d_fn
|
|
vllm.model_executor.layers.fla.ops.fused_recurrent.fused_recurrent_gated_delta_rule_fwd_kernel = fused_recurrent_gated_delta_rule_fwd_kernel
|
|
vllm.model_executor.layers.fla.ops.layernorm_guard.LayerNormFn = LayerNormFn
|
|
vllm.model_executor.layers.fla.ops.chunk.chunk_gated_delta_rule = torch_chunk_gated_delta_rule
|