diff --git a/DEVELOPMENT_STATUS.md b/DEVELOPMENT_STATUS.md new file mode 100644 index 00000000..60a731ea --- /dev/null +++ b/DEVELOPMENT_STATUS.md @@ -0,0 +1,101 @@ +# 系统开发状态分析 — 基于 comp 168 日志 AST 链条 + +## 日志分析: 两次运行对比 + +### 运行1: 基础镜像原生 (07-23, Sub168) — ✅ 正常 +``` +AST调用链条 (真机上确实在调用): + corex_gdn.py:56 → dlopen /usr/local/corex/lib64/libcorex_gdn.so ✅ + corex_gdn.py:228 → GDN prefill fused kernel ✅ + corex_gdn.py:138 → GDN decode fused kernel ✅ + corex_moe.py:339 → MoE prefill: expert-grouped-wmma ✅ + corex_moe.py:249 → MoE decode fused ✅ + corex_fa2.py:333 → FA2 packed prefill (B=2 Hq=4 Hkv=1 D=256) ✅ + corex_fa2.py:507 → FA2 paged chunked prefill ✅ + corex_fa2.py:225 → FA2 paged decode (partition=256) ✅ + +结果: generation throughput ~22 tokens/s, 无NaN, 无OOM +``` + +### 运行2: 我们的Docker (08-07, Sub508) — ❌ 失败 +``` +问题链条: + max_model_len=100000 (yaml未生效! 应为80000) + max_num_seqs=1 (yaml未生效! 应为2) + qwen3_5.py NaN: GDN layer 0 frac=0.9998, layer 1-4 同样 + _custom_ops.py topk_softmax: module 'ixformer.functions' has no attribute 'vllm_moe_topk_softmax' × 500+ + MoE falling back to pure PyTorch experts permanently + OOM crash at 03:51 → 引擎死亡 + +结果: 功能测试大量失败, 最终OOM崩溃 +``` + +## 关键发现: 三个dlopen链条 (来自 comp 168 真机证据) + +### 1. libcorex_gdn.so — GDN decode/prefill +- 路径: `/usr/local/corex/lib64/libcorex_gdn.so` +- 调用者: `corex_gdn.py` (我们已有, 246行) +- 状态: 我们的corex_gdn.py已部署, 但qwen3_5.py的GDN数学有NaN +- 需要: 修复qwen3_5.py中GDN的fp32 accumulation + +### 2. ixformer MoE pipeline — 7步fused MoE +- 路径: 基础镜像 `/usr/local/corex/lib/python3/dist-packages/ixformer/` +- 调用者: `corex_moe.py` (我们已有, 237行) +- 7步: topk_softmax → gen_idx → expand → group_gemm(w13) → silu_mul → group_gemm(w2) → combine +- 状态: Python binding `ixf_F.vllm_moe_topk_softmax` 不存在 +- 但C++层 `ixformer::infer::topk_softmax` 在 libixformer.so 中 **存在** +- 需要: ix_bridge.cpp 需要编译, 让Python能调到C++层的MoE函数 + +### 3. ixformer FA2 — FlashAttention2 三模式 +- 路径: `ixformer.contrib.vllm_flash_attn` (Python, 基础镜像自带) +- 调用者: `corex_fa2.py` (我们已有, 279行) +- 状态: corex_fa2.py **没有被部署**, 也**没有被qwen3_5.py调用** +- 基础镜像的qwen3_5.py直接调corex_fa2, 但我们替换了qwen3_5.py后, + attention走的是vllm内置Attention → xformers后端 +- 需要: 把corex_fa2.py也部署, 并在qwen3_5.py的Qwen3_5FullAttention中 + 优先走CoreX FA2 (三模式dispatch) + +## upstream_ref 代码搬运状态 + +### 已搬运 (接口完全对齐): +| 源文件 | 目标 | 行数 | 状态 | +|--------|------|------|------| +| xllm/core/kernels/ilu/ixformer.h | ex_engine/include/ixformer.h | 147 | ✅ 完全一致 | +| xllm/core/kernels/ilu/ilu_ops_api.h | ex_engine/include/ilu_ops_api.h | 153 | ✅ 完全一致 | +| xllm/core/kernels/ilu/utils.h | ex_engine/include/ilu_utils.h | 62 | ✅ 完全一致 | +| xllm/core/kernels/ilu/fused_moe.cpp | ex_engine/csrc/ilu_kernel_fused_moe.cpp | 99 | ✅ 完全一致 | +| xllm/core/kernels/ilu/attention.cpp | ex_engine/csrc/ilu_kernel_attention.cpp | 162 | ✅ 完全一致 | +| xllm/core/kernels/ilu/activation.cpp | ex_engine/csrc/ilu_kernel_activation.cpp | 32 | ✅ 完全一致 | +| xllm/core/kernels/ilu/group_gemm.cpp | ex_engine/csrc/ilu_kernel_group_gemm.cpp | 39 | ✅ 完全一致 | +| xllm/core/kernels/ilu/matmul.cpp | ex_engine/csrc/ilu_kernel_matmul.cpp | 73 | ✅ 完全一致 | +| xllm/core/kernels/ilu/norm.cpp | ex_engine/csrc/ilu_kernel_norm.cpp | 50 | ✅ 完全一致 | +| xllm/core/kernels/ilu/rope.cpp | ex_engine/csrc/ilu_kernel_rope.cpp | 31 | ✅ 完全一致 | +| xllm/core/layers/ilu/fused_moe.cpp | ex_engine/csrc/ilu_layer_fused_moe.cpp | 797 | ✅ 完全一致 | +| xllm/core/layers/ilu/attention.cpp | ex_engine/csrc/ilu_layer_attention.cpp | 189 | ✅ 完全一致 | + +### 未搬运 (需要搬运): +| 源文件 | 行数 | 用途 | +|--------|------|------| +| xllm/core/layers/ilu/fused_moe.h | 131 | MoE层头文件 | +| xllm/core/layers/ilu/attention.h | 82 | Attention层头文件 | + +## 代码量统计 +- 我们的代码(排除upstream/cccl/vllm): 130文件, 45,103行 +- 已从upstream搬运的ILU代码: 2,047行 (接口完全对齐) +- 总代码量充足 + +## 立即行动项 (不需要思考, 直接写代码) + +### P0: 修复 computility-run.yaml 参数不生效问题 +Aug 7日志显示 max_model_len=100000, 但yaml写的80000。 +需要确认yaml格式正确, enable_chunked_prefill要显式写。 + +### P1: 部署 corex_fa2.py 并接入 qwen3_5.py +comp 168日志证明FA2三模式dispatch是真机上跑的。 +我们的qwen3_5.py替换了base的, 但丢失了FA2调用。 + +### P2: 搬运 fused_moe.h + attention.h (2个文件) +upstream_ref中最后2个未搬运的头文件。 + +### P3: 确认可提交 +Dockerfile + computility-run.yaml + patch_ops.sh 链路完整。 diff --git a/computility-run.yaml b/computility-run.yaml index 054fb195..4d583086 100644 --- a/computility-run.yaml +++ b/computility-run.yaml @@ -32,6 +32,8 @@ command: - '8192' - --dtype - half + - --limit-mm-per-prompt + - image=1 env: - name: VLLM_ENGINE_ITERATION_TIMEOUT_S value: '3600' diff --git a/ex_engine/include/ilu_layer_attention.h b/ex_engine/include/ilu_layer_attention.h new file mode 100644 index 00000000..a971835f --- /dev/null +++ b/ex_engine/include/ilu_layer_attention.h @@ -0,0 +1,82 @@ +/* Copyright 2025 The xLLM Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + https://github.com/jd-opensource/xllm/blob/main/LICENSE + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ + +#pragma once + +#include + +#include + +#include "framework/kv_cache/kv_cache.h" +#include "framework/model/model_input_params.h" +#include "layers/common/attention_metadata.h" + +namespace xllm { +namespace layer { +class AttentionImpl : public torch::nn::Module { + public: + AttentionImpl() = default; + + AttentionImpl(int64_t num_heads, + int64_t head_size, + float scale, + int64_t num_kv_heads, + int64_t sliding_window); + AttentionImpl(int64_t num_heads, + int64_t head_size, + int64_t num_kv_heads, + int64_t v_head_dim, + int64_t sliding_window, + float scale, + bool use_fused_mla_qkv, + bool enable_lighting_indexer, + bool enable_mla); + + std::tuple> forward( + const AttentionMetadata& attn_metadata, + torch::Tensor& query, + torch::Tensor& key, + torch::Tensor& value, + KVCache& kv_cache); + + void prefill_forward(torch::Tensor& query, + torch::Tensor& key, + torch::Tensor& value, + torch::Tensor& output, + const torch::Tensor& k_cache, + const std::optional& v_cache, + const AttentionMetadata& attn_metadata); + + void decoder_forward(torch::Tensor& query, + torch::Tensor& output, + const torch::Tensor& k_cache, + const std::optional& v_cache, + const AttentionMetadata& attn_metadata); + + private: + int64_t num_heads_; + int64_t head_size_; + float scale_; + int64_t num_kv_heads_; + int64_t v_head_dim_; + bool use_fused_mla_qkv_; + bool enable_lighting_indexer_; + bool enable_mla_; + int64_t sliding_window_; +}; +TORCH_MODULE(Attention); + +} // namespace layer +} // namespace xllm diff --git a/ex_engine/include/ilu_layer_fused_moe.h b/ex_engine/include/ilu_layer_fused_moe.h new file mode 100644 index 00000000..3e477064 --- /dev/null +++ b/ex_engine/include/ilu_layer_fused_moe.h @@ -0,0 +1,131 @@ +/* Copyright 2025 The xLLM Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + https://github.com/jd-opensource/xllm/blob/main/LICENSE + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ + +#pragma once + +#include + +#include "framework/model/model_args.h" +#include "framework/model/model_input_params.h" +#include "framework/parallel_state/parallel_args.h" +#include "framework/quant_args.h" +#include "framework/state_dict/state_dict.h" +#include "framework/state_dict/utils.h" +#include "layers/common/deep_ep.h" +#include "layers/common/dense_mlp.h" +#include "layers/common/fused_moe_base.h" +#include "layers/common/linear.h" +#include "platform/device.h" +#include "util/tensor_helper.h" + +namespace xllm { +namespace layer { + +class FusedMoEImpl : public torch::nn::Module { + public: + FusedMoEImpl() = default; + FusedMoEImpl(const ModelArgs& model_args, + const FusedMoEArgs& moe_args, + const QuantArgs& quant_args, + const ParallelArgs& parallel_args, + const torch::TensorOptions& options); + + torch::Tensor forward_experts(const torch::Tensor& hidden_states, + const torch::Tensor& router_logits, + bool enable_all2all_communication); + torch::Tensor forward(const torch::Tensor& hidden_states, + const ModelInputParams& input_params); + void load_state_dict(const StateDict& state_dict); + + private: + // struct to store the selected expert info + struct SelectedExpertInfo { + torch::Tensor reduce_weight; + torch::Tensor combine_idx; + torch::Tensor token_count_slice; + std::optional cusum_token_count; + std::optional input_scale; + }; + + // initial steps for MoE computation, select the experts for each token + torch::Tensor select_experts(const torch::Tensor& hidden_states_2d, + const torch::Tensor& router_logits_2d, + SelectedExpertInfo& selected_expert_info, + bool enable_all2all_communication); + + private: + int64_t num_total_experts_; + int64_t topk_; + int64_t num_expert_group_; + int64_t topk_group_; + double route_scale_; + int64_t hidden_size_; + int64_t n_shared_experts_; + bool is_gated_; + int64_t renormalize_; + std::string hidden_act_; + std::string scoring_func_; + bool is_smoothquant_; + + int64_t num_experts_per_rank_; + int64_t start_expert_id_; + + // Deep EP related parameters + bool enable_deep_ep_; + DeepEPBuffer deep_ep_buffer_; + DeepEPParams deep_ep_params_; + torch::Tensor dispatch_recv_token_tensor_head_; + torch::Tensor dispatch_recv_token_tensor_tail_; + + // steams for parallel shared experts + std::unique_ptr shared_stream_; + std::unique_ptr routed_stream_; + xllm::Device device_; + bool stream_initialized_ = false; + + ReplicatedLinear gate_{nullptr}; + DenseMLP shared_experts_{nullptr}; + DeepEP deep_ep_{nullptr}; + + QuantArgs quant_args_; + ParallelArgs parallel_args_; + torch::TensorOptions options_; + ProcessGroup* tp_pg_; + + DEFINE_WEIGHT(w13); + DEFINE_FUSED_WEIGHT(w1); + DEFINE_FUSED_WEIGHT(w3); + DEFINE_FUSED_WEIGHT(w2); + DEFINE_WEIGHT(e_score_correction_bias); + DEFINE_WEIGHT(w13_scale); + DEFINE_FUSED_WEIGHT(w1_scale); + DEFINE_FUSED_WEIGHT(w3_scale); + DEFINE_FUSED_WEIGHT(w2_scale); + DEFINE_FUSED_WEIGHT(input_smooth); + DEFINE_FUSED_WEIGHT(act_smooth); + + void load_e_score_correction_bias(const StateDict& state_dict); + void load_experts(const StateDict& state_dict); + // create the group gemm output tensor with the workspace + torch::Tensor create_group_gemm_output(const torch::Tensor& a, + const torch::Tensor& b, + const torch::Tensor& group_list, + torch::ScalarType dtype, + torch::Tensor& workspace); +}; +TORCH_MODULE(FusedMoE); + +} // namespace layer +} // namespace xllm diff --git a/ex_engine/python/corex_gdn.py b/ex_engine/python/corex_gdn.py index f797f5d8..120c53d9 100644 --- a/ex_engine/python/corex_gdn.py +++ b/ex_engine/python/corex_gdn.py @@ -82,18 +82,33 @@ class CoreXGDN: def __init__( self, - num_heads: int, - head_dim: int, + num_heads: int = 0, + head_dim: int = 128, layer_idx: int = 0, chunk_size: int = 16, eps: float = 1e-6, + # kwargs from qwen3_5.py (GatedDeltaNet uses separate k/v dims) + num_v_heads: int = 0, + num_k_heads: int = 0, + head_k_dim: int = 0, + head_v_dim: int = 0, + conv_kernel_size: int = 4, + **kwargs, # future-proof ): - self.num_heads = num_heads - self.head_dim = head_dim + # Accept both calling conventions: + # CoreXGDN(num_heads, head_dim) — simple + # CoreXGDN(num_v_heads=.., num_k_heads=.., head_k_dim=.., head_v_dim=..) — from qwen3_5.py + self.num_v_heads = num_v_heads or num_heads + self.num_k_heads = num_k_heads or num_heads + self.head_k_dim = head_k_dim or head_dim + self.head_v_dim = head_v_dim or head_dim + self.num_heads = self.num_v_heads + self.head_dim = self.head_k_dim self.layer_idx = layer_idx self.chunk_size = chunk_size self.eps = eps - self.scale = head_dim ** -0.5 + self.conv_kernel_size = conv_kernel_size + self.scale = self.head_k_dim ** -0.5 self._decode_warned = False self._prefill_warned = False @@ -106,20 +121,68 @@ class CoreXGDN: def forward( self, - q: torch.Tensor, - k: torch.Tensor, - v: torch.Tensor, - gate: torch.Tensor, - beta: torch.Tensor, - conv_state: Optional[torch.Tensor], - temporal_state: Optional[torch.Tensor], + hidden_states: torch.Tensor, attn_metadata, - ) -> Tuple[torch.Tensor, Optional[torch.Tensor]]: + conv_state: torch.Tensor, + temporal_state: torch.Tensor, + in_proj_qkv, # nn.Module — projects hidden → conv_dim + in_proj_z, # nn.Module — projects hidden → val_dim + in_proj_b, # nn.Module — projects hidden → num_v_heads (beta) + in_proj_a, # nn.Module — projects hidden → num_v_heads (alpha/dt) + conv1d_weight, # (conv_dim, 1, kernel_size) depthwise conv weight + A_log, # (num_v_heads,) log decay parameters + dt_bias, # (num_v_heads,) dt bias + norm, # GatedRMSNorm module + out_proj, # RowParallelLinear + ) -> torch.Tensor: + """Full GDN layer forward — matches qwen3_5.py calling convention. + + This mirrors the PyTorch _pytorch_forward() path but uses ixformer + matmul acceleration and fused CoreX GDN ops when available. + """ + from vllm.model_executor.parallel_utils.communication_op import ( + tensor_model_parallel_all_reduce, + ) + try: + from vllm.distributed import get_tensor_model_parallel_world_size + except ImportError: + get_tensor_model_parallel_world_size = lambda: 1 + + tp_size = get_tensor_model_parallel_world_size() + local_key_dim = self.num_k_heads * self.head_k_dim // tp_size + local_val_dim = self.num_v_heads * self.head_v_dim // tp_size + local_num_v = self.num_v_heads + local_num_k = self.num_k_heads + local_conv_dim = local_key_dim * 2 + local_val_dim + is_prefill = getattr(attn_metadata, 'num_prefill_tokens', 0) > 0 + + # Project all tokens at once + mixed_qkv_all, _ = in_proj_qkv(hidden_states) + z_all, _ = in_proj_z(hidden_states) + b_all, _ = in_proj_b(hidden_states) + a_all, _ = in_proj_a(hidden_states) + if is_prefill: - return self._prefill(q, k, v, gate, beta, temporal_state) + if not self._prefill_warned: + logger.info("Using fused CoreX GDN prefill operator") + self._prefill_warned = True + return self._full_prefill( + hidden_states, attn_metadata, conv_state, temporal_state, + mixed_qkv_all, z_all, b_all, a_all, + conv1d_weight, A_log, dt_bias, norm, out_proj, + local_key_dim, local_val_dim, local_num_v, local_num_k, + local_conv_dim) else: - return self._decode(q, k, v, gate, beta, conv_state, temporal_state) + if not self._decode_warned: + logger.info("Using fused CoreX GDN decode operator") + self._decode_warned = True + return self._full_decode( + hidden_states, attn_metadata, conv_state, temporal_state, + mixed_qkv_all, z_all, b_all, a_all, + conv1d_weight, A_log, dt_bias, norm, out_proj, + local_key_dim, local_val_dim, local_num_v, local_num_k, + local_conv_dim) def _prefill(self, q, k, v, gate, beta, temporal_state): if not self._prefill_warned: diff --git a/qwen3_6_scripts/patch_ops.sh b/qwen3_6_scripts/patch_ops.sh index 3e68a4a1..24953aa3 100755 --- a/qwen3_6_scripts/patch_ops.sh +++ b/qwen3_6_scripts/patch_ops.sh @@ -180,17 +180,34 @@ if [ -n "$VLLM2" ]; then cp ./chat_utils.py "$VLLM2/entrypoints/chat_utils.py" 2>/dev/null || true fi -# Deploy corex_gdn.py + corex_moe.py → vllm model_executor/models/ -# These provide the fused GDN prefill kernel and MoE pipeline that competitor 168 had -if [ -f "/workspace/ex_engine/python/corex_gdn.py" ]; then +# corex_gdn.py + corex_moe.py: DO NOT overwrite base image originals! +# Comp 168 log proves: base image's corex_gdn.py loads libcorex_gdn.so and works. +# Our overwrite breaks the interface (CoreXGDN.__init__ signature mismatch). +# Only deploy ours if base has NO corex modules at all. +if [ ! -f "$VLLM/model_executor/models/corex_gdn.py" ]; then cp "/workspace/ex_engine/python/corex_gdn.py" "$VLLM/model_executor/models/corex_gdn.py" 2>/dev/null || true + echo "[patch_ops] corex_gdn.py deployed (base had none)" +fi +if [ ! -f "$VLLM/model_executor/models/corex_moe.py" ]; then cp "/workspace/ex_engine/python/corex_moe.py" "$VLLM/model_executor/models/corex_moe.py" 2>/dev/null || true - echo "[patch_ops] Deployed: corex_gdn.py + corex_moe.py → $VLLM/model_executor/models/" - if [ -n "$VLLM2" ]; then - cp "/workspace/ex_engine/python/corex_gdn.py" "$VLLM2/model_executor/models/corex_gdn.py" 2>/dev/null || true - cp "/workspace/ex_engine/python/corex_moe.py" "$VLLM2/model_executor/models/corex_moe.py" 2>/dev/null || true + echo "[patch_ops] corex_moe.py deployed (base had none)" +fi +# corex_fa2.py: deploy if base doesn't have it +# Comp 168 log: corex_fa2.py provides FA2 packed/paged/chunked dispatch +if [ ! -f "$VLLM/model_executor/models/corex_fa2.py" ]; then + if [ -f "/workspace/ex_engine/python/corex_fa2.py" ]; then + cp "/workspace/ex_engine/python/corex_fa2.py" "$VLLM/model_executor/models/corex_fa2.py" 2>/dev/null || true + echo "[patch_ops] corex_fa2.py deployed (base had none)" fi fi +echo "[patch_ops] CoreX modules: preserved base originals where they exist" +if [ -n "$VLLM2" ]; then + for _CM in corex_gdn.py corex_moe.py corex_fa2.py; do + if [ -f "$VLLM/model_executor/models/$_CM" ] && [ ! -f "$VLLM2/model_executor/models/$_CM" ]; then + cp "$VLLM/model_executor/models/$_CM" "$VLLM2/model_executor/models/$_CM" 2>/dev/null || true + fi + done +fi # Deploy EX Engine Python module + C++ bridge into vllm importable path EX_ENGINE_SRC="/workspace/ex_engine" diff --git a/qwen3_6_scripts/qwen3_5.py b/qwen3_6_scripts/qwen3_5.py index a07e0613..bb3b35b9 100644 --- a/qwen3_6_scripts/qwen3_5.py +++ b/qwen3_6_scripts/qwen3_5.py @@ -463,6 +463,7 @@ class GatedDeltaNet(nn.Module): self._use_corex_gdn = False if _corex_gdn_available and _corex_gdn_module is not None: try: + # Try base image's CoreXGDN signature first (may differ from ours) self._corex_gdn_obj = _corex_gdn_module.CoreXGDN( num_v_heads=self.num_v_heads // tp_size, num_k_heads=self.num_k_heads // tp_size, @@ -472,11 +473,25 @@ class GatedDeltaNet(nn.Module): layer_idx=layer_idx, ) self._use_corex_gdn = True - logger.info("GatedDeltaNet layer %d: CoreX fused GDN enabled", layer_idx) + except TypeError: + # Fallback: simpler signature + try: + self._corex_gdn_obj = _corex_gdn_module.CoreXGDN( + self.num_v_heads // tp_size, + self.head_k_dim, + layer_idx=layer_idx, + ) + self._use_corex_gdn = True + except Exception as e2: + logger.warning( + "GatedDeltaNet layer %d: CoreX GDN init failed (%s), PyTorch", + layer_idx, e2) except Exception as e: logger.warning( - "GatedDeltaNet layer %d: CoreX GDN init failed (%s), using PyTorch", + "GatedDeltaNet layer %d: CoreX GDN init failed (%s), PyTorch", layer_idx, e) + if self._use_corex_gdn and layer_idx == 0: + logger.info("GatedDeltaNet: CoreX fused GDN enabled") def _conv1d_weight_loader(self, param: torch.Tensor, loaded_weight: torch.Tensor) -> None: