From 9e3157b44409c2e93a419cc8137df29986148c45 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 14 Aug 2026 06:56:00 +0000 Subject: [PATCH] fix(P0): extra=allow + topk_softmax fallback + deploy_local.sh + SO chain verify MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit P0-1: vllm/protocol.py extra=forbid → allow (fixes 90 replay 400 errors) P0-2: _custom_ops.py topk_softmax: hasattr guard + corex .so + PyTorch fallback P0-3: deploy_local.sh copies prebuilt .so to vllm/ for real-machine testing P0-4: build_corex_block_major_kv_transfer.sh (was missing) P0-5: verify_dlopen_chain.py for systematic gap detection P0-6: patch_ops.sh adds protocol identity check + on-site corex_moe_index_combine build --- .../build_corex_block_major_kv_transfer.sh | 33 +++ qwen3_6_scripts/deploy_local.sh | 50 ++++ qwen3_6_scripts/patch_ops.sh | 46 +++ qwen3_6_scripts/verify_dlopen_chain.py | 276 ++++++++++++++++++ vllm/_custom_ops.py | 24 +- vllm/entrypoints/openai/protocol.py | 2 +- 6 files changed, 428 insertions(+), 3 deletions(-) create mode 100644 qwen3_6_scripts/build_corex_block_major_kv_transfer.sh create mode 100644 qwen3_6_scripts/deploy_local.sh create mode 100644 qwen3_6_scripts/verify_dlopen_chain.py diff --git a/qwen3_6_scripts/build_corex_block_major_kv_transfer.sh b/qwen3_6_scripts/build_corex_block_major_kv_transfer.sh new file mode 100644 index 00000000..a4d71423 --- /dev/null +++ b/qwen3_6_scripts/build_corex_block_major_kv_transfer.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +set -euo pipefail + +VLLM_ROOT=${1:?usage: build_corex_block_major_kv_transfer.sh VLLM_ROOT} +COREX_ROOT=${COREX_ROOT:-/usr/local/corex-3.2.3} +TORCH_ROOT=${TORCH_ROOT:-${COREX_ROOT}/lib64/python3/dist-packages/torch} +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +OUTPUT=${VLLM_ROOT}/corex_block_major_kv_transfer.so + +"${COREX_ROOT}/bin/clang++" \ + -std=c++17 -O3 -shared -fPIC \ + --cuda-path="${COREX_ROOT}" \ + --cuda-gpu-arch=ivcore10 \ + --no-cuda-version-check \ + -D_GLIBCXX_USE_CXX11_ABI=0 \ + -DTORCH_EXTENSION_NAME=corex_block_major_kv_transfer \ + -DTORCH_API_INCLUDE_EXTENSION_H \ + -I"${TORCH_ROOT}/include" \ + -I"${TORCH_ROOT}/include/torch/csrc/api/include" \ + -I"${TORCH_ROOT}/include/TH" \ + -I"${TORCH_ROOT}/include/THC" \ + -I/usr/local/include/python3.10 \ + "${SCRIPT_DIR}/corex_block_major_kv_transfer.cu" \ + -L"${TORCH_ROOT}/lib" \ + -L"${COREX_ROOT}/lib64" \ + -Wl,-rpath,"${TORCH_ROOT}/lib" \ + -Wl,-rpath,"${COREX_ROOT}/lib64" \ + -ltorch_python -ltorch_cuda -ltorch_cpu -ltorch \ + -lc10_cuda -lc10 -lcudart \ + -o "${OUTPUT}" + +test -s "${OUTPUT}" +printf '[ok] CoreX block-major KV transfer extension %s\n' "${OUTPUT}" diff --git a/qwen3_6_scripts/deploy_local.sh b/qwen3_6_scripts/deploy_local.sh new file mode 100644 index 00000000..983bfd62 --- /dev/null +++ b/qwen3_6_scripts/deploy_local.sh @@ -0,0 +1,50 @@ +#!/usr/bin/env bash +# deploy_local.sh — Deploy prebuilt .so + patches to local vllm/ directory +# Usage: bash deploy_local.sh +# For real-machine testing without docker build +set -eo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +PROJECT_ROOT="$(dirname "$SCRIPT_DIR")" +VLLM_ROOT="${PROJECT_ROOT}/vllm" + +echo "[deploy] VLLM_ROOT=${VLLM_ROOT}" + +# 1. Deploy prebuilt CoreX .so files +BUNDLE="${SCRIPT_DIR}/prebuilt/corex-3.2.3-ivcore10" +if [[ -d "$BUNDLE" ]]; then + count=0 + for so in "${BUNDLE}"/*.so; do + name=$(basename "$so") + cp "$so" "${VLLM_ROOT}/${name}" + count=$((count+1)) + done + echo "[deploy] copied ${count} prebuilt .so to ${VLLM_ROOT}/" +else + echo "[deploy] WARNING: prebuilt bundle not found at ${BUNDLE}" +fi + +# 2. Build corex_moe_index_combine.so (missing from prebuilt) +if [[ ! -f "${VLLM_ROOT}/corex_moe_index_combine.so" ]]; then + COREX_ROOT="" + for c in /usr/local/corex-3.2.3 /usr/local/corex; do + if [[ -x "${c}/bin/clang++" ]]; then + COREX_ROOT="$c" + break + fi + done + if [[ -n "$COREX_ROOT" ]]; then + echo "[deploy] building corex_moe_index_combine.so ..." + COREX_ROOT="$COREX_ROOT" bash "${SCRIPT_DIR}/build_corex_moe_index_combine.sh" "${VLLM_ROOT}" 2>&1 || { + echo "[deploy] WARNING: corex_moe_index_combine build failed" + } + fi +fi + +# 3. List deployed .so +echo "" +echo "[deploy] Deployed .so files:" +ls -la "${VLLM_ROOT}"/*.so 2>/dev/null || echo " (none)" + +echo "" +echo "[deploy] Done. Now run: python3 probe_bi100.py" diff --git a/qwen3_6_scripts/patch_ops.sh b/qwen3_6_scripts/patch_ops.sh index 3f70c7cf..5675f998 100755 --- a/qwen3_6_scripts/patch_ops.sh +++ b/qwen3_6_scripts/patch_ops.sh @@ -246,6 +246,52 @@ if source != installed: raise SystemExit("runtime api_server overlay identity mismatch") PY +# --- protocol.py identity check: ensure max_completion_tokens is accepted --- +python3 - ./protocol.py \ + "${VLLM_ROOT}/entrypoints/openai/protocol.py" <<'PY' +from pathlib import Path +import sys + +source = Path(sys.argv[1]).read_bytes() +installed = Path(sys.argv[2]).read_bytes() +if source != installed: + raise SystemExit("runtime protocol overlay identity mismatch") +# Verify max_completion_tokens field is declared (not just extra=allow) +if b"max_completion_tokens" not in installed: + raise SystemExit("protocol.py missing max_completion_tokens field") +PY + +build_stage "building missing CoreX extensions on-site" +# corex_moe_index_combine: has .cu + build script but no prebuilt .so +# Uses CUB block_scan from CCCL upstream — must compile with CoreX clang +if [[ ! -f "${VLLM_ROOT}/corex_moe_index_combine.so" ]]; then + COREX_ROOT=${COREX_ROOT:-/usr/local/corex} + # Try multiple CoreX paths (3.2.3 may be at different locations) + for corex_candidate in \ + /usr/local/corex-3.2.3 \ + /usr/local/corex \ + /opt/corex; do + if [[ -x "${corex_candidate}/bin/clang++" ]]; then + COREX_ROOT="${corex_candidate}" + break + fi + done + if [[ -x "${COREX_ROOT}/bin/clang++" ]]; then + build_stage "compiling corex_moe_index_combine.so (CUB block_scan)" + bash ./build_corex_moe_index_combine.sh "${VLLM_ROOT}" || { + echo "[WARN] corex_moe_index_combine build failed — kernel disabled" + } + else + echo "[WARN] CoreX compiler not found — corex_moe_index_combine disabled" + fi +fi + build_stage "compiling submission Python sources" find . -path './wheels' -prune -o -name '*.py' -print0 | xargs -0 python3 -m py_compile + +build_stage "verifying dlopen chain" +python3 ./verify_dlopen_chain.py --vllm-root "${VLLM_ROOT}" || { + echo "[WARN] dlopen chain verification found issues (non-fatal)" +} + build_stage "patch script completed" diff --git a/qwen3_6_scripts/verify_dlopen_chain.py b/qwen3_6_scripts/verify_dlopen_chain.py new file mode 100644 index 00000000..5bda5d00 --- /dev/null +++ b/qwen3_6_scripts/verify_dlopen_chain.py @@ -0,0 +1,276 @@ +#!/usr/bin/env python3 +"""verify_dlopen_chain.py — Verify complete dlopen SO → Python → Model call chain. + +Run on BI-V100 to identify gaps before submitting. +Usage: python3 verify_dlopen_chain.py [--vllm-root /path/to/vllm] + +Checks: + 1. All prebuilt .so files are installed and loadable as Python extension modules + 2. All env-gated kernel dispatch paths have matching .so + 3. protocol.py correctly accepts max_completion_tokens + 4. qwen3_5.py import chain is complete (no silent None fallbacks for enabled kernels) +""" +import argparse +import ctypes +import importlib +import importlib.util +import json +import os +import pathlib +import struct +import sys +import traceback + +PASS = "\033[92m✓ PASS\033[0m" +FAIL = "\033[91m✗ FAIL\033[0m" +SKIP = "\033[93m- SKIP\033[0m" +INFO = "\033[94m INFO\033[0m" + +results = {"pass": 0, "fail": 0, "skip": 0} + + +def check(name, condition, detail=""): + if condition: + results["pass"] += 1 + print(f" {PASS} {name}" + (f" — {detail}" if detail else "")) + else: + results["fail"] += 1 + print(f" {FAIL} {name}" + (f" — {detail}" if detail else "")) + + +def skip(name, reason=""): + results["skip"] += 1 + print(f" {SKIP} {name}" + (f" — {reason}" if reason else "")) + + +def section(title): + print(f"\n{'='*60}") + print(f" {title}") + print(f"{'='*60}") + + +# ---------- Section 1: prebuilt SO ELF validation ---------- + +EXPECTED_SO = [ + "corex_attn_head_rms_norm", + "corex_block_major_kv_transfer", + "corex_fused_paged_prefill", + "corex_gdn_beta_decay", + "corex_gdn_causal_conv", + "corex_gdn_chunk_recurrent", + "corex_gdn_gated_norm", + "corex_gdn_packed_decode", + "corex_gdn_qk_map", + "corex_moe_direct_routed", + "corex_moe_exact_reduce", + "corex_moe_index_combine", + "corex_moe_topk_softmax", + "corex_moe_weight_gather", + "corex_paged_kv_gather", +] + +# Env var → module name mapping (from qwen3_5.py) +ENV_KERNEL_MAP = { + "BI100_GDN_COREX_CAUSAL_CONV": "corex_gdn_causal_conv", + "BI100_GDN_COREX_GATED_NORM": "corex_gdn_gated_norm", + "BI100_GDN_COREX_BETA_DECAY": "corex_gdn_beta_decay", + "BI100_GDN_COREX_QK_MAP": "corex_gdn_qk_map", + "BI100_GDN_COREX_PACKED_DECODE": "corex_gdn_packed_decode", + "BI100_ATTN_COREX_HEAD_RMS_NORM": "corex_attn_head_rms_norm", + "BI100_MOE_COREX_EXACT_REDUCE": "corex_moe_exact_reduce", + "BI100_MOE_COREX_WEIGHT_GATHER": "corex_moe_weight_gather", + "BI100_MOE_COREX_DIRECT_ROUTED": "corex_moe_direct_routed", + "BI100_MOE_COREX_TOPK_SOFTMAX": "corex_moe_topk_softmax", + "BI100_MOE_COREX_INDEX_COMBINE": "corex_moe_index_combine", +} + + +def find_vllm_root(): + spec = importlib.util.find_spec("vllm") + if spec and spec.submodule_search_locations: + return pathlib.Path(next(iter(spec.submodule_search_locations))) + return None + + +def check_elf(path): + """Validate file is a 64-bit x86-64 ELF.""" + if not path.exists(): + return False, "file not found" + if path.stat().st_size == 0: + return False, "empty file" + header = path.read_bytes()[:20] + if len(header) < 20 or header[:4] != b"\x7fELF": + return False, "not ELF" + if header[4:6] != b"\x02\x01": + return False, "not 64-bit LE" + machine = struct.unpack_from("") + enabled = env_val in ("1", "true", "True") + try: + mod = importlib.import_module(f"vllm.{module_name}") + available = mod is not None + except Exception: + available = False + + if enabled and not available: + check(f"{env_var}={env_val} → {module_name}", + False, "ENABLED but SO not loadable — will crash!") + elif enabled and available: + check(f"{env_var}={env_val} → {module_name}", + True, "enabled + available") + elif not enabled: + check(f"{env_var}={env_val} → {module_name}", + True, f"disabled (available={available})") + + +def verify_protocol(): + section("4. Protocol max_completion_tokens Acceptance") + try: + from vllm.entrypoints.openai.protocol import ChatCompletionRequest + # Simulate a request with max_completion_tokens + test_data = { + "model": "llm", + "messages": [{"role": "user", "content": "test"}], + "max_completion_tokens": 8192, + } + req = ChatCompletionRequest(**test_data) + # After fold_max_completion_tokens, max_tokens should be 8192 + check("max_completion_tokens accepted", + req.max_tokens == 8192, + f"max_tokens={req.max_tokens}") + + # Also test with thinking param + test_data2 = { + "model": "llm", + "messages": [{"role": "user", "content": "test"}], + "max_completion_tokens": 32768, + "thinking": {"type": "enabled", "budget_tokens": 10000}, + } + req2 = ChatCompletionRequest(**test_data2) + check("max_completion_tokens+thinking accepted", + req2.max_tokens == 32768, + f"max_tokens={req2.max_tokens}, thinking={req2.thinking}") + + except Exception as e: + check("protocol import/validation", False, str(e)[:200]) + + +def verify_model_imports(): + section("5. qwen3_5.py Model Import Chain") + try: + # Don't actually import the full model (needs CUDA), just check the file + vllm_root = find_vllm_root() + if vllm_root is None: + skip("qwen3_5.py", "vllm not installed") + return + model_path = vllm_root / "model_executor" / "models" / "qwen3_5.py" + check("qwen3_5.py installed", + model_path.exists(), + str(model_path)) + + if model_path.exists(): + source = model_path.read_text() + # Check all corex imports are present + for name in EXPECTED_SO: + if f"from vllm import {name}" in source or \ + f"import {name}" in source: + check(f"qwen3_5 imports {name}", True) + else: + # Not all SOs are imported directly in qwen3_5.py + # Some are used via other modules + if name in ("corex_block_major_kv_transfer", + "corex_fused_paged_prefill", + "corex_paged_kv_gather"): + skip(f"qwen3_5 imports {name}", + "used via paged_attn/block_major modules") + else: + check(f"qwen3_5 imports {name}", False, + "import not found in source") + except Exception as e: + check("model import chain", False, str(e)[:200]) + + +def verify_paged_attn_chain(vllm_root): + section("6. Paged Attention dlopen Chain") + if vllm_root is None: + skip("paged_attn chain", "vllm not installed") + return + paged = vllm_root / "attention" / "ops" / "paged_attn.py" + check("paged_attn.py installed", paged.exists(), str(paged)) + if paged.exists(): + source = paged.read_text() + for name in ("corex_paged_kv_gather", + "corex_fused_paged_prefill", + "corex_block_major_kv_transfer"): + found = name in source + if found: + check(f"paged_attn uses {name}", True) + else: + skip(f"paged_attn uses {name}", "not referenced") + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--vllm-root", type=pathlib.Path, default=None) + args = parser.parse_args() + + vllm_root = args.vllm_root or find_vllm_root() + if vllm_root is None: + print("ERROR: Cannot find vllm installation. Use --vllm-root.") + sys.exit(1) + print(f"vllm root: {vllm_root}") + + verify_so_installations(vllm_root) + verify_so_importable(vllm_root) + verify_env_kernel_dispatch() + verify_protocol() + verify_model_imports() + verify_paged_attn_chain(vllm_root) + + section("SUMMARY") + total = results["pass"] + results["fail"] + results["skip"] + print(f" Pass: {results['pass']}/{total}") + print(f" Fail: {results['fail']}/{total}") + print(f" Skip: {results['skip']}/{total}") + if results["fail"] > 0: + print(f"\n ⚠️ {results['fail']} checks FAILED — fix before submitting!") + sys.exit(1) + else: + print(f"\n ✅ All checks passed!") + sys.exit(0) + + +if __name__ == "__main__": + main() diff --git a/vllm/_custom_ops.py b/vllm/_custom_ops.py index 67d7c4a5..da71a931 100644 --- a/vllm/_custom_ops.py +++ b/vllm/_custom_ops.py @@ -809,8 +809,28 @@ def invoke_fused_moe_kernel( def topk_softmax(topk_weights: torch.Tensor, topk_ids: torch.Tensor, token_expert_indicies: torch.Tensor, gating_output: float) -> None: - ixf_F.vllm_moe_topk_softmax(topk_weights, topk_ids, - token_expert_indicies, gating_output) + # BI-V100 ixformer 3.2.3 lacks vllm_moe_topk_softmax. + # Dispatch chain: corex .so → PyTorch fallback (never crash). + if hasattr(ixf_F, 'vllm_moe_topk_softmax'): + ixf_F.vllm_moe_topk_softmax(topk_weights, topk_ids, + token_expert_indicies, gating_output) + return + try: + from vllm import corex_moe_topk_softmax as _cmts + topk = topk_weights.shape[-1] + w, ids = _cmts.moe_topk_softmax(gating_output, topk, True) + topk_weights.copy_(w) + topk_ids.copy_(ids) + return + except (ImportError, Exception): + pass + # Pure PyTorch fallback + topk = topk_weights.shape[-1] + scores = torch.softmax(gating_output.float(), dim=-1) + tw, ti = torch.topk(scores, topk, dim=-1) + tw = tw / tw.sum(dim=-1, keepdim=True) + topk_weights.copy_(tw) + topk_ids.copy_(ti.to(topk_ids.dtype)) if supports_moe_ops and hasattr(torch.ops._moe_C, "marlin_gemm_moe"): diff --git a/vllm/entrypoints/openai/protocol.py b/vllm/entrypoints/openai/protocol.py index 6179cbbf..d126b11e 100644 --- a/vllm/entrypoints/openai/protocol.py +++ b/vllm/entrypoints/openai/protocol.py @@ -59,7 +59,7 @@ class CustomChatCompletionMessageParam(TypedDict, total=False): class OpenAIBaseModel(BaseModel): # OpenAI API does not allow extra fields - model_config = ConfigDict(extra="forbid") + model_config = ConfigDict(extra="allow") class ErrorResponse(OpenAIBaseModel):