[Bugfix] fix custom op GmmSwigluQuantWeightNzTensorList (#4593)

### What this PR does / why we need it? 1. Fixes the environment path used to locate custom op shared libraries. 2. Uses empty tensor initialization for op outputs instead of zero-initialization for better efficiency. - vLLM version: v0.11.2 - vLLM main: https://github.com/vllm-project/vllm/commit/v0.11.2 --------- Signed-off-by: QianChenxi <chenxi.qian.cq@outlook.com>
2025-12-02 22:02:04 +08:00
parent b84c9afbf5
commit 4588cdac02
6 changed files with 18 additions and 27 deletions
--- a/csrc/CMakeLists.txt
+++ b/csrc/CMakeLists.txt
@@ -15,7 +15,7 @@ option(BUILD_OPEN_PROJECT         "Build open ascend ops project."  ON)
 option(ENABLE_CCACHE              "Enable ccache capability"        ON)
 set(ASCEND_COMPUTE_UNIT           "ascend910b"                    CACHE   STRING   "soc that need to be compiled")
 set(ASCEND_OP_NAME                "ALL"                           CACHE   STRING   "operators that need to be compiled")
-set(VENDOR_NAME                   "customize"                     CACHE   STRING   "vendor name")
+set(VENDOR_NAME                   "vllm-ascend"                   CACHE   STRING   "vendor name")

 include(cmake/config.cmake)
 include(cmake/func.cmake)
--- a/csrc/build_aclnn.sh
+++ b/csrc/build_aclnn.sh
@@ -31,4 +31,3 @@ bash build.sh -n $CUSTOM_OPS -c $SOC_ARG

 # install custom ops to vllm_ascend/_cann_ops_custom
 ./output/CANN-custom_ops*.run --install-path=$ROOT_DIR/vllm_ascend/_cann_ops_custom
-source $ROOT_DIR/vllm_ascend/_cann_ops_custom/vendors/customize/bin/set_env.bash
--- a/csrc/torch_binding.cpp
+++ b/csrc/torch_binding.cpp
@@ -568,9 +568,9 @@ std::tuple<at::Tensor, at::Tensor, at::Tensor> grouped_matmul_swiglu_quant_weigh
    int m = x_size[0];
    int k = x_size[1];

-    at::Tensor output = at::zeros({m, n/2}, x.options().dtype(at::kChar));
-    at::Tensor output_scale = at::zeros({m}, x.options().dtype(at::kFloat));
-    at::Tensor output_offset = at::zeros({m}, x.options().dtype(at::kFloat));
+    at::Tensor output = at::empty({m, n/2}, x.options().dtype(at::kChar));
+    at::Tensor output_scale = at::empty({m}, x.options().dtype(at::kFloat));
+    at::Tensor output_offset = at::empty({m}, x.options().dtype(at::kFloat));

    EXEC_NPU_CMD(
        aclnnGroupedMatmulSwigluQuantWeightNzTensorList,