init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -0,0 +1 @@

View File

@@ -0,0 +1,128 @@
import json
import logging
from pathlib import Path
from typing import Any
logger = logging.getLogger(__name__)
PORT_ENV_KEYS = {"SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"}
INFRA_ENV_KEYS = {
"HCCL_IF_IP",
"HCCL_SOCKET_IFNAME",
"GLOO_SOCKET_IFNAME",
"TP_SOCKET_IFNAME",
"LOCAL_IP",
"NIC_NAME",
"MASTER_IP",
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
}
PERF_METRIC_RENAME: dict[str, str] = {
"Benchmark Duration": "Benchmark_Duration(BD)",
"Prefill Token Throughput": "Prefill_Token_Throughput(PTT)",
"Input Token Throughput": "Input_Token_Throughput(ITT)",
"Output Token Throughput": "Output_Token_Throughput(OTT)",
"Total Token Throughput": "Total_Token_Throughput(TTT)",
}
def extract_hardware(runner: str) -> str:
runner_lower = runner.lower()
for label in ("a3", "a2"):
if label in runner_lower:
return label.upper()
return runner
def get_vllm_version() -> str:
try:
import vllm
return vllm.__version__
except Exception:
return ""
def task_passed(case_config: dict[str, Any], result: Any) -> bool:
if result == "":
return False
case_type = case_config.get("case_type")
baseline = case_config.get("baseline")
threshold = case_config.get("threshold")
if baseline is None or threshold is None:
return True
if case_type == "accuracy" and isinstance(result, (int, float)):
return abs(float(result) - float(baseline)) <= float(threshold)
if case_type == "performance" and isinstance(result, list) and len(result) == 2:
_, result_json = result
throughput_str = result_json.get("Output Token Throughput", {}).get("total", "")
try:
throughput_val = float(throughput_str.replace("token/s", "").strip())
return throughput_val >= float(threshold) * float(baseline)
except (ValueError, AttributeError):
return False
return True
def build_task_entry(case_key: str, case_config: dict[str, Any], result: Any) -> dict[str, Any]:
dataset_path = case_config.get("dataset_path", "")
dataset_conf = case_config.get("dataset_conf", "")
if dataset_path:
task_name = dataset_path.split("/", 1)[-1]
elif dataset_conf:
task_name = dataset_conf.split("/")[0]
else:
task_name = case_key
case_type = case_config.get("case_type", "unknown")
metrics: dict[str, float] = {}
if result == "":
pass
elif case_type == "accuracy" and isinstance(result, (int, float)):
metrics["accuracy"] = round(float(result), 4)
elif case_type == "performance" and isinstance(result, list) and len(result) == 2:
_, result_json = result
for metric_name, metric_data in result_json.items():
if not isinstance(metric_data, dict):
continue
total_str = metric_data.get("total", "")
try:
value = float(total_str.replace("token/s", "").replace("ms", "").replace("s", "").strip())
metrics[PERF_METRIC_RENAME.get(metric_name, metric_name)] = round(value, 4)
except (ValueError, AttributeError):
pass
test_input_keys = ("num_prompts", "max_out_len", "batch_size", "request_rate")
test_input = {key: case_config[key] for key in test_input_keys if key in case_config}
target: dict[str, Any] = {}
if case_config.get("baseline") is not None:
target["baseline"] = case_config["baseline"]
if case_config.get("threshold") is not None:
target["threshold"] = case_config["threshold"]
entry: dict[str, Any] = {"name": task_name, "metrics": metrics, "test_input": test_input}
if target:
entry["target"] = target
entry["pass_fail"] = "pass" if task_passed(case_config, result) else "fail"
return entry
def filter_environment(envs: dict[str, Any]) -> dict[str, Any]:
exclude = PORT_ENV_KEYS | INFRA_ENV_KEYS
return {key: value for key, value in envs.items() if key not in exclude}
def write_results_json(
output: dict[str, Any],
*,
job_name: str,
output_dir: Path | None = None,
) -> Path:
if output_dir is None:
output_dir = Path("/root/.cache/benchmark_results") / job_name
output_dir.mkdir(parents=True, exist_ok=True)
output_path = output_dir / f"{job_name}.json"
output_path.write_text(json.dumps(output, indent=2, ensure_ascii=False), encoding="utf-8")
logger.info("Benchmark results saved to PVC at %s", output_path)
print(f"Benchmark results saved to PVC at {output_path}")
return output_path

View File

@@ -0,0 +1,174 @@
apiVersion: leaderworkerset.x-k8s.io/v1
kind: LeaderWorkerSet
metadata:
name: {{ lws_name | default("vllm") }}
namespace: vllm-project
spec:
replicas: {{ replicas | default(1) }}
leaderWorkerTemplate:
size: {{ size | default(2) }}
restartPolicy: None
leaderTemplate:
metadata:
labels:
role: leader
spec:
tolerations:
- key: "dedicated"
operator: "Equal"
value: "night"
effect: "NoSchedule"
containers:
- name: vllm-leader
imagePullPolicy: Always
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
env:
- name: CONFIG_YAML_PATH
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
- name: CONFIG_BASE_PATH
value: "{{ config_base_path | default("") }}"
- name: LOG_PREFIX
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
- name: WORKSPACE
value: "/vllm-workspace"
- name: FAIL_TAG
value: {{ fail_tag | default("FAIL_TAG") }}
- name: IS_PR_TEST
value: "{{ is_pr_test | default("false") }}"
- name: VLLM_ASCEND_REF
value: {{ vllm_ascend_ref | default("main") }}
- name: VLLM_ASCEND_REMOTE_URL
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
- name: BENCHMARK_JOB_NAME
value: {{ benchmark_job_name | default("") }}
- name: VLLM_CI_RUNNER
value: {{ runner | default("linux-aarch64-a3-0") }}
- name: VLLM_ASCEND_VERSION
value: {{ vllm_ascend_ref | default("main") }}
- name: AOP_MULTI_ENABLED
value: "{{ aop_multi_enabled }}"
- name: GOOD_TABLE
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
command:
- sh
- -c
- |
bash /root/.cache/tests/run.sh
resources:
limits:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
memory: 512Gi
ephemeral-storage: 100Gi
requests:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
ephemeral-storage: 100Gi
cpu: 125
ports:
- containerPort: 8080
# readinessProbe:
# tcpSocket:
# port: 8080
# initialDelaySeconds: 15
# periodSeconds: 10
volumeMounts:
- mountPath: /root/.cache
name: shared-volume
- mountPath: /usr/local/Ascend/driver/tools
name: driver-tools
- mountPath: /dev/shm
name: dshm
volumes:
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 512Gi
- name: shared-volume
persistentVolumeClaim:
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
- name: driver-tools
hostPath:
path: /usr/local/Ascend/driver/tools
workerTemplate:
spec:
tolerations:
- key: "dedicated"
operator: "Equal"
value: "night"
effect: "NoSchedule"
containers:
- name: vllm-worker
imagePullPolicy: Always
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
env:
- name: CONFIG_YAML_PATH
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
- name: CONFIG_BASE_PATH
value: "{{ config_base_path | default("") }}"
- name: LOG_PREFIX
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
- name: WORKSPACE
value: "/vllm-workspace"
- name: FAIL_TAG
value: {{ fail_tag | default("FAIL_TAG") }}
- name: IS_PR_TEST
value: "{{ is_pr_test | default("false") }}"
- name: VLLM_ASCEND_REF
value: {{ vllm_ascend_ref | default("main") }}
- name: VLLM_ASCEND_REMOTE_URL
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
- name: BENCHMARK_JOB_NAME
value: {{ benchmark_job_name | default("") }}
- name: VLLM_CI_RUNNER
value: {{ runner | default("linux-aarch64-a3-0") }}
- name: AOP_MULTI_ENABLED
value: "{{ aop_multi_enabled }}"
- name: GOOD_TABLE
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
command:
- sh
- -c
- |
bash /root/.cache/tests/run.sh
resources:
limits:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
memory: 512Gi
ephemeral-storage: 100Gi
requests:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
ephemeral-storage: 100Gi
cpu: 125
volumeMounts:
- mountPath: /root/.cache
name: shared-volume
- mountPath: /usr/local/Ascend/driver/tools
name: driver-tools
- mountPath: /dev/shm
name: dshm
volumes:
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 512Gi
- name: shared-volume
persistentVolumeClaim:
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
- name: driver-tools
hostPath:
path: /usr/local/Ascend/driver/tools
---
apiVersion: v1
kind: Service
metadata:
name: {{ lws_name | default("vllm") }}-leader
namespace: vllm-project
spec:
ports:
- name: http
port: 8080
protocol: TCP
targetPort: 8080
selector:
leaderworkerset.sigs.k8s.io/name: {{ lws_name | default("vllm") }}
role: leader
type: ClusterIP

View File

@@ -0,0 +1,466 @@
#!/bin/bash
set -euo pipefail
# Color definitions
GREEN="\033[0;32m"
BLUE="\033[0;34m"
YELLOW="\033[0;33m"
RED="\033[0;31m"
NC="\033[0m" # No Color
INTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/internal_dp/scripts/test_multi_node.py"
EXTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py"
if [ -z "${MULTI_NODE_TEST_PATH:-}" ]; then
if [[ "${CONFIG_BASE_PATH:-}" == *"external_dp/config"* || "${CONFIG_YAML_PATH:-}" == *"external_dp/config"* ]]; then
MULTI_NODE_TEST_PATH="$EXTERNAL_DP_TEST_PATH"
else
MULTI_NODE_TEST_PATH="$INTERNAL_DP_TEST_PATH"
fi
fi
# Configuration
export LD_LIBRARY_PATH=/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:$LD_LIBRARY_PATH
export LD_LIBRARY_PATH=/usr/local/lib:$LD_LIBRARY_PATH
# cann and atb environment setup
source /usr/local/Ascend/ascend-toolkit/set_env.sh
source /usr/local/Ascend/cann-9.1.0/share/info/ascendnpu-ir/bin/set_env.sh
set +eu
source /usr/local/Ascend/nnal/atb/set_env.sh
set -eu
# Home path for aisbench
export BENCHMARK_HOME=${WORKSPACE}/vllm-ascend/benchmark
# Logging configurations
export VLLM_LOGGING_LEVEL="INFO"
# Reduce glog verbosity for mooncake
export GLOG_minloglevel=1
# Set transformers to offline mode to avoid downloading models during tests
export HF_HUB_OFFLINE="1"
# Default is 600s
export VLLM_ENGINE_READY_TIMEOUT_S=1800
# Function to print section headers
print_section() {
echo -e "\n${BLUE}=== $1 ===${NC}"
}
print_failure() {
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: $1${NC}"
exit 1
}
# Function to print success messages
print_success() {
echo -e "${GREEN}✓ $1${NC}"
}
# Function to print error messages and exit
print_error() {
echo -e "${RED}✗ ERROR: $1${NC}"
exit 1
}
show_vllm_info() {
cd "$WORKSPACE"
echo "Installed vLLM-related Python packages:"
pip list | grep vllm || echo "No vllm packages found."
echo ""
echo "============================"
echo "vLLM Git information"
echo "============================"
cd vllm
if [ -d .git ]; then
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
echo "Commit hash: $(git rev-parse HEAD)"
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
echo "Message: $(git log -1 --pretty=format:'%s')"
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
echo "Remote: $(git remote -v | head -n1)"
echo ""
else
echo "No .git directory found in vllm"
fi
cd ..
echo ""
echo "============================"
echo "vLLM-Ascend Git information"
echo "============================"
cd vllm-ascend
if [ -d .git ]; then
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
echo "Commit hash: $(git rev-parse HEAD)"
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
echo "Message: $(git log -1 --pretty=format:'%s')"
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
echo "Remote: $(git remote -v | head -n1)"
echo ""
else
echo "No .git directory found in vllm-ascend"
fi
cd ..
}
check_npu_info() {
echo "====> Check NPU info"
npu-smi info
cat "/usr/local/Ascend/ascend-toolkit/latest/$(uname -i)-linux/ascend_toolkit_install.info"
}
check_and_config() {
echo "====> Configure mirrors and git proxy"
git config --global url."https://ghfast.top/https://github.com/".insteadOf "https://github.com/"
pip config set global.index-url https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple
export PIP_EXTRA_INDEX_URL="https://mirrors.huaweicloud.com/ascend/repos/pypi"
}
install_extra_components() {
echo "====> Installing extra components for DeepSeek-v3.2-exp-bf16"
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/CANN-custom_ops-sfa-linux.aarch64.run; then
echo "Failed to download CANN-custom_ops-sfa-linux.aarch64.run"
return 1
fi
chmod +x ./CANN-custom_ops-sfa-linux.aarch64.run
./CANN-custom_ops-sfa-linux.aarch64.run --quiet
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/custom_ops-1.0-cp311-cp311-linux_aarch64.whl; then
echo "Failed to download custom_ops wheel"
return 1
fi
pip install custom_ops-1.0-cp311-cp311-linux_aarch64.whl
export ASCEND_CUSTOM_OPP_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize${ASCEND_CUSTOM_OPP_PATH:+:${ASCEND_CUSTOM_OPP_PATH}}"
export LD_LIBRARY_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize/op_api/lib/${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
source /usr/local/Ascend/ascend-toolkit/set_env.sh
rm -f CANN-custom_ops-sfa-linux.aarch64.run \
custom_ops-1.0-cp311-cp311-linux_aarch64.whl
echo "====> Extra components installation completed"
}
checkout_src() {
echo "====> Checkout source code"
mkdir -p "$WORKSPACE"
cd "$WORKSPACE"
pip uninstall -y vllm-ascend || true
cp -r "$WORKSPACE/vllm-ascend/benchmark" /tmp/aisbench-backup || true
rm -rf "$WORKSPACE/vllm-ascend"
if [ ! -d "$WORKSPACE/vllm-ascend" ]; then
echo "Cloning vllm-ascend from $VLLM_ASCEND_REMOTE_URL"
git clone --depth 1 --recurse-submodules "$VLLM_ASCEND_REMOTE_URL" "$WORKSPACE/vllm-ascend"
cd "$WORKSPACE/vllm-ascend"
PR_REF=$(git ls-remote origin 'refs/pull/*/head' | grep "^${VLLM_ASCEND_REF}" | awk '{print $2}' | head -1)
if [ -n "$PR_REF" ]; then
git fetch --depth 1 origin "$PR_REF"
git checkout FETCH_HEAD
else
git fetch origin '+refs/pull/*/head:refs/remotes/pull/*' 2>/dev/null || true
git checkout "$VLLM_ASCEND_REF"
fi
git submodule update --init --recursive
fi
}
install_vllm_ascend() {
echo "====> Install vllm-ascend"
pip install -r "$WORKSPACE/vllm-ascend/requirements-dev.txt"
pip install -e "$WORKSPACE/vllm-ascend"
}
install_aisbench() {
echo "====> Install AISBench benchmark"
BENCH_DIR="$WORKSPACE/vllm-ascend/benchmark"
cp -r /tmp/aisbench-backup "$BENCH_DIR"
cd "$BENCH_DIR"
pip install -e . \
-r requirements/api.txt \
-r requirements/extra.txt
python3 -m pip cache purge || echo "WARNING: pip cache purge failed, but proceeding..."
}
show_triton_ascend_info() {
echo "====> Check triton ascend info"
clang -v
which bishengir-compile
pip show triton-ascend
}
kill_npu_processes() {
pgrep python3 | xargs -r kill -9
pgrep VLLM | xargs -r kill -9
sleep 4
}
run_tests_with_log() {
set +e
kill_npu_processes
mkdir -p "${LOG_PREFIX}"
echo "====> Run pytest entry: $MULTI_NODE_TEST_PATH"
local log_file="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-?}_pytest.log"
pytest -sv --show-capture=no "$MULTI_NODE_TEST_PATH" 2>&1 | tee "$log_file"
ret=$?
echo "pytest exit code: ret=${ret}"
set -e
if [ "${LWS_WORKER_INDEX:-}" = "0" ]; then
if [ $ret -eq 0 ]; then
print_success "All tests passed!"
touch "${LOG_PREFIX}/aop_done" 2>/dev/null
else
echo "Leader: waiting 10s for worker logs..."
sleep 10
if [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
set +e; aop_pipeline; set -e
fi
local done_file="${LOG_PREFIX}/aop_done"
touch "$done_file"
echo "Leader: notifying workers (${done_file})"
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: Some tests failed${NC}"
exit 1
fi
elif [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
if [ $ret -eq 0 ]; then
echo "Worker: test passed, waiting for leader..."
local wait_timeout=30
while [ $wait_timeout -gt 0 ] && [ ! -f "${LOG_PREFIX}/aop_done" ]; do
sleep 1
wait_timeout=$((wait_timeout - 1))
done
fi
if [ ! -f "${LOG_PREFIX}/aop_done" ]; then
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
local release="${LOG_PREFIX}/aop_done"
mkdir -p "$coord"
touch "${coord}/worker_ready_${LWS_WORKER_INDEX}"
echo "Worker: signalling ready at ${coord}/worker_ready_${LWS_WORKER_INDEX}"
echo "Worker: joining bisect as worker node (index ${LWS_WORKER_INDEX})..."
cd "$WORKSPACE/vllm-ascend"
python -m tests.e2e.nightly.bisect.auto_bisect \
--scene multi_node \
--config-yaml "${CONFIG_YAML_PATH}" \
--bad-commit HEAD \
--coord-dir "${coord}" \
--release-file "${release}"
while [ ! -f "$release" ]; do sleep 5; done
echo "Worker: release signal received, exiting"
exit 1
else
echo "Worker: leader finished successfully, exiting"
fi
fi
}
# Run AOP decision pipeline on failure: classify → check age → bisect-or-exit
# Same logic as _e2e_nightly_multi_node.yaml AOP hooks.
aop_pipeline() {
local rules="$WORKSPACE/vllm-ascend/tests/e2e/nightly/scripts/rules-env.txt"
local table="${GOOD_TABLE:-}"
# Strip branch prefix from BENCHMARK_JOB_NAME (e.g. "main-Qwen3.5-27B-w8a8-A2" → "Qwen3.5-27B-w8a8-A2")
local case_name="${BENCHMARK_JOB_NAME#*-}"
if [ -z "$case_name" ] || [ "$case_name" = "$BENCHMARK_JOB_NAME" ]; then
case_name="${CONFIG_YAML_PATH%.yaml}"
fi
echo "============================================"
echo " AOP Pipeline (Pod) - START"
echo " Config : ${CONFIG_YAML_PATH}"
echo " Case name : ${case_name}"
echo " Rules file : ${rules}"
echo " Table file : ${table}"
echo " Log prefix : ${LOG_PREFIX}"
echo " BENCHMARK_JOB_NAME: ${BENCHMARK_JOB_NAME:-}"
echo "============================================"
# ---- Step 1: Classify ----
echo ""
echo "--- [1/3] Classify: scanning pod logs for env patterns ---"
echo " Rules content:"
if [ -f "$rules" ]; then
grep -vE '^[[:space:]]*(#|$)' "$rules" | sed 's/^/ > /'
else
echo " (rules file not found)"
fi
echo ""
echo " Pod logs found:"
local found_any=0
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
if [ -f "$f" ]; then
echo " - ${f} ($(wc -l < "$f") lines)"
found_any=1
fi
done
[ "$found_any" -eq 0 ] && echo " (no pod logs found)"
local env_count=0
if [ -f "$rules" ]; then
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
if [ -f "$f" ]; then
local n
n=$(grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -ciEf - "$f" 2>/dev/null || echo 0)
n=${n%%[!0-9]*}
echo " Scan ${f}: ${n} matches"
env_count=$((env_count + n))
if [ "$n" -gt 0 ]; then
echo " Matched lines:"
grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -niEf - "$f" | head -5 | sed 's/^/ /'
fi
fi
done
fi
echo " Classify result: env_count=${env_count}"
if [ "$found_any" -eq 0 ]; then
echo " Decision: no pod logs → SKIP"
echo "=== AOP Pipeline (Pod) - END (no logs) ==="
return 1
fi
if [ "$env_count" -gt 0 ]; then
echo " Decision: env_failure → SKIP"
echo "=== AOP Pipeline (Pod) - END (env skip) ==="
return 1
fi
# ---- Step 2: Check age ----
echo ""
echo "--- [2/3] Check commit age ---"
echo " Looking up: ${case_name}"
local skip_age=0
if [ ! -f "$table" ]; then
echo " Table file not found: ${table}"
echo " Decision: no table → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# Only consider success rows
local success_rows
success_rows=$(grep "^${case_name}," "$table" | grep -F ',success,' || true)
if [ -z "$success_rows" ]; then
echo " No success row found for '${case_name}'"
echo " Decision: no success entry → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# Pick most recent success row
local best_date=""
while IFS= read -r row; do
local d
d=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
[ -z "$d" ] && continue
if [ -z "$best_date" ] || [[ "$d" > "$best_date" ]]; then
best_date="$d"
fi
done <<< "$success_rows"
if [ -z "$best_date" ]; then
echo " No valid date in success rows"
echo " Decision: no date → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
echo " Matched row: $(grep -m1 "$best_date" <<< "$success_rows")"
local last_ts now_ts age_days
last_ts=$(date -d "$best_date" +%s 2>/dev/null || echo 0)
if [ "$last_ts" = "0" ] || [ -z "$last_ts" ]; then
echo " Date parse failed: ${best_date}"
echo " Decision: invalid date → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
now_ts=$(date +%s)
age_days=$(( (now_ts - last_ts) / 86400 ))
echo " Last success: ${best_date} (${age_days} days ago, threshold: 3 days)"
if [ "$age_days" -gt 3 ]; then
echo " Decision: old commit (> 3 days) → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# ---- Step 3: Bisect ----
echo ""
echo "--- [3/3] Run bisect ---"
echo " Scene : multi_node"
echo " Config : ${CONFIG_YAML_PATH}"
echo " Bad commit : HEAD"
echo " Name : ${case_name}"
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
echo " Coord dir : ${coord}"
# Wait for all workers to signal ready
echo " Waiting for workers..."
for i in $(seq 1 30); do
local ready_count=0
for f in "${coord}"/worker_ready_*; do
[ -e "$f" ] && ready_count=$((ready_count + 1))
done
echo " [${i}/30] ready workers: ${ready_count}"
if [ "$ready_count" -ge 1 ]; then break; fi
sleep 2
done
cd "$WORKSPACE/vllm-ascend"
local bisect_rc=0
python -m tests.e2e.nightly.bisect.auto_bisect \
--scene multi_node \
--config-yaml "${CONFIG_YAML_PATH}" \
--bad-commit HEAD \
--good-table "${table}" \
--name "${case_name}" \
--coord-dir "${coord}" || bisect_rc=$?
echo " bisect completed (exit code: ${bisect_rc})"
echo "=== AOP Pipeline (Pod) - END ==="
return 1
}
clear_logs() {
print_section "Clearing logs from previous runs"
rm -fr "$HOME/ascend/log" || true
}
backup_ascend_logs() {
if [ -n "${LOG_PREFIX:-}" ]; then
local dest="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-unknown}_plogs"
mkdir -p "$dest"
cp -r /root/ascend/log/. "$dest/" 2>/dev/null || true
echo "Ascend logs backed up to $dest"
fi
}
main() {
trap backup_ascend_logs EXIT
check_npu_info
clear_logs
check_and_config
if [[ "$IS_PR_TEST" == "true" ]]; then
checkout_src
install_vllm_ascend
install_aisbench
fi
show_vllm_info
show_triton_ascend_info
if [[ "$CONFIG_YAML_PATH" == *"DeepSeek-V3_2-Exp-bf16.yaml" ]]; then
install_extra_components
fi
cd "$WORKSPACE/vllm-ascend"
run_tests_with_log
}
main "$@"

View File

@@ -0,0 +1,183 @@
import logging
import os
import socket
import time
from contextlib import contextmanager
from pathlib import Path
from typing import Any
import yaml
logger = logging.getLogger(__name__)
@contextmanager
def temp_env(env_dict: dict[str, Any]):
old_env = {}
for key, value in env_dict.items():
old_env[key] = os.environ.get(key)
os.environ[key] = str(value)
try:
yield
finally:
for key, value in old_env.items():
if value is None:
os.environ.pop(key, None)
else:
os.environ[key] = value
def setup_logger() -> None:
logging.basicConfig(
level=logging.INFO,
format="[%(asctime)s] [%(levelname)s] %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
)
def load_yaml_mapping(
yaml_path: str | None,
*,
default_name: str,
default_base_path: str,
description: str,
) -> dict[str, Any]:
if not yaml_path:
yaml_path = os.getenv("CONFIG_YAML_PATH", default_name)
path = Path(yaml_path)
if not path.is_absolute() and not path.exists():
base_path = os.getenv("CONFIG_BASE_PATH") or default_base_path
path = Path(base_path) / yaml_path
logger.info("Loading %s yaml: %s", description, path)
with path.open(encoding="utf-8") as f:
data = yaml.safe_load(f)
if not isinstance(data, dict):
raise TypeError(f"{description} must be a mapping: {path}")
return data
def dns_resolver(retries: int = 240, base_delay: float = 0.5):
def resolve(dns: str) -> str:
delay = base_delay
for attempt in range(retries):
try:
return socket.gethostbyname(dns)
except socket.gaierror:
if attempt == retries - 1:
raise
time.sleep(delay)
delay = min(delay * 1.5, 5)
raise RuntimeError(f"Unable to resolve DNS: {dns}")
return resolve
def get_cluster_dns_list(world_size: int) -> list[str]:
if world_size < 1:
raise ValueError(f"world_size must be >= 1, got {world_size}")
leader_dns = os.getenv("LWS_LEADER_ADDRESS")
if not leader_dns:
raise RuntimeError("environment variable LWS_LEADER_ADDRESS is not set")
parts = leader_dns.split(".")
if len(parts) < 3:
raise ValueError(f"invalid leader DNS format: {leader_dns}")
leader_name, group_name, namespace = parts[0], parts[1], parts[2]
worker_dns_list = [f"{leader_name}-{idx}.{group_name}.{namespace}" for idx in range(1, world_size)]
return [leader_dns, *worker_dns_list]
def get_cluster_ips(world_size: int = 2) -> list[str]:
resolver = dns_resolver()
return [resolver(dns) for dns in get_cluster_dns_list(world_size)]
def resolve_cluster_ips(
raw_config: dict[str, Any],
num_nodes: int,
explicit_cluster_ips: list[str] | None = None,
*,
cluster_hosts_log_message: str | None = None,
dns_log_message: str = "Resolving cluster IPs via DNS...",
) -> list[str]:
if explicit_cluster_ips is not None:
if len(explicit_cluster_ips) != num_nodes:
raise AssertionError("cluster_ips size mismatch")
return explicit_cluster_ips
cluster_hosts = raw_config.get("cluster_hosts")
if cluster_hosts:
if cluster_hosts_log_message:
logger.info(cluster_hosts_log_message)
if len(cluster_hosts) != num_nodes:
raise AssertionError("cluster_hosts size mismatch")
return list(cluster_hosts)
logger.info(dns_log_message)
return get_cluster_ips(num_nodes)
def get_available_port(start_port: int = 6000, end_port: int = 7000) -> int:
for port in range(start_port, end_port):
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
try:
s.bind(("", port))
return port
except OSError:
continue
raise RuntimeError("No available port found")
def get_cur_ip(retries: int = 20, base_delay: float = 0.5) -> str:
delay = base_delay
for attempt in range(retries):
try:
with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as s:
s.connect(("8.8.8.8", 80))
return s.getsockname()[0]
except Exception:
try:
return socket.gethostbyname(socket.gethostname())
except Exception:
if attempt == retries - 1:
raise RuntimeError("Failed to determine local IP address")
time.sleep(delay)
delay = min(delay * 1.5, 5)
raise RuntimeError("Failed to determine local IP address")
def get_net_interface(ip: str | None = None) -> str:
import psutil
if ip is None:
ip = get_cur_ip()
for iface, addrs in psutil.net_if_addrs().items():
for addr in addrs:
if addr.family == socket.AF_INET and addr.address == ip:
return iface
raise RuntimeError(f"No network interface found for IP {ip}")
def get_all_ipv4() -> list[str]:
ipv4s = {"127.0.0.1"}
hostname = socket.gethostname()
for info in socket.getaddrinfo(hostname, None, family=socket.AF_INET):
ipv4s.add(info[4][0])
return list(ipv4s)
def resolve_current_node_index(cluster_ips: list[str]) -> int:
worker_index = os.environ.get("LWS_WORKER_INDEX")
if worker_index:
return int(worker_index)
local_ips = set(get_all_ipv4())
for index, ip in enumerate(cluster_ips):
if ip in local_ips:
return index
raise RuntimeError("Unable to determine current node index")