1
tests/e2e/nightly/multi_node/scripts/__init__.py
Normal file
1
tests/e2e/nightly/multi_node/scripts/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
|
||||
128
tests/e2e/nightly/multi_node/scripts/benchmark_results.py
Normal file
128
tests/e2e/nightly/multi_node/scripts/benchmark_results.py
Normal file
@@ -0,0 +1,128 @@
|
||||
import json
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PORT_ENV_KEYS = {"SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"}
|
||||
INFRA_ENV_KEYS = {
|
||||
"HCCL_IF_IP",
|
||||
"HCCL_SOCKET_IFNAME",
|
||||
"GLOO_SOCKET_IFNAME",
|
||||
"TP_SOCKET_IFNAME",
|
||||
"LOCAL_IP",
|
||||
"NIC_NAME",
|
||||
"MASTER_IP",
|
||||
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
|
||||
}
|
||||
PERF_METRIC_RENAME: dict[str, str] = {
|
||||
"Benchmark Duration": "Benchmark_Duration(BD)",
|
||||
"Prefill Token Throughput": "Prefill_Token_Throughput(PTT)",
|
||||
"Input Token Throughput": "Input_Token_Throughput(ITT)",
|
||||
"Output Token Throughput": "Output_Token_Throughput(OTT)",
|
||||
"Total Token Throughput": "Total_Token_Throughput(TTT)",
|
||||
}
|
||||
|
||||
|
||||
def extract_hardware(runner: str) -> str:
|
||||
runner_lower = runner.lower()
|
||||
for label in ("a3", "a2"):
|
||||
if label in runner_lower:
|
||||
return label.upper()
|
||||
return runner
|
||||
|
||||
|
||||
def get_vllm_version() -> str:
|
||||
try:
|
||||
import vllm
|
||||
|
||||
return vllm.__version__
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def task_passed(case_config: dict[str, Any], result: Any) -> bool:
|
||||
if result == "":
|
||||
return False
|
||||
case_type = case_config.get("case_type")
|
||||
baseline = case_config.get("baseline")
|
||||
threshold = case_config.get("threshold")
|
||||
if baseline is None or threshold is None:
|
||||
return True
|
||||
if case_type == "accuracy" and isinstance(result, (int, float)):
|
||||
return abs(float(result) - float(baseline)) <= float(threshold)
|
||||
if case_type == "performance" and isinstance(result, list) and len(result) == 2:
|
||||
_, result_json = result
|
||||
throughput_str = result_json.get("Output Token Throughput", {}).get("total", "")
|
||||
try:
|
||||
throughput_val = float(throughput_str.replace("token/s", "").strip())
|
||||
return throughput_val >= float(threshold) * float(baseline)
|
||||
except (ValueError, AttributeError):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def build_task_entry(case_key: str, case_config: dict[str, Any], result: Any) -> dict[str, Any]:
|
||||
dataset_path = case_config.get("dataset_path", "")
|
||||
dataset_conf = case_config.get("dataset_conf", "")
|
||||
if dataset_path:
|
||||
task_name = dataset_path.split("/", 1)[-1]
|
||||
elif dataset_conf:
|
||||
task_name = dataset_conf.split("/")[0]
|
||||
else:
|
||||
task_name = case_key
|
||||
|
||||
case_type = case_config.get("case_type", "unknown")
|
||||
metrics: dict[str, float] = {}
|
||||
if result == "":
|
||||
pass
|
||||
elif case_type == "accuracy" and isinstance(result, (int, float)):
|
||||
metrics["accuracy"] = round(float(result), 4)
|
||||
elif case_type == "performance" and isinstance(result, list) and len(result) == 2:
|
||||
_, result_json = result
|
||||
for metric_name, metric_data in result_json.items():
|
||||
if not isinstance(metric_data, dict):
|
||||
continue
|
||||
total_str = metric_data.get("total", "")
|
||||
try:
|
||||
value = float(total_str.replace("token/s", "").replace("ms", "").replace("s", "").strip())
|
||||
metrics[PERF_METRIC_RENAME.get(metric_name, metric_name)] = round(value, 4)
|
||||
except (ValueError, AttributeError):
|
||||
pass
|
||||
|
||||
test_input_keys = ("num_prompts", "max_out_len", "batch_size", "request_rate")
|
||||
test_input = {key: case_config[key] for key in test_input_keys if key in case_config}
|
||||
|
||||
target: dict[str, Any] = {}
|
||||
if case_config.get("baseline") is not None:
|
||||
target["baseline"] = case_config["baseline"]
|
||||
if case_config.get("threshold") is not None:
|
||||
target["threshold"] = case_config["threshold"]
|
||||
|
||||
entry: dict[str, Any] = {"name": task_name, "metrics": metrics, "test_input": test_input}
|
||||
if target:
|
||||
entry["target"] = target
|
||||
entry["pass_fail"] = "pass" if task_passed(case_config, result) else "fail"
|
||||
return entry
|
||||
|
||||
|
||||
def filter_environment(envs: dict[str, Any]) -> dict[str, Any]:
|
||||
exclude = PORT_ENV_KEYS | INFRA_ENV_KEYS
|
||||
return {key: value for key, value in envs.items() if key not in exclude}
|
||||
|
||||
|
||||
def write_results_json(
|
||||
output: dict[str, Any],
|
||||
*,
|
||||
job_name: str,
|
||||
output_dir: Path | None = None,
|
||||
) -> Path:
|
||||
if output_dir is None:
|
||||
output_dir = Path("/root/.cache/benchmark_results") / job_name
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
output_path = output_dir / f"{job_name}.json"
|
||||
output_path.write_text(json.dumps(output, indent=2, ensure_ascii=False), encoding="utf-8")
|
||||
logger.info("Benchmark results saved to PVC at %s", output_path)
|
||||
print(f"Benchmark results saved to PVC at {output_path}")
|
||||
return output_path
|
||||
174
tests/e2e/nightly/multi_node/scripts/lws.yaml.jinja2
Normal file
174
tests/e2e/nightly/multi_node/scripts/lws.yaml.jinja2
Normal file
@@ -0,0 +1,174 @@
|
||||
apiVersion: leaderworkerset.x-k8s.io/v1
|
||||
kind: LeaderWorkerSet
|
||||
metadata:
|
||||
name: {{ lws_name | default("vllm") }}
|
||||
namespace: vllm-project
|
||||
spec:
|
||||
replicas: {{ replicas | default(1) }}
|
||||
leaderWorkerTemplate:
|
||||
size: {{ size | default(2) }}
|
||||
restartPolicy: None
|
||||
leaderTemplate:
|
||||
metadata:
|
||||
labels:
|
||||
role: leader
|
||||
spec:
|
||||
tolerations:
|
||||
- key: "dedicated"
|
||||
operator: "Equal"
|
||||
value: "night"
|
||||
effect: "NoSchedule"
|
||||
containers:
|
||||
- name: vllm-leader
|
||||
imagePullPolicy: Always
|
||||
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
|
||||
env:
|
||||
- name: CONFIG_YAML_PATH
|
||||
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
|
||||
- name: CONFIG_BASE_PATH
|
||||
value: "{{ config_base_path | default("") }}"
|
||||
- name: LOG_PREFIX
|
||||
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
|
||||
- name: WORKSPACE
|
||||
value: "/vllm-workspace"
|
||||
- name: FAIL_TAG
|
||||
value: {{ fail_tag | default("FAIL_TAG") }}
|
||||
- name: IS_PR_TEST
|
||||
value: "{{ is_pr_test | default("false") }}"
|
||||
- name: VLLM_ASCEND_REF
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: VLLM_ASCEND_REMOTE_URL
|
||||
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
|
||||
- name: BENCHMARK_JOB_NAME
|
||||
value: {{ benchmark_job_name | default("") }}
|
||||
- name: VLLM_CI_RUNNER
|
||||
value: {{ runner | default("linux-aarch64-a3-0") }}
|
||||
- name: VLLM_ASCEND_VERSION
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: AOP_MULTI_ENABLED
|
||||
value: "{{ aop_multi_enabled }}"
|
||||
- name: GOOD_TABLE
|
||||
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- |
|
||||
bash /root/.cache/tests/run.sh
|
||||
resources:
|
||||
limits:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
memory: 512Gi
|
||||
ephemeral-storage: 100Gi
|
||||
requests:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
ephemeral-storage: 100Gi
|
||||
cpu: 125
|
||||
ports:
|
||||
- containerPort: 8080
|
||||
# readinessProbe:
|
||||
# tcpSocket:
|
||||
# port: 8080
|
||||
# initialDelaySeconds: 15
|
||||
# periodSeconds: 10
|
||||
volumeMounts:
|
||||
- mountPath: /root/.cache
|
||||
name: shared-volume
|
||||
- mountPath: /usr/local/Ascend/driver/tools
|
||||
name: driver-tools
|
||||
- mountPath: /dev/shm
|
||||
name: dshm
|
||||
volumes:
|
||||
- name: dshm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 512Gi
|
||||
- name: shared-volume
|
||||
persistentVolumeClaim:
|
||||
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
|
||||
- name: driver-tools
|
||||
hostPath:
|
||||
path: /usr/local/Ascend/driver/tools
|
||||
workerTemplate:
|
||||
spec:
|
||||
tolerations:
|
||||
- key: "dedicated"
|
||||
operator: "Equal"
|
||||
value: "night"
|
||||
effect: "NoSchedule"
|
||||
containers:
|
||||
- name: vllm-worker
|
||||
imagePullPolicy: Always
|
||||
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
|
||||
env:
|
||||
- name: CONFIG_YAML_PATH
|
||||
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
|
||||
- name: CONFIG_BASE_PATH
|
||||
value: "{{ config_base_path | default("") }}"
|
||||
- name: LOG_PREFIX
|
||||
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
|
||||
- name: WORKSPACE
|
||||
value: "/vllm-workspace"
|
||||
- name: FAIL_TAG
|
||||
value: {{ fail_tag | default("FAIL_TAG") }}
|
||||
- name: IS_PR_TEST
|
||||
value: "{{ is_pr_test | default("false") }}"
|
||||
- name: VLLM_ASCEND_REF
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: VLLM_ASCEND_REMOTE_URL
|
||||
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
|
||||
- name: BENCHMARK_JOB_NAME
|
||||
value: {{ benchmark_job_name | default("") }}
|
||||
- name: VLLM_CI_RUNNER
|
||||
value: {{ runner | default("linux-aarch64-a3-0") }}
|
||||
- name: AOP_MULTI_ENABLED
|
||||
value: "{{ aop_multi_enabled }}"
|
||||
- name: GOOD_TABLE
|
||||
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- |
|
||||
bash /root/.cache/tests/run.sh
|
||||
resources:
|
||||
limits:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
memory: 512Gi
|
||||
ephemeral-storage: 100Gi
|
||||
requests:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
ephemeral-storage: 100Gi
|
||||
cpu: 125
|
||||
volumeMounts:
|
||||
- mountPath: /root/.cache
|
||||
name: shared-volume
|
||||
- mountPath: /usr/local/Ascend/driver/tools
|
||||
name: driver-tools
|
||||
- mountPath: /dev/shm
|
||||
name: dshm
|
||||
volumes:
|
||||
- name: dshm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 512Gi
|
||||
- name: shared-volume
|
||||
persistentVolumeClaim:
|
||||
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
|
||||
- name: driver-tools
|
||||
hostPath:
|
||||
path: /usr/local/Ascend/driver/tools
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: {{ lws_name | default("vllm") }}-leader
|
||||
namespace: vllm-project
|
||||
spec:
|
||||
ports:
|
||||
- name: http
|
||||
port: 8080
|
||||
protocol: TCP
|
||||
targetPort: 8080
|
||||
selector:
|
||||
leaderworkerset.sigs.k8s.io/name: {{ lws_name | default("vllm") }}
|
||||
role: leader
|
||||
type: ClusterIP
|
||||
466
tests/e2e/nightly/multi_node/scripts/run.sh
Normal file
466
tests/e2e/nightly/multi_node/scripts/run.sh
Normal file
@@ -0,0 +1,466 @@
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
# Color definitions
|
||||
GREEN="\033[0;32m"
|
||||
BLUE="\033[0;34m"
|
||||
YELLOW="\033[0;33m"
|
||||
RED="\033[0;31m"
|
||||
NC="\033[0m" # No Color
|
||||
|
||||
INTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/internal_dp/scripts/test_multi_node.py"
|
||||
EXTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py"
|
||||
|
||||
if [ -z "${MULTI_NODE_TEST_PATH:-}" ]; then
|
||||
if [[ "${CONFIG_BASE_PATH:-}" == *"external_dp/config"* || "${CONFIG_YAML_PATH:-}" == *"external_dp/config"* ]]; then
|
||||
MULTI_NODE_TEST_PATH="$EXTERNAL_DP_TEST_PATH"
|
||||
else
|
||||
MULTI_NODE_TEST_PATH="$INTERNAL_DP_TEST_PATH"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Configuration
|
||||
export LD_LIBRARY_PATH=/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:$LD_LIBRARY_PATH
|
||||
export LD_LIBRARY_PATH=/usr/local/lib:$LD_LIBRARY_PATH
|
||||
# cann and atb environment setup
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh
|
||||
source /usr/local/Ascend/cann-9.1.0/share/info/ascendnpu-ir/bin/set_env.sh
|
||||
|
||||
set +eu
|
||||
source /usr/local/Ascend/nnal/atb/set_env.sh
|
||||
set -eu
|
||||
|
||||
# Home path for aisbench
|
||||
export BENCHMARK_HOME=${WORKSPACE}/vllm-ascend/benchmark
|
||||
|
||||
# Logging configurations
|
||||
export VLLM_LOGGING_LEVEL="INFO"
|
||||
# Reduce glog verbosity for mooncake
|
||||
export GLOG_minloglevel=1
|
||||
# Set transformers to offline mode to avoid downloading models during tests
|
||||
export HF_HUB_OFFLINE="1"
|
||||
# Default is 600s
|
||||
export VLLM_ENGINE_READY_TIMEOUT_S=1800
|
||||
|
||||
# Function to print section headers
|
||||
print_section() {
|
||||
echo -e "\n${BLUE}=== $1 ===${NC}"
|
||||
}
|
||||
|
||||
print_failure() {
|
||||
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: $1${NC}"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Function to print success messages
|
||||
print_success() {
|
||||
echo -e "${GREEN}✓ $1${NC}"
|
||||
}
|
||||
|
||||
# Function to print error messages and exit
|
||||
print_error() {
|
||||
echo -e "${RED}✗ ERROR: $1${NC}"
|
||||
exit 1
|
||||
}
|
||||
|
||||
show_vllm_info() {
|
||||
cd "$WORKSPACE"
|
||||
echo "Installed vLLM-related Python packages:"
|
||||
pip list | grep vllm || echo "No vllm packages found."
|
||||
|
||||
echo ""
|
||||
echo "============================"
|
||||
echo "vLLM Git information"
|
||||
echo "============================"
|
||||
cd vllm
|
||||
if [ -d .git ]; then
|
||||
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
|
||||
echo "Commit hash: $(git rev-parse HEAD)"
|
||||
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
|
||||
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
|
||||
echo "Message: $(git log -1 --pretty=format:'%s')"
|
||||
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
|
||||
echo "Remote: $(git remote -v | head -n1)"
|
||||
echo ""
|
||||
else
|
||||
echo "No .git directory found in vllm"
|
||||
fi
|
||||
cd ..
|
||||
|
||||
echo ""
|
||||
echo "============================"
|
||||
echo "vLLM-Ascend Git information"
|
||||
echo "============================"
|
||||
cd vllm-ascend
|
||||
if [ -d .git ]; then
|
||||
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
|
||||
echo "Commit hash: $(git rev-parse HEAD)"
|
||||
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
|
||||
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
|
||||
echo "Message: $(git log -1 --pretty=format:'%s')"
|
||||
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
|
||||
echo "Remote: $(git remote -v | head -n1)"
|
||||
echo ""
|
||||
else
|
||||
echo "No .git directory found in vllm-ascend"
|
||||
fi
|
||||
cd ..
|
||||
}
|
||||
|
||||
check_npu_info() {
|
||||
echo "====> Check NPU info"
|
||||
npu-smi info
|
||||
cat "/usr/local/Ascend/ascend-toolkit/latest/$(uname -i)-linux/ascend_toolkit_install.info"
|
||||
}
|
||||
|
||||
check_and_config() {
|
||||
echo "====> Configure mirrors and git proxy"
|
||||
git config --global url."https://ghfast.top/https://github.com/".insteadOf "https://github.com/"
|
||||
pip config set global.index-url https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple
|
||||
export PIP_EXTRA_INDEX_URL="https://mirrors.huaweicloud.com/ascend/repos/pypi"
|
||||
}
|
||||
|
||||
install_extra_components() {
|
||||
echo "====> Installing extra components for DeepSeek-v3.2-exp-bf16"
|
||||
|
||||
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/CANN-custom_ops-sfa-linux.aarch64.run; then
|
||||
echo "Failed to download CANN-custom_ops-sfa-linux.aarch64.run"
|
||||
return 1
|
||||
fi
|
||||
chmod +x ./CANN-custom_ops-sfa-linux.aarch64.run
|
||||
./CANN-custom_ops-sfa-linux.aarch64.run --quiet
|
||||
|
||||
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/custom_ops-1.0-cp311-cp311-linux_aarch64.whl; then
|
||||
echo "Failed to download custom_ops wheel"
|
||||
return 1
|
||||
fi
|
||||
pip install custom_ops-1.0-cp311-cp311-linux_aarch64.whl
|
||||
|
||||
export ASCEND_CUSTOM_OPP_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize${ASCEND_CUSTOM_OPP_PATH:+:${ASCEND_CUSTOM_OPP_PATH}}"
|
||||
export LD_LIBRARY_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize/op_api/lib/${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh
|
||||
|
||||
rm -f CANN-custom_ops-sfa-linux.aarch64.run \
|
||||
custom_ops-1.0-cp311-cp311-linux_aarch64.whl
|
||||
echo "====> Extra components installation completed"
|
||||
}
|
||||
|
||||
checkout_src() {
|
||||
echo "====> Checkout source code"
|
||||
mkdir -p "$WORKSPACE"
|
||||
cd "$WORKSPACE"
|
||||
pip uninstall -y vllm-ascend || true
|
||||
cp -r "$WORKSPACE/vllm-ascend/benchmark" /tmp/aisbench-backup || true
|
||||
rm -rf "$WORKSPACE/vllm-ascend"
|
||||
|
||||
if [ ! -d "$WORKSPACE/vllm-ascend" ]; then
|
||||
echo "Cloning vllm-ascend from $VLLM_ASCEND_REMOTE_URL"
|
||||
git clone --depth 1 --recurse-submodules "$VLLM_ASCEND_REMOTE_URL" "$WORKSPACE/vllm-ascend"
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
PR_REF=$(git ls-remote origin 'refs/pull/*/head' | grep "^${VLLM_ASCEND_REF}" | awk '{print $2}' | head -1)
|
||||
if [ -n "$PR_REF" ]; then
|
||||
git fetch --depth 1 origin "$PR_REF"
|
||||
git checkout FETCH_HEAD
|
||||
else
|
||||
git fetch origin '+refs/pull/*/head:refs/remotes/pull/*' 2>/dev/null || true
|
||||
git checkout "$VLLM_ASCEND_REF"
|
||||
fi
|
||||
git submodule update --init --recursive
|
||||
fi
|
||||
}
|
||||
|
||||
install_vllm_ascend() {
|
||||
echo "====> Install vllm-ascend"
|
||||
pip install -r "$WORKSPACE/vllm-ascend/requirements-dev.txt"
|
||||
pip install -e "$WORKSPACE/vllm-ascend"
|
||||
}
|
||||
|
||||
install_aisbench() {
|
||||
echo "====> Install AISBench benchmark"
|
||||
|
||||
BENCH_DIR="$WORKSPACE/vllm-ascend/benchmark"
|
||||
|
||||
cp -r /tmp/aisbench-backup "$BENCH_DIR"
|
||||
|
||||
cd "$BENCH_DIR"
|
||||
pip install -e . \
|
||||
-r requirements/api.txt \
|
||||
-r requirements/extra.txt
|
||||
|
||||
python3 -m pip cache purge || echo "WARNING: pip cache purge failed, but proceeding..."
|
||||
|
||||
}
|
||||
|
||||
show_triton_ascend_info() {
|
||||
echo "====> Check triton ascend info"
|
||||
clang -v
|
||||
which bishengir-compile
|
||||
pip show triton-ascend
|
||||
}
|
||||
|
||||
kill_npu_processes() {
|
||||
pgrep python3 | xargs -r kill -9
|
||||
pgrep VLLM | xargs -r kill -9
|
||||
|
||||
sleep 4
|
||||
}
|
||||
|
||||
run_tests_with_log() {
|
||||
set +e
|
||||
kill_npu_processes
|
||||
mkdir -p "${LOG_PREFIX}"
|
||||
echo "====> Run pytest entry: $MULTI_NODE_TEST_PATH"
|
||||
local log_file="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-?}_pytest.log"
|
||||
pytest -sv --show-capture=no "$MULTI_NODE_TEST_PATH" 2>&1 | tee "$log_file"
|
||||
ret=$?
|
||||
echo "pytest exit code: ret=${ret}"
|
||||
set -e
|
||||
if [ "${LWS_WORKER_INDEX:-}" = "0" ]; then
|
||||
if [ $ret -eq 0 ]; then
|
||||
print_success "All tests passed!"
|
||||
touch "${LOG_PREFIX}/aop_done" 2>/dev/null
|
||||
else
|
||||
echo "Leader: waiting 10s for worker logs..."
|
||||
sleep 10
|
||||
if [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
|
||||
set +e; aop_pipeline; set -e
|
||||
fi
|
||||
local done_file="${LOG_PREFIX}/aop_done"
|
||||
touch "$done_file"
|
||||
echo "Leader: notifying workers (${done_file})"
|
||||
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: Some tests failed${NC}"
|
||||
exit 1
|
||||
fi
|
||||
elif [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
|
||||
if [ $ret -eq 0 ]; then
|
||||
echo "Worker: test passed, waiting for leader..."
|
||||
local wait_timeout=30
|
||||
while [ $wait_timeout -gt 0 ] && [ ! -f "${LOG_PREFIX}/aop_done" ]; do
|
||||
sleep 1
|
||||
wait_timeout=$((wait_timeout - 1))
|
||||
done
|
||||
fi
|
||||
if [ ! -f "${LOG_PREFIX}/aop_done" ]; then
|
||||
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
|
||||
local release="${LOG_PREFIX}/aop_done"
|
||||
mkdir -p "$coord"
|
||||
touch "${coord}/worker_ready_${LWS_WORKER_INDEX}"
|
||||
echo "Worker: signalling ready at ${coord}/worker_ready_${LWS_WORKER_INDEX}"
|
||||
echo "Worker: joining bisect as worker node (index ${LWS_WORKER_INDEX})..."
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
python -m tests.e2e.nightly.bisect.auto_bisect \
|
||||
--scene multi_node \
|
||||
--config-yaml "${CONFIG_YAML_PATH}" \
|
||||
--bad-commit HEAD \
|
||||
--coord-dir "${coord}" \
|
||||
--release-file "${release}"
|
||||
while [ ! -f "$release" ]; do sleep 5; done
|
||||
echo "Worker: release signal received, exiting"
|
||||
exit 1
|
||||
else
|
||||
echo "Worker: leader finished successfully, exiting"
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
# Run AOP decision pipeline on failure: classify → check age → bisect-or-exit
|
||||
# Same logic as _e2e_nightly_multi_node.yaml AOP hooks.
|
||||
aop_pipeline() {
|
||||
local rules="$WORKSPACE/vllm-ascend/tests/e2e/nightly/scripts/rules-env.txt"
|
||||
local table="${GOOD_TABLE:-}"
|
||||
# Strip branch prefix from BENCHMARK_JOB_NAME (e.g. "main-Qwen3.5-27B-w8a8-A2" → "Qwen3.5-27B-w8a8-A2")
|
||||
local case_name="${BENCHMARK_JOB_NAME#*-}"
|
||||
if [ -z "$case_name" ] || [ "$case_name" = "$BENCHMARK_JOB_NAME" ]; then
|
||||
case_name="${CONFIG_YAML_PATH%.yaml}"
|
||||
fi
|
||||
|
||||
echo "============================================"
|
||||
echo " AOP Pipeline (Pod) - START"
|
||||
echo " Config : ${CONFIG_YAML_PATH}"
|
||||
echo " Case name : ${case_name}"
|
||||
echo " Rules file : ${rules}"
|
||||
echo " Table file : ${table}"
|
||||
echo " Log prefix : ${LOG_PREFIX}"
|
||||
echo " BENCHMARK_JOB_NAME: ${BENCHMARK_JOB_NAME:-}"
|
||||
echo "============================================"
|
||||
|
||||
# ---- Step 1: Classify ----
|
||||
echo ""
|
||||
echo "--- [1/3] Classify: scanning pod logs for env patterns ---"
|
||||
echo " Rules content:"
|
||||
if [ -f "$rules" ]; then
|
||||
grep -vE '^[[:space:]]*(#|$)' "$rules" | sed 's/^/ > /'
|
||||
else
|
||||
echo " (rules file not found)"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo " Pod logs found:"
|
||||
local found_any=0
|
||||
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
|
||||
if [ -f "$f" ]; then
|
||||
echo " - ${f} ($(wc -l < "$f") lines)"
|
||||
found_any=1
|
||||
fi
|
||||
done
|
||||
[ "$found_any" -eq 0 ] && echo " (no pod logs found)"
|
||||
|
||||
local env_count=0
|
||||
if [ -f "$rules" ]; then
|
||||
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
|
||||
if [ -f "$f" ]; then
|
||||
local n
|
||||
n=$(grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -ciEf - "$f" 2>/dev/null || echo 0)
|
||||
n=${n%%[!0-9]*}
|
||||
echo " Scan ${f}: ${n} matches"
|
||||
env_count=$((env_count + n))
|
||||
if [ "$n" -gt 0 ]; then
|
||||
echo " Matched lines:"
|
||||
grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -niEf - "$f" | head -5 | sed 's/^/ /'
|
||||
fi
|
||||
fi
|
||||
done
|
||||
fi
|
||||
echo " Classify result: env_count=${env_count}"
|
||||
|
||||
if [ "$found_any" -eq 0 ]; then
|
||||
echo " Decision: no pod logs → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (no logs) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
if [ "$env_count" -gt 0 ]; then
|
||||
echo " Decision: env_failure → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (env skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# ---- Step 2: Check age ----
|
||||
echo ""
|
||||
echo "--- [2/3] Check commit age ---"
|
||||
echo " Looking up: ${case_name}"
|
||||
local skip_age=0
|
||||
if [ ! -f "$table" ]; then
|
||||
echo " Table file not found: ${table}"
|
||||
echo " Decision: no table → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Only consider success rows
|
||||
local success_rows
|
||||
success_rows=$(grep "^${case_name}," "$table" | grep -F ',success,' || true)
|
||||
if [ -z "$success_rows" ]; then
|
||||
echo " No success row found for '${case_name}'"
|
||||
echo " Decision: no success entry → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Pick most recent success row
|
||||
local best_date=""
|
||||
while IFS= read -r row; do
|
||||
local d
|
||||
d=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
|
||||
[ -z "$d" ] && continue
|
||||
if [ -z "$best_date" ] || [[ "$d" > "$best_date" ]]; then
|
||||
best_date="$d"
|
||||
fi
|
||||
done <<< "$success_rows"
|
||||
|
||||
if [ -z "$best_date" ]; then
|
||||
echo " No valid date in success rows"
|
||||
echo " Decision: no date → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
echo " Matched row: $(grep -m1 "$best_date" <<< "$success_rows")"
|
||||
local last_ts now_ts age_days
|
||||
last_ts=$(date -d "$best_date" +%s 2>/dev/null || echo 0)
|
||||
if [ "$last_ts" = "0" ] || [ -z "$last_ts" ]; then
|
||||
echo " Date parse failed: ${best_date}"
|
||||
echo " Decision: invalid date → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
now_ts=$(date +%s)
|
||||
age_days=$(( (now_ts - last_ts) / 86400 ))
|
||||
echo " Last success: ${best_date} (${age_days} days ago, threshold: 3 days)"
|
||||
|
||||
if [ "$age_days" -gt 3 ]; then
|
||||
echo " Decision: old commit (> 3 days) → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# ---- Step 3: Bisect ----
|
||||
echo ""
|
||||
echo "--- [3/3] Run bisect ---"
|
||||
echo " Scene : multi_node"
|
||||
echo " Config : ${CONFIG_YAML_PATH}"
|
||||
echo " Bad commit : HEAD"
|
||||
echo " Name : ${case_name}"
|
||||
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
|
||||
echo " Coord dir : ${coord}"
|
||||
|
||||
# Wait for all workers to signal ready
|
||||
echo " Waiting for workers..."
|
||||
for i in $(seq 1 30); do
|
||||
local ready_count=0
|
||||
for f in "${coord}"/worker_ready_*; do
|
||||
[ -e "$f" ] && ready_count=$((ready_count + 1))
|
||||
done
|
||||
echo " [${i}/30] ready workers: ${ready_count}"
|
||||
if [ "$ready_count" -ge 1 ]; then break; fi
|
||||
sleep 2
|
||||
done
|
||||
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
local bisect_rc=0
|
||||
python -m tests.e2e.nightly.bisect.auto_bisect \
|
||||
--scene multi_node \
|
||||
--config-yaml "${CONFIG_YAML_PATH}" \
|
||||
--bad-commit HEAD \
|
||||
--good-table "${table}" \
|
||||
--name "${case_name}" \
|
||||
--coord-dir "${coord}" || bisect_rc=$?
|
||||
echo " bisect completed (exit code: ${bisect_rc})"
|
||||
echo "=== AOP Pipeline (Pod) - END ==="
|
||||
return 1
|
||||
}
|
||||
|
||||
clear_logs() {
|
||||
print_section "Clearing logs from previous runs"
|
||||
rm -fr "$HOME/ascend/log" || true
|
||||
}
|
||||
|
||||
backup_ascend_logs() {
|
||||
if [ -n "${LOG_PREFIX:-}" ]; then
|
||||
local dest="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-unknown}_plogs"
|
||||
mkdir -p "$dest"
|
||||
cp -r /root/ascend/log/. "$dest/" 2>/dev/null || true
|
||||
echo "Ascend logs backed up to $dest"
|
||||
fi
|
||||
}
|
||||
|
||||
main() {
|
||||
trap backup_ascend_logs EXIT
|
||||
check_npu_info
|
||||
clear_logs
|
||||
check_and_config
|
||||
if [[ "$IS_PR_TEST" == "true" ]]; then
|
||||
checkout_src
|
||||
install_vllm_ascend
|
||||
install_aisbench
|
||||
fi
|
||||
show_vllm_info
|
||||
show_triton_ascend_info
|
||||
if [[ "$CONFIG_YAML_PATH" == *"DeepSeek-V3_2-Exp-bf16.yaml" ]]; then
|
||||
install_extra_components
|
||||
fi
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
run_tests_with_log
|
||||
}
|
||||
|
||||
main "$@"
|
||||
183
tests/e2e/nightly/multi_node/scripts/utils.py
Normal file
183
tests/e2e/nightly/multi_node/scripts/utils.py
Normal file
@@ -0,0 +1,183 @@
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import time
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def temp_env(env_dict: dict[str, Any]):
|
||||
old_env = {}
|
||||
for key, value in env_dict.items():
|
||||
old_env[key] = os.environ.get(key)
|
||||
os.environ[key] = str(value)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
for key, value in old_env.items():
|
||||
if value is None:
|
||||
os.environ.pop(key, None)
|
||||
else:
|
||||
os.environ[key] = value
|
||||
|
||||
|
||||
def setup_logger() -> None:
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="[%(asctime)s] [%(levelname)s] %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
)
|
||||
|
||||
|
||||
def load_yaml_mapping(
|
||||
yaml_path: str | None,
|
||||
*,
|
||||
default_name: str,
|
||||
default_base_path: str,
|
||||
description: str,
|
||||
) -> dict[str, Any]:
|
||||
if not yaml_path:
|
||||
yaml_path = os.getenv("CONFIG_YAML_PATH", default_name)
|
||||
|
||||
path = Path(yaml_path)
|
||||
if not path.is_absolute() and not path.exists():
|
||||
base_path = os.getenv("CONFIG_BASE_PATH") or default_base_path
|
||||
path = Path(base_path) / yaml_path
|
||||
|
||||
logger.info("Loading %s yaml: %s", description, path)
|
||||
with path.open(encoding="utf-8") as f:
|
||||
data = yaml.safe_load(f)
|
||||
if not isinstance(data, dict):
|
||||
raise TypeError(f"{description} must be a mapping: {path}")
|
||||
return data
|
||||
|
||||
|
||||
def dns_resolver(retries: int = 240, base_delay: float = 0.5):
|
||||
def resolve(dns: str) -> str:
|
||||
delay = base_delay
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
return socket.gethostbyname(dns)
|
||||
except socket.gaierror:
|
||||
if attempt == retries - 1:
|
||||
raise
|
||||
time.sleep(delay)
|
||||
delay = min(delay * 1.5, 5)
|
||||
raise RuntimeError(f"Unable to resolve DNS: {dns}")
|
||||
|
||||
return resolve
|
||||
|
||||
|
||||
def get_cluster_dns_list(world_size: int) -> list[str]:
|
||||
if world_size < 1:
|
||||
raise ValueError(f"world_size must be >= 1, got {world_size}")
|
||||
|
||||
leader_dns = os.getenv("LWS_LEADER_ADDRESS")
|
||||
if not leader_dns:
|
||||
raise RuntimeError("environment variable LWS_LEADER_ADDRESS is not set")
|
||||
|
||||
parts = leader_dns.split(".")
|
||||
if len(parts) < 3:
|
||||
raise ValueError(f"invalid leader DNS format: {leader_dns}")
|
||||
|
||||
leader_name, group_name, namespace = parts[0], parts[1], parts[2]
|
||||
worker_dns_list = [f"{leader_name}-{idx}.{group_name}.{namespace}" for idx in range(1, world_size)]
|
||||
return [leader_dns, *worker_dns_list]
|
||||
|
||||
|
||||
def get_cluster_ips(world_size: int = 2) -> list[str]:
|
||||
resolver = dns_resolver()
|
||||
return [resolver(dns) for dns in get_cluster_dns_list(world_size)]
|
||||
|
||||
|
||||
def resolve_cluster_ips(
|
||||
raw_config: dict[str, Any],
|
||||
num_nodes: int,
|
||||
explicit_cluster_ips: list[str] | None = None,
|
||||
*,
|
||||
cluster_hosts_log_message: str | None = None,
|
||||
dns_log_message: str = "Resolving cluster IPs via DNS...",
|
||||
) -> list[str]:
|
||||
if explicit_cluster_ips is not None:
|
||||
if len(explicit_cluster_ips) != num_nodes:
|
||||
raise AssertionError("cluster_ips size mismatch")
|
||||
return explicit_cluster_ips
|
||||
|
||||
cluster_hosts = raw_config.get("cluster_hosts")
|
||||
if cluster_hosts:
|
||||
if cluster_hosts_log_message:
|
||||
logger.info(cluster_hosts_log_message)
|
||||
if len(cluster_hosts) != num_nodes:
|
||||
raise AssertionError("cluster_hosts size mismatch")
|
||||
return list(cluster_hosts)
|
||||
|
||||
logger.info(dns_log_message)
|
||||
return get_cluster_ips(num_nodes)
|
||||
|
||||
|
||||
def get_available_port(start_port: int = 6000, end_port: int = 7000) -> int:
|
||||
for port in range(start_port, end_port):
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
try:
|
||||
s.bind(("", port))
|
||||
return port
|
||||
except OSError:
|
||||
continue
|
||||
raise RuntimeError("No available port found")
|
||||
|
||||
|
||||
def get_cur_ip(retries: int = 20, base_delay: float = 0.5) -> str:
|
||||
delay = base_delay
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as s:
|
||||
s.connect(("8.8.8.8", 80))
|
||||
return s.getsockname()[0]
|
||||
except Exception:
|
||||
try:
|
||||
return socket.gethostbyname(socket.gethostname())
|
||||
except Exception:
|
||||
if attempt == retries - 1:
|
||||
raise RuntimeError("Failed to determine local IP address")
|
||||
time.sleep(delay)
|
||||
delay = min(delay * 1.5, 5)
|
||||
raise RuntimeError("Failed to determine local IP address")
|
||||
|
||||
|
||||
def get_net_interface(ip: str | None = None) -> str:
|
||||
import psutil
|
||||
|
||||
if ip is None:
|
||||
ip = get_cur_ip()
|
||||
|
||||
for iface, addrs in psutil.net_if_addrs().items():
|
||||
for addr in addrs:
|
||||
if addr.family == socket.AF_INET and addr.address == ip:
|
||||
return iface
|
||||
raise RuntimeError(f"No network interface found for IP {ip}")
|
||||
|
||||
|
||||
def get_all_ipv4() -> list[str]:
|
||||
ipv4s = {"127.0.0.1"}
|
||||
hostname = socket.gethostname()
|
||||
for info in socket.getaddrinfo(hostname, None, family=socket.AF_INET):
|
||||
ipv4s.add(info[4][0])
|
||||
return list(ipv4s)
|
||||
|
||||
|
||||
def resolve_current_node_index(cluster_ips: list[str]) -> int:
|
||||
worker_index = os.environ.get("LWS_WORKER_INDEX")
|
||||
if worker_index:
|
||||
return int(worker_index)
|
||||
|
||||
local_ips = set(get_all_ipv4())
|
||||
for index, ip in enumerate(cluster_ips):
|
||||
if ip in local_ips:
|
||||
return index
|
||||
raise RuntimeError("Unable to determine current node index")
|
||||
Reference in New Issue
Block a user