init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -0,0 +1,59 @@
#!/bin/bash
# ============================================================
# aop_capture.sh - Capture test results from log files
#
# Called from _e2e_nightly_single_node.yaml [AOP] steps.
# Writes outputs to $GITHUB_OUTPUT.
#
# Usage: aop_capture.sh <yaml_outcome> <pytest_outcome>
# ============================================================
set -euo pipefail
YAML_OUTCOME="$1"
PYTEST_OUTCOME="$2"
LOG_DIR="/tmp/test-logs"
echo "============================================"
echo " Test Result Summary"
echo " YAML-driven : ${YAML_OUTCOME:-skipped}"
echo " Pytest-driven: ${PYTEST_OUTCOME:-skipped}"
echo "============================================"
parse_log() {
local log_file="$1"
local prefix="$2"
if [ ! -f "$log_file" ]; then
return 0
fi
echo ""
echo "--- ${prefix} tail (last 40 lines) ---"
tail -n 40 "$log_file"
echo "--- end ---"
local summary
summary=$(grep -E '=+.*(passed|failed|error).*=+' "$log_file" | tail -1 || true)
echo "${prefix}_summary=${summary}" >> "$GITHUB_OUTPUT"
echo "${prefix}_summary: ${summary}"
local failures
failures=$(grep -c 'FAILED' "$log_file" || true)
echo "${prefix}_failures=${failures}" >> "$GITHUB_OUTPUT"
}
parse_log "${LOG_DIR}/pytest-driven.log" "pytest"
parse_log "${LOG_DIR}/yaml-test.log" "yaml"
# Final verdict + which test failed
if [ "$YAML_OUTCOME" = "failure" ] || [ "$PYTEST_OUTCOME" = "failure" ]; then
echo "result=failure" >> "$GITHUB_OUTPUT"
FAILED=""
[ "$YAML_OUTCOME" = "failure" ] && FAILED="${FAILED}yaml,"
[ "$PYTEST_OUTCOME" = "failure" ] && FAILED="${FAILED}pytest,"
echo "failed_test=${FAILED%,}" >> "$GITHUB_OUTPUT"
elif [ "$YAML_OUTCOME" = "success" ] || [ "$PYTEST_OUTCOME" = "success" ]; then
echo "result=success" >> "$GITHUB_OUTPUT"
else
echo "result=skipped" >> "$GITHUB_OUTPUT"
fi

View File

@@ -0,0 +1,60 @@
#!/bin/bash
# ============================================================
# aop_classify.sh - Check if failure is environmental
#
# Args: $1 = failed_test (from capture: "yaml", "pytest", "yaml,pytest")
#
# Only scans the log files that actually failed.
# Writes failure_type to $GITHUB_OUTPUT.
# ============================================================
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
RULES="$SCRIPT_DIR/rules-env.txt"
LOG_DIR="/tmp/test-logs"
FAILED_TEST="${1:-yaml,pytest}"
check_log() {
local log_file="$1"
local label="$2"
if [ ! -f "$log_file" ] || [ ! -s "$log_file" ]; then
echo " [$label] log empty/missing -> skipped"
return 0
fi
local count
count=$(grep -vE '^[[:space:]]*(#|$)' "$RULES" | grep -ciEf - "$log_file" 2>/dev/null || echo 0)
echo " [$label] env patterns matched: ${count}"
if [ "$count" -gt 0 ]; then
echo " [$label] --- matches ---"
grep -vE '^[[:space:]]*(#|$)' "$RULES" | grep -niEf - "$log_file" | head -10
return 1
fi
return 0
}
echo "=== Failure Classification ==="
echo "[DEBUG] rules file : ${RULES}"
echo "[DEBUG] rules exists: $(test -f "$RULES" && echo yes || echo no)"
echo "[DEBUG] rules lines : $(grep -c . "$RULES" 2>/dev/null || echo 0)"
echo "[DEBUG] rules content:"
cat -n "$RULES" 2>/dev/null || echo "(file not found)"
ENV_FOUND=0
if [[ "$FAILED_TEST" == *pytest* ]]; then
check_log "${LOG_DIR}/pytest-driven.log" "pytest-driven" || ENV_FOUND=1
fi
if [[ "$FAILED_TEST" == *yaml* ]]; then
check_log "${LOG_DIR}/yaml-test.log" "yaml-test" || ENV_FOUND=1
fi
if [ "$ENV_FOUND" -eq 1 ]; then
echo "=== Result: env_failure ==="
echo "failure_type=env_failure" >> "$GITHUB_OUTPUT"
else
echo "=== Result: not_env_failure ==="
echo "failure_type=not_env_failure" >> "$GITHUB_OUTPUT"
fi

View File

@@ -0,0 +1,98 @@
#!/bin/bash
# ============================================================
# aop_commit_age.sh - Look up last successful time in good_table.csv
#
# CSV format (good_table.csv):
# name,yaml/path,link,status,vLLM Git information,vLLM-Ascend Git information,time
#
# Finds rows matching config_name where status=success,
# picks the most recent "time" column, and calculates age.
#
# Args: $1 = config_name
# $2 = csv_path
#
# Writes to $GITHUB_OUTPUT:
# commit_age_days - days since last success
# is_old - true if > 3 days
# last_status - status from table
# last_date - date from table
# ============================================================
set -euo pipefail
CONFIG_NAME="${1:-}"
CSV_PATH="${GOOD_TABLE:-$2}"
if [ -z "$CONFIG_NAME" ]; then
echo "ERROR: no config name provided"
exit 1
fi
echo ">>> Looking up config : ${CONFIG_NAME}"
echo ">>> CSV path : ${CSV_PATH}"
if [ ! -f "$CSV_PATH" ]; then
echo ">>> CSV not found → skip"
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
exit 0
fi
# Find matching rows, only consider success rows (match name column only)
ROWS=$(grep "^${CONFIG_NAME}," "$CSV_PATH" | grep -F ',success,' || true)
if [ -z "$ROWS" ]; then
echo ">>> No success row for '${CONFIG_NAME}' → skip"
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
exit 0
fi
# Pick most recent success row
BEST_ROW=""
BEST_DATE=""
while IFS= read -r row; do
date_str=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
[ -z "$date_str" ] && continue
if [ -z "$BEST_DATE" ] || [[ "$date_str" > "$BEST_DATE" ]]; then
BEST_ROW="$row"
BEST_DATE="$date_str"
fi
done <<< "$ROWS"
if [ -z "$BEST_ROW" ]; then
echo ">>> No valid date in success rows → skip"
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
exit 0
fi
LAST_STATUS=$(echo "$BEST_ROW" | awk -F',' '{print $4}' | xargs)
LAST_DATE="$BEST_DATE"
echo ">>> Matched row: ${BEST_ROW}"
echo "last_status=${LAST_STATUS}" >> "$GITHUB_OUTPUT"
echo "last_date=${LAST_DATE}" >> "$GITHUB_OUTPUT"
LAST_TS=$(date -d "$LAST_DATE" +%s 2>/dev/null || true)
if [ -z "$LAST_TS" ]; then
echo ">>> Could not parse date: ${LAST_DATE} → skip"
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
exit 0
fi
NOW=$(date +%s)
AGE_DAYS=$(( (NOW - LAST_TS) / 86400 ))
echo "commit_age_days=${AGE_DAYS}" >> "$GITHUB_OUTPUT"
if [ "$AGE_DAYS" -gt 3 ]; then
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo ">>> ${CONFIG_NAME} last_status=${LAST_STATUS} date=${LAST_DATE} age=${AGE_DAYS}d (> 3 days) → old"
else
echo "is_old=false" >> "$GITHUB_OUTPUT"
echo ">>> ${CONFIG_NAME} last_status=${LAST_STATUS} date=${LAST_DATE} age=${AGE_DAYS}d (<= 3 days) → recent"
fi

View File

@@ -0,0 +1,93 @@
#!/bin/bash
# ============================================================
# aop_process.sh - Handle a recent real failure + auto bisect
#
# Args:
# $1 failure_type
# $2 commit_age_days
# $3 runner
# $4 tests
# $5 config_file_path
# $6 pytest_summary
# $7 yaml_summary
# $8 scene (single_node | multi_node)
# $9 bad_commit (commit SHA, default HEAD)
# $10 num_nodes (multi_node only)
# $11 coord_dir (multi_node only)
# $12 case_name (optional)
# ============================================================
set -euo pipefail
FT="${1:-unknown}"
AGE="${2:-?}"
RUNNER="${3:-?}"
TESTS="${4:-}"
CONFIG="${5:-}"
PYTEST_SUMMARY="${6:-}"
YAML_SUMMARY="${7:-}"
SCENE="${8:-single_node}"
BAD_COMMIT="${9:-HEAD}"
NUM_NODES="${10:-}"
COORD_DIR="${11:-}"
NAME="${12:-}"
echo "================================================"
echo " PROCESS - needs attention"
echo " Failure type : ${FT}"
echo " Commit age : ${AGE} days"
echo " Runner : ${RUNNER}"
echo " Tests : ${TESTS:-N/A}"
echo " Config : ${CONFIG:-N/A}"
echo " Scene : ${SCENE}"
echo " Bad commit : ${BAD_COMMIT}"
echo " PyTest : ${PYTEST_SUMMARY:-N/A}"
echo " YAML : ${YAML_SUMMARY:-N/A}"
echo "================================================"
echo "::group::Failed test details"
for f in /tmp/test-logs/pytest-driven.log /tmp/test-logs/yaml-test.log /tmp/test-logs/multi-node.log; do
if [ -f "$f" ]; then
grep -A 10 'FAILED' "$f" || true
fi
done
echo "::endgroup::"
# =====================================================
# Auto bisect
# =====================================================
# Extract case_name if not provided (single_node requires it)
if [ -z "$NAME" ] && [ "$SCENE" = "single_node" ]; then
if [ -n "$TESTS" ]; then
# py-driven: tests/e2e/.../test_xxx.py → test_xxx
NAME=$(basename "$TESTS" .py)
elif [ -n "$CONFIG" ]; then
# YAML-driven: Qwen3-32B-Int8.yaml → Qwen3-32B-Int8
NAME=$(basename "$CONFIG" .yaml)
fi
if [ -z "$NAME" ]; then
echo "WARNING: could not extract case_name, bisect may fail"
else
echo "Extracted name: ${NAME}"
fi
fi
GOOD_TABLE="${GOOD_TABLE:-}"
BISECT_CMD=(
python -m tests.e2e.nightly.bisect.auto_bisect
--scene "${SCENE}"
--bad-commit "${BAD_COMMIT}"
--good-table "${GOOD_TABLE}"
)
[ -n "$CONFIG" ] && BISECT_CMD+=(--config-yaml "$CONFIG")
[ -n "$NAME" ] && BISECT_CMD+=(--name "$NAME")
[ -n "$NUM_NODES" ] && BISECT_CMD+=(--num-nodes "$NUM_NODES")
[ -n "$COORD_DIR" ] && BISECT_CMD+=(--coord-dir "$COORD_DIR")
echo ""
echo "=== Running auto bisect ==="
echo "${BISECT_CMD[@]}"
"${BISECT_CMD[@]}"

View File

@@ -0,0 +1,39 @@
#!/bin/bash
# ============================================================
# aop_skip.sh - Log skip reason and show failure details
#
# Args: failure_type last_status last_date age_days
# pytest_summary yaml_summary
# ============================================================
set -euo pipefail
FT="${1:-unknown}"
LAST_STATUS="${2:-?}"
LAST_DATE="${3:-?}"
AGE="${4:-?}"
PYTEST_SUMMARY="${5:-}"
YAML_SUMMARY="${6:-}"
case "$FT" in
env_failure) REASON="environment issue" ;;
*) REASON="last run > 3 days ago" ;;
esac
echo "================================================"
echo " SKIP - no further action"
echo " Failure type : ${FT}"
echo " Last status : ${LAST_STATUS}"
echo " Last date : ${LAST_DATE}"
echo " Age (days) : ${AGE}"
echo " Reason : ${REASON}"
echo " PyTest : ${PYTEST_SUMMARY:-N/A}"
echo " YAML : ${YAML_SUMMARY:-N/A}"
echo "================================================"
echo "::group::Failed test details"
for f in /tmp/test-logs/pytest-driven.log /tmp/test-logs/yaml-test.log; do
if [ -f "$f" ]; then
grep -A 10 'FAILED' "$f" || true
fi
done
echo "::endgroup::"

View File

@@ -0,0 +1,6 @@
# Environment failure patterns (network / hardware / infra)
# One regex per line. Lines starting with # are comments.
# Used by aop_classify.sh via grep -Ef
RuntimeError: Timeout
TimeoutError: Timed out waiting for engine core processes to start

View File

@@ -0,0 +1,155 @@
#!/usr/bin/env python3
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
"""Update /root/.cache/vllm-ascend/main/nightly/good_table.csv with a
successful test entry. Creates the file (with header) if it does not exist;
replaces the existing row for the same test name if it does.
CSV columns:
name, yaml/path, link, status,
vLLM Git information, vLLM-Ascend Git information, time
"""
import argparse
import csv
import os
import subprocess
from datetime import datetime, timedelta, timezone
HEADER = [
"name",
"yaml/path",
"link",
"status",
"vLLM Git information",
"vLLM-Ascend Git information",
"time",
]
def git_head(repo_dir: str) -> str:
try:
return subprocess.check_output(
["git", "rev-parse", "HEAD"],
cwd=repo_dir,
stderr=subprocess.DEVNULL,
text=True,
).strip()
except Exception:
return "N/A"
def current_timestamp() -> str:
tz = timezone(timedelta(hours=8))
ts = datetime.now(tz).strftime("%Y-%m-%d %H:%M:%S %z")
# Reformat +0800 → +08:00 to match existing CSV entries
return ts[:-2] + ":" + ts[-2:]
def load_rows(csv_path: str) -> list[list[str]]:
if not os.path.isfile(csv_path):
return []
with open(csv_path, newline="", encoding="utf-8") as f:
reader = csv.reader(f)
rows = list(reader)
# Drop the header row if present
if rows and rows[0] == HEADER:
rows = rows[1:]
return rows
def save_rows(csv_path: str, rows: list[list[str]]) -> None:
os.makedirs(os.path.dirname(csv_path), exist_ok=True)
with open(csv_path, "w", newline="", encoding="utf-8") as f:
writer = csv.writer(f)
writer.writerow(HEADER)
writer.writerows(rows)
_DEFAULT_SINGLE_NODE_CONFIG_BASE = "tests/e2e/nightly/single_node/models/configs"
_DEFAULT_MULTI_NODE_CONFIG_BASES = (
"tests/e2e/nightly/multi_node/internal_dp/config",
"tests/e2e/nightly/multi_node/external_dp/config",
)
def resolve_test_path(
test_path: str,
config_base_path: str,
scene: str = "single_node",
repo_dir: str = ".",
) -> str:
"""Return the full relative path for the yaml/path CSV column.
Upper-level workflows pass config_file_path as a bare filename
(e.g. ``Qwen3.5-27B-w8a8-A2.yaml``). When no directory component is
present we prepend the config base path so the CSV matches the format
used by the existing hand-curated good_table entries.
"""
if os.sep in test_path or "/" in test_path:
return test_path
if config_base_path.strip():
return f"{config_base_path.strip()}/{test_path}"
if scene == "multi_node":
for base in _DEFAULT_MULTI_NODE_CONFIG_BASES:
if os.path.isfile(os.path.join(repo_dir, base, test_path)):
return f"{base}/{test_path}"
return f"{_DEFAULT_MULTI_NODE_CONFIG_BASES[0]}/{test_path}"
return f"{_DEFAULT_SINGLE_NODE_CONFIG_BASE}/{test_path}"
def main() -> None:
parser = argparse.ArgumentParser(description="Update good_table.csv on test success")
parser.add_argument("--cache-csv", required=True)
parser.add_argument("--test-name", required=True)
parser.add_argument("--test-path", required=True)
parser.add_argument("--config-base-path", default="")
parser.add_argument("--scene", default="single_node", choices=["single_node", "multi_node"])
parser.add_argument("--run-link", required=True)
parser.add_argument("--vllm-dir", default="/vllm-workspace/vllm")
parser.add_argument("--vllm-ascend-dir", default="/vllm-workspace/vllm-ascend")
parser.add_argument("--vllm-ascend-version", default="")
parser.add_argument("--vllm-version", default="")
args = parser.parse_args()
vllm_hash = args.vllm_version.strip() or git_head(args.vllm_dir)
vllm_ascend_hash = args.vllm_ascend_version.strip() or git_head(args.vllm_ascend_dir)
timestamp = current_timestamp()
test_path = resolve_test_path(args.test_path, args.config_base_path, args.scene, args.vllm_ascend_dir)
new_row = [
args.test_name,
test_path,
args.run_link,
"success",
vllm_hash,
vllm_ascend_hash,
timestamp,
]
is_new = not os.path.isfile(args.cache_csv)
rows = load_rows(args.cache_csv)
rows = [r for r in rows if r and r[0] != args.test_name]
rows.append(new_row)
save_rows(args.cache_csv, rows)
action = "Created" if is_new else "Updated"
print(f">>> {action} {args.cache_csv}: name={args.test_name} status=success time={timestamp}")
if __name__ == "__main__":
main()