Add NVIDIA vLLM image and ModelHub release workflow
Some checks failed
Docker Build and Push / docker (push) Failing after 2s
Some checks failed
Docker Build and Push / docker (push) Failing after 2s
This commit is contained in:
2
.dockerignore
Normal file
2
.dockerignore
Normal file
@@ -0,0 +1,2 @@
|
|||||||
|
**
|
||||||
|
!Dockerfile
|
||||||
133
.gitea/workflows/docker-build-push.yml
Normal file
133
.gitea/workflows/docker-build-push.yml
Normal file
@@ -0,0 +1,133 @@
|
|||||||
|
name: Docker Build and Push
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags:
|
||||||
|
- "v*"
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
docker:
|
||||||
|
runs-on: amd64-ubuntu-24.04
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Clone repository
|
||||||
|
run: |
|
||||||
|
git clone "${{ gitea.server_url }}/${{ gitea.repository }}.git" .
|
||||||
|
git checkout "${{ gitea.ref_name }}"
|
||||||
|
|
||||||
|
- name: Set image metadata
|
||||||
|
run: |
|
||||||
|
IMAGE_NAME="$(echo "${{ gitea.repository }}" | tr '[:upper:]' '[:lower:]' | tr '_' '-')"
|
||||||
|
IMAGE="${DOCKER_REGISTRY}/${DOCKER_USERNAME}/${IMAGE_NAME}:${{ gitea.ref_name }}"
|
||||||
|
|
||||||
|
echo "IMAGE_NAME=${IMAGE_NAME}" >> "$GITEA_ENV"
|
||||||
|
echo "IMAGE=${IMAGE}" >> "$GITEA_ENV"
|
||||||
|
|
||||||
|
- name: Load and Validate Task Info
|
||||||
|
run: |
|
||||||
|
set -a
|
||||||
|
. .gitea/workflows/task_info.env
|
||||||
|
set +a
|
||||||
|
|
||||||
|
for name in FRAMEWORK GPU_TYPE TASK_TYPE; do
|
||||||
|
eval "value=\${${name}:-}"
|
||||||
|
if [ "$name" = "FRAMEWORK" ] && [ -z "$value" ]; then
|
||||||
|
echo "${name} is empty in .gitea/workflows/task_info.env"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "${name}=${value}" >> "$GITEA_ENV"
|
||||||
|
done
|
||||||
|
|
||||||
|
- name: Validate Image Verify Metadata
|
||||||
|
run: |
|
||||||
|
if [ -z "${FIXED_TOKEN:-}" ]; then
|
||||||
|
echo "FIXED_TOKEN is not configured on runner"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if ! response="$(curl --silent --show-error --location --get 'https://modelhub.org.cn/adminApi/image-verify/validate' \
|
||||||
|
--header "Xc-Token: ${FIXED_TOKEN}" \
|
||||||
|
--data-urlencode "gpuType=${GPU_TYPE:-}" \
|
||||||
|
--data-urlencode "taskType=${TASK_TYPE:-}")"; then
|
||||||
|
echo "failed to call image verify validate API"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
VALIDATE_RESPONSE="$response" python3 - <<'PY'
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
raw = os.environ.get("VALIDATE_RESPONSE", "")
|
||||||
|
try:
|
||||||
|
body = json.loads(raw)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
print("image verify validate API returned invalid JSON")
|
||||||
|
print(raw)
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
if body.get("code") == 0 and body.get("data") is True:
|
||||||
|
print("image verify metadata validation passed")
|
||||||
|
sys.exit(0)
|
||||||
|
|
||||||
|
message = body.get("message") or "unknown error"
|
||||||
|
print(f"image verify metadata validation failed: {message}")
|
||||||
|
print(raw)
|
||||||
|
sys.exit(1)
|
||||||
|
PY
|
||||||
|
|
||||||
|
- name: Login to Docker Registry
|
||||||
|
run: |
|
||||||
|
echo "$DOCKER_PASSWORD" | docker login "$DOCKER_REGISTRY" \
|
||||||
|
-u "$DOCKER_USERNAME" \
|
||||||
|
--password-stdin
|
||||||
|
|
||||||
|
- name: Build Docker Image
|
||||||
|
run: |
|
||||||
|
docker build -t "$IMAGE" .
|
||||||
|
|
||||||
|
- name: Push Docker Image
|
||||||
|
run: |
|
||||||
|
for attempt in 1 2 3; do
|
||||||
|
echo "Starting docker push attempt ${attempt}/3 for ${IMAGE}"
|
||||||
|
docker push "$IMAGE" &
|
||||||
|
PUSH_PID=$!
|
||||||
|
|
||||||
|
while kill -0 "$PUSH_PID" 2>/dev/null; do
|
||||||
|
echo "docker push is still running at $(date -u '+%Y-%m-%dT%H:%M:%SZ')"
|
||||||
|
sleep 60
|
||||||
|
done
|
||||||
|
|
||||||
|
if wait "$PUSH_PID"; then
|
||||||
|
echo "docker push completed successfully"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "docker push failed on attempt ${attempt}/3"
|
||||||
|
sleep 30
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "docker push failed after 3 attempts"
|
||||||
|
exit 1
|
||||||
|
|
||||||
|
- name: Notify Image Verify
|
||||||
|
run: |
|
||||||
|
if [ -z "${FIXED_TOKEN:-}" ]; then
|
||||||
|
echo "FIXED_TOKEN is not configured on runner"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
curl --silent --show-error --fail-with-body --location --request POST 'https://modelhub.org.cn//adminApi/image-verify' \
|
||||||
|
--header "Xc-Token: ${FIXED_TOKEN}" \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--data-raw "{
|
||||||
|
\"framework\": \"${FRAMEWORK}\",
|
||||||
|
\"gpuType\": \"${GPU_TYPE}\",
|
||||||
|
\"imageUrl\": \"${IMAGE}\",
|
||||||
|
\"taskType\": \"${TASK_TYPE}\",
|
||||||
|
\"createBy\": \"${{ gitea.actor }}\",
|
||||||
|
\"repoUrl\": \"${{ gitea.server_url }}/${{ gitea.repository }}\",
|
||||||
|
\"tag\": \"${{ github.ref_name }}\"
|
||||||
|
}"
|
||||||
|
|
||||||
3
.gitea/workflows/task_info.env
Normal file
3
.gitea/workflows/task_info.env
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
FRAMEWORK=vllm
|
||||||
|
GPU_TYPE="NVIDIA H800"
|
||||||
|
TASK_TYPE=text-generation
|
||||||
5
Dockerfile
Normal file
5
Dockerfile
Normal file
@@ -0,0 +1,5 @@
|
|||||||
|
FROM harbor.4pd.io/hardcore-tech/vllm/vllm-openai:v0.25.0-cu129
|
||||||
|
|
||||||
|
# Text inference does not require the incompatible TorchCodec video decoder.
|
||||||
|
RUN python3 -m pip uninstall -y torchcodec
|
||||||
|
RUN python3 -c "import vllm.entrypoints.openai.api_server; print('API server import OK')"
|
||||||
21
README.md
Normal file
21
README.md
Normal file
@@ -0,0 +1,21 @@
|
|||||||
|
# Step-3.5-Flash NVIDIA inference image
|
||||||
|
|
||||||
|
Rebuilds the text-inference environment validated on seven NVIDIA H800 GPUs:
|
||||||
|
vLLM 0.25.0, CUDA 12.9, BF16, tensor parallelism 1 and pipeline parallelism 7.
|
||||||
|
|
||||||
|
TorchCodec is removed because the original image fails to import it with a
|
||||||
|
missing `libnvrtc.so.13` dependency. The build verifies the API server import.
|
||||||
|
Model weights are not included. Mount them and supply serving arguments at runtime.
|
||||||
|
|
||||||
|
## ModelHub release
|
||||||
|
|
||||||
|
The workflow is copied from https://dev.modelhub.org.cn/4pdadmin/cicd_demo.
|
||||||
|
Push a new `v*` Git tag to trigger image build, push, and review submission.
|
||||||
|
The runner supplies `DOCKER_REGISTRY`, `DOCKER_USERNAME`, `DOCKER_PASSWORD`,
|
||||||
|
and `FIXED_TOKEN`. It must be able to pull the Harbor base image.
|
||||||
|
ModelHub validates `GPU_TYPE="NVIDIA H800"` and `TASK_TYPE=text-generation`
|
||||||
|
before building. Approval is required before selecting the image for evaluation.
|
||||||
|
|
||||||
|
The image inherits its base image's entrypoint; the validated deployment overrides
|
||||||
|
it with `python3 -m vllm.entrypoints.openai.api_server`. Docker run options and
|
||||||
|
host model paths are not embedded into this image.
|
||||||
Reference in New Issue
Block a user