commit 7a7c4986b9cba98db8288b9b754f7e21d737f142 Author: zhousha <736730048@qq.com> Date: Tue Sep 29 15:39:44 2026 +0800 Add NVIDIA vLLM image and ModelHub release workflow diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..4d9b1a2 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,2 @@ +** +!Dockerfile diff --git a/.gitea/workflows/docker-build-push.yml b/.gitea/workflows/docker-build-push.yml new file mode 100644 index 0000000..a844b5a --- /dev/null +++ b/.gitea/workflows/docker-build-push.yml @@ -0,0 +1,133 @@ +name: Docker Build and Push + +on: + push: + tags: + - "v*" + +jobs: + docker: + runs-on: amd64-ubuntu-24.04 + + steps: + - name: Clone repository + run: | + git clone "${{ gitea.server_url }}/${{ gitea.repository }}.git" . + git checkout "${{ gitea.ref_name }}" + + - name: Set image metadata + run: | + IMAGE_NAME="$(echo "${{ gitea.repository }}" | tr '[:upper:]' '[:lower:]' | tr '_' '-')" + IMAGE="${DOCKER_REGISTRY}/${DOCKER_USERNAME}/${IMAGE_NAME}:${{ gitea.ref_name }}" + + echo "IMAGE_NAME=${IMAGE_NAME}" >> "$GITEA_ENV" + echo "IMAGE=${IMAGE}" >> "$GITEA_ENV" + + - name: Load and Validate Task Info + run: | + set -a + . .gitea/workflows/task_info.env + set +a + + for name in FRAMEWORK GPU_TYPE TASK_TYPE; do + eval "value=\${${name}:-}" + if [ "$name" = "FRAMEWORK" ] && [ -z "$value" ]; then + echo "${name} is empty in .gitea/workflows/task_info.env" + exit 1 + fi + + echo "${name}=${value}" >> "$GITEA_ENV" + done + + - name: Validate Image Verify Metadata + run: | + if [ -z "${FIXED_TOKEN:-}" ]; then + echo "FIXED_TOKEN is not configured on runner" + exit 1 + fi + + if ! response="$(curl --silent --show-error --location --get 'https://modelhub.org.cn/adminApi/image-verify/validate' \ + --header "Xc-Token: ${FIXED_TOKEN}" \ + --data-urlencode "gpuType=${GPU_TYPE:-}" \ + --data-urlencode "taskType=${TASK_TYPE:-}")"; then + echo "failed to call image verify validate API" + exit 1 + fi + + VALIDATE_RESPONSE="$response" python3 - <<'PY' + import json + import os + import sys + + raw = os.environ.get("VALIDATE_RESPONSE", "") + try: + body = json.loads(raw) + except json.JSONDecodeError: + print("image verify validate API returned invalid JSON") + print(raw) + sys.exit(1) + + if body.get("code") == 0 and body.get("data") is True: + print("image verify metadata validation passed") + sys.exit(0) + + message = body.get("message") or "unknown error" + print(f"image verify metadata validation failed: {message}") + print(raw) + sys.exit(1) + PY + + - name: Login to Docker Registry + run: | + echo "$DOCKER_PASSWORD" | docker login "$DOCKER_REGISTRY" \ + -u "$DOCKER_USERNAME" \ + --password-stdin + + - name: Build Docker Image + run: | + docker build -t "$IMAGE" . + + - name: Push Docker Image + run: | + for attempt in 1 2 3; do + echo "Starting docker push attempt ${attempt}/3 for ${IMAGE}" + docker push "$IMAGE" & + PUSH_PID=$! + + while kill -0 "$PUSH_PID" 2>/dev/null; do + echo "docker push is still running at $(date -u '+%Y-%m-%dT%H:%M:%SZ')" + sleep 60 + done + + if wait "$PUSH_PID"; then + echo "docker push completed successfully" + exit 0 + fi + + echo "docker push failed on attempt ${attempt}/3" + sleep 30 + done + + echo "docker push failed after 3 attempts" + exit 1 + + - name: Notify Image Verify + run: | + if [ -z "${FIXED_TOKEN:-}" ]; then + echo "FIXED_TOKEN is not configured on runner" + exit 1 + fi + + curl --silent --show-error --fail-with-body --location --request POST 'https://modelhub.org.cn//adminApi/image-verify' \ + --header "Xc-Token: ${FIXED_TOKEN}" \ + --header 'Content-Type: application/json' \ + --data-raw "{ + \"framework\": \"${FRAMEWORK}\", + \"gpuType\": \"${GPU_TYPE}\", + \"imageUrl\": \"${IMAGE}\", + \"taskType\": \"${TASK_TYPE}\", + \"createBy\": \"${{ gitea.actor }}\", + \"repoUrl\": \"${{ gitea.server_url }}/${{ gitea.repository }}\", + \"tag\": \"${{ github.ref_name }}\" + }" + diff --git a/.gitea/workflows/task_info.env b/.gitea/workflows/task_info.env new file mode 100644 index 0000000..bdbc5ca --- /dev/null +++ b/.gitea/workflows/task_info.env @@ -0,0 +1,3 @@ +FRAMEWORK=vllm +GPU_TYPE="NVIDIA H800" +TASK_TYPE=text-generation diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..e9838a7 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,5 @@ +FROM harbor.4pd.io/hardcore-tech/vllm/vllm-openai:v0.25.0-cu129 + +# Text inference does not require the incompatible TorchCodec video decoder. +RUN python3 -m pip uninstall -y torchcodec +RUN python3 -c "import vllm.entrypoints.openai.api_server; print('API server import OK')" diff --git a/README.md b/README.md new file mode 100644 index 0000000..4a69ff7 --- /dev/null +++ b/README.md @@ -0,0 +1,21 @@ +# Step-3.5-Flash NVIDIA inference image + +Rebuilds the text-inference environment validated on seven NVIDIA H800 GPUs: +vLLM 0.25.0, CUDA 12.9, BF16, tensor parallelism 1 and pipeline parallelism 7. + +TorchCodec is removed because the original image fails to import it with a +missing `libnvrtc.so.13` dependency. The build verifies the API server import. +Model weights are not included. Mount them and supply serving arguments at runtime. + +## ModelHub release + +The workflow is copied from https://dev.modelhub.org.cn/4pdadmin/cicd_demo. +Push a new `v*` Git tag to trigger image build, push, and review submission. +The runner supplies `DOCKER_REGISTRY`, `DOCKER_USERNAME`, `DOCKER_PASSWORD`, +and `FIXED_TOKEN`. It must be able to pull the Harbor base image. +ModelHub validates `GPU_TYPE="NVIDIA H800"` and `TASK_TYPE=text-generation` +before building. Approval is required before selecting the image for evaluation. + +The image inherits its base image's entrypoint; the validated deployment overrides +it with `python3 -m vllm.entrypoints.openai.api_server`. Docker run options and +host model paths are not embedded into this image.