diff --git a/pack/.post_operation/20260920_vllm_install_router/cuda/Dockerfile b/pack/.post_operation/20260920_vllm_install_router/cuda/Dockerfile new file mode 100644 index 0000000..4aa41d5 --- /dev/null +++ b/pack/.post_operation/20260920_vllm_install_router/cuda/Dockerfile @@ -0,0 +1,61 @@ +ARG CMAKE_MAX_JOBS +ARG CUDA_VERSION=12.9 +ARG VLLM_VERSION=0.27.1 +ARG VLLM_ROUTER_VERSION=0.1.15 + +FROM gpustack/runner:cuda${CUDA_VERSION}-vllm${VLLM_VERSION} AS vllm +SHELL ["/bin/bash", "-eo", "pipefail", "-c"] + +ARG TARGETPLATFORM +ARG TARGETOS +ARG TARGETARCH + +## Install vLLM Router +## +## `pack/cuda/Dockerfile.vllm` only gained the router after 0.27.1 was cut, so the released 0.27.1 +## images carry none of it: a disaggregated group deployed on them has no router to run, and either +## falls back to an engine example script (no metrics, no circuit breaker, no health check) or does +## not start at all. 0.29.0 already ships it, which is why 0.27.1 is the only CUDA runner here. +## +## Pinned to the same 0.1.15 the recipe installs, so that a 0.27.1 image and a 0.29.0 image render +## the same router. The wheel is prebuilt for both aarch64 and x86_64, and it depends on nothing +## but pure-Python web packages (fastapi, uvicorn, aiohttp, orjson, requests, setproctitle) -- it +## does not depend on vLLM, so this install cannot move the torch/vLLM stack already in the image. + +ARG VLLM_ROUTER_VERSION + +RUN </dev/null + + # Review + uv pip tree + + # Cleanup + rm -rf /var/tmp/* \ + && rm -rf /tmp/* +EOF + +## Probe Dependencies + +ARG DEPENDENCY_PACKAGES="" +RUN --mount=type=bind,from=shared,source=probe_dependencies.sh,target=/tmp/probe_dependencies.sh \ + DEPENDENCY_PACKAGES="${DEPENDENCY_PACKAGES}" bash /tmp/probe_dependencies.sh + +## Entrypoint + +WORKDIR / +ENTRYPOINT [ "tini", "--" ] + +## Export Dependencies + +FROM scratch AS vllm-deps + +COPY --from=vllm /etc/gpustack-runner/dependencies.json / diff --git a/pack/.post_operation/20260920_vllm_install_router/matrix.yaml b/pack/.post_operation/20260920_vllm_install_router/matrix.yaml new file mode 100644 index 0000000..73c6959 --- /dev/null +++ b/pack/.post_operation/20260920_vllm_install_router/matrix.yaml @@ -0,0 +1,32 @@ +rules: + + # CUDA 0.27.1 only, and deliberately so. + # + # The router install lives in `pack/cuda/Dockerfile.vllm` and `pack/cann/Dockerfile.vllm` -- the + # two recipes GPUStack renders a PD mode for. Of the released images that miss it: + # + # - cuda 0.27.1 is no longer in `pack/matrix.yaml` (CUDA stands at 0.29.0, which already ships + # the router), so rewriting the tag here is the only way to put a router into it. + # - cann 0.23.0 is still exactly what `pack/matrix.yaml` builds, and the recipe already carries + # the router, so a normal `for_release` pack of backend `cann` rebuilds those four variants + # with it. That is a plain release rather than a mutation, so it does not belong here. + # - rocm never installed a router at all; its 0.29.0 image has none either, and patching only + # 0.27.1 would leave the newer image the odd one out. + + ## Packed NVIDIA CUDA 13.0. + ## + - backend: "cuda" + services: + - "vllm" + args: + - "CUDA_VERSION=13.0" + - "VLLM_VERSION=0.27.1" + + ## Packed NVIDIA CUDA 12.9. + ## + - backend: "cuda" + services: + - "vllm" + args: + - "CUDA_VERSION=12.9" + - "VLLM_VERSION=0.27.1" diff --git a/pack/.post_operation/README.md b/pack/.post_operation/README.md index ce7bd43..ac1e69a 100644 --- a/pack/.post_operation/README.md +++ b/pack/.post_operation/README.md @@ -101,3 +101,4 @@ mutated stay as they were until the next release rebuilds those images. - [x] 2026-03-03: Fix malformed ARM64 image for vLLM 0.15.1 of CUDA released images. - [x] 2026-09-01: Pin `numpy` to 1.26.4 and remove CUDA-only NIXL EP packages for vLLM 0.18.1 of DTK 26.04 released images. - [x] 2026-09-16: Patch vLLM 0.24.0/0.25.1/0.27.1/0.29.0 of CUDA/ROCm released images to fix mooncake prom metrics issue. +- [ ] 2026-09-20: Install `vllm-router` package for vLLM 0.27.1 of CUDA released images.