diff --git a/README.md b/README.md index e4397def..44bfb341 100644 --- a/README.md +++ b/README.md @@ -56,8 +56,8 @@ The following table lists the supported accelerated backends and their correspon | CUDA Version
(Variant) | vLLM | SGLang | VoxBox | |------------------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------------------------------------|----------| -| 13.0 | **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1`,
`0.18.1` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1` | | -| 12.9 | **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1`,
`0.18.1`, `0.17.1`,
`0.16.0`, `0.15.1`,
`0.14.1`, `0.13.0`,
`0.12.0`, `0.11.2` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1`,
`0.5.9`, `0.5.8.post1`,
`0.5.7`, `0.5.6.post2` | | +| 13.0 | **`0.30.0`**, **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1`,
`0.18.1` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1` | | +| 12.9 | **`0.30.0`**, **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1`,
`0.18.1`, `0.17.1`,
`0.16.0`, `0.15.1`,
`0.14.1`, `0.13.0`,
`0.12.0`, `0.11.2` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1`,
`0.5.9`, `0.5.8.post1`,
`0.5.7`, `0.5.6.post2` | | | 12.8 | `0.17.1`, `0.16.0`,
`0.15.1`, `0.14.1`,
`0.13.0`, `0.12.0`,
`0.11.2`, `0.10.2` | `0.5.9`, `0.5.8.post1`,
`0.5.7`, `0.5.6.post2`,
`0.5.5.post3` | `0.0.21` | | 12.6 | `0.15.1`, `0.14.1`,
`0.13.0`, `0.12.0`,
`0.11.2`, `0.10.2` | | `0.0.21` | @@ -96,7 +96,7 @@ The following table lists the supported accelerated backends and their correspon | ROCm Version
(Variant) | vLLM | SGLang | |------------------------------|---------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------| -| 7.2 | **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1` | +| 7.2 | **`0.30.0`**, **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1` | | 7.1 | `0.17.1` | | | 7.0 | `0.18.1`, `0.16.0`,
`0.15.1`, `0.14.1`,
`0.13.0`, `0.12.0`,
`0.11.2` | `0.5.9`, `0.5.8.post1`,
`0.5.7`, `0.5.6.post2` | | 6.4 | `0.16.0`, `0.15.1`,
`0.14.1`, `0.13.0`,
`0.12.0`, `0.11.2`,
`0.10.2` | `0.5.8.post1`, `0.5.7`,
`0.5.6.post2`, `0.5.5.post3` | diff --git a/pack/cuda/Dockerfile.sglang b/pack/cuda/Dockerfile.sglang index 930f841e..7f747b6f 100644 --- a/pack/cuda/Dockerfile.sglang +++ b/pack/cuda/Dockerfile.sglang @@ -9,7 +9,7 @@ ARG SGLANG_TORCH_CUDA_VERSION=${CUDA_VERSION} ## Keep >= 0.4.6 and in sync with the ROCm and vLLM pins: SGLang's lmc_radix_cache.py imports ## `lmcache.integration.sglang.multi_process_adapter`, added in 0.4.6, and the MP wire protocol ## has no version handshake. The SGLang base ships no lmcache, so it is always built from source. -ARG SGLANG_LMCACHE_VERSION=0.5.4 +ARG SGLANG_LMCACHE_VERSION=0.5.5 ARG SGLANG_NVIDIA_HPCX_VERSION=2.24.1_cuda13 ARG SGLANG_AWS_EFA_VERSION=1.46.0 @@ -192,6 +192,10 @@ RUN < cuda-bindings<14,>=13.0.3` requirement. + ## 0.5.5 generates its gRPC stubs at build time (setup.py _BuildPyWithGrpcStubs); + ## provide the pinned codegen tools its build-system requirements name, since + ## `--no-isolation` skips installing them. + uv pip install grpcio==1.84.0 grpcio-tools==1.84.0 pushd /tmp/lmcache \ && sed -i "s/\"torch==.*\"/\"torch\"/g" /tmp/lmcache/pyproject.toml \ && cat /tmp/lmcache/pyproject.toml \ @@ -373,7 +377,7 @@ RUN < cuda-bindings<14,>=13.0.3` requirement. ## Note: separate statements, not an `&&` chain —— `bash -e` ignores a failing non-final ## command of a list, which would leave /workspace empty and fail later at install time. + ## 0.5.5 generates its gRPC stubs at build time (setup.py _BuildPyWithGrpcStubs); + ## provide the pinned codegen tools its build-system requirements name, since + ## `--no-isolation` skips installing them. + uv pip install grpcio==1.84.0 grpcio-tools==1.84.0 cd /tmp/lmcache ## Drop the build-time torch pin: we compile against the base's own torch on purpose. sed -i "s/\"torch==.*\"/\"torch\"/g" /tmp/lmcache/pyproject.toml cat /tmp/lmcache/pyproject.toml + ## torch 2.14's headers require C++20 (#error in torch/all.h), but lmcache 0.5.5 + ## hardcodes -std=c++17 for its g++ sources (pybind.cpp); nvcc already defaults + ## to C++20 under CUDA 13. Harmless on the cu129 base's torch 2.13. + sed -i 's/-std=c++17/-std=c++20/g' \ + /tmp/lmcache/setup_extensions/common_cpp.py \ + /tmp/lmcache/setup_extensions/build_profiles/cuda.py ## `--skip-dependency-check`: the base does not ship torch's full dependency closure. python -m build --no-isolation --skip-dependency-check --wheel tree -hs /tmp/lmcache/dist @@ -473,7 +532,7 @@ soundfile mistral_common[audio] # tokenizer extras -fastokens==0.2.0 +fastokens==0.3.2 EOT uv pip install \ -r /tmp/requirements.txt diff --git a/pack/matrix.yaml b/pack/matrix.yaml index 6032af33..cba5265b 100644 --- a/pack/matrix.yaml +++ b/pack/matrix.yaml @@ -101,8 +101,8 @@ rules: - "linux/amd64" args: - "ROCM_VERSION=7.2.3" - - "VLLM_VERSION=0.29.0" - - "VLLM_BASE_IMAGE=vllm/vllm-openai-rocm:v0.29.0" + - "VLLM_VERSION=0.30.0" + - "VLLM_BASE_IMAGE=vllm/vllm-openai-rocm:v0.30.0" ## AMD ROCm 7.2.1 - SGLang 0.5.18 (rocm720-mi30x) ## - backend: "rocm" @@ -119,15 +119,17 @@ rules: # NVIDIA CUDA # - ## NVIDIA CUDA 13.0.1 + ## NVIDIA CUDA 13.0.3 + ## (torch is upgraded from the base's 2.13.0 to 2.14.0 —— see the + ## "Upgrade Torch" step in pack/cuda/Dockerfile.vllm) ## - backend: "cuda" services: - "vllm" args: - - "CUDA_VERSION=13.0.1" - - "VLLM_VERSION=0.29.0" - - "VLLM_BASE_IMAGE=vllm/vllm-openai:v0.29.0-ubuntu2404" + - "CUDA_VERSION=13.0.3" + - "VLLM_VERSION=0.30.0" + - "VLLM_BASE_IMAGE=vllm/vllm-openai:v0.30.0-ubuntu2404" ## NVIDIA CUDA 13.0.1 - SGLang 0.5.18 ## - backend: "cuda" @@ -138,14 +140,18 @@ rules: - "SGLANG_VERSION=0.5.18" - "SGLANG_BASE_IMAGE=lmsysorg/sglang:v0.5.18-cu130" ## NVIDIA CUDA 12.9.1 + ## (stays on the base's torch 2.13.0: 2.14.0 is published for cu130 only, + ## so VLLM_TORCH_VERSION is pinned to the base's version and the + ## "Upgrade Torch" step in pack/cuda/Dockerfile.vllm is a no-op) ## - backend: "cuda" services: - "vllm" args: - "CUDA_VERSION=12.9.1" - - "VLLM_VERSION=0.29.0" - - "VLLM_BASE_IMAGE=vllm/vllm-openai:v0.29.0-cu129-ubuntu2404" + - "VLLM_VERSION=0.30.0" + - "VLLM_BASE_IMAGE=vllm/vllm-openai:v0.30.0-cu129-ubuntu2404" + - "VLLM_TORCH_VERSION=2.13.0" ## NVIDIA CUDA 12.9.1 - SGLang 0.5.18 ## - backend: "cuda" diff --git a/pack/rocm/Dockerfile.sglang b/pack/rocm/Dockerfile.sglang index 426280b4..6022e1cd 100644 --- a/pack/rocm/Dockerfile.sglang +++ b/pack/rocm/Dockerfile.sglang @@ -8,7 +8,7 @@ ARG SGLANG_TORCH_VERSION=2.9.1 ARG SGLANG_TORCH_ROCM_VERSION=${ROCM_VERSION} ## Keep >= 0.4.6 and in sync with the CUDA pin —— see pack/cuda/Dockerfile.sglang. ## Always built from source: upstream's ROCm wheel targets torch 2.11.0+rocm7.2. -ARG SGLANG_LMCACHE_VERSION=0.5.4 +ARG SGLANG_LMCACHE_VERSION=0.5.5 FROM ${SGLANG_BASE_IMAGE} AS sglang-build SHELL ["/bin/bash", "-eo", "pipefail", "-c"] @@ -198,6 +198,10 @@ RUN <