diff --git a/README.md b/README.md
index e4397def..44bfb341 100644
--- a/README.md
+++ b/README.md
@@ -56,8 +56,8 @@ The following table lists the supported accelerated backends and their correspon
| CUDA Version
(Variant) | vLLM | SGLang | VoxBox |
|------------------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------------------------------------|----------|
-| 13.0 | **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1`,
`0.18.1` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1` | |
-| 12.9 | **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1`,
`0.18.1`, `0.17.1`,
`0.16.0`, `0.15.1`,
`0.14.1`, `0.13.0`,
`0.12.0`, `0.11.2` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1`,
`0.5.9`, `0.5.8.post1`,
`0.5.7`, `0.5.6.post2` | |
+| 13.0 | **`0.30.0`**, **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1`,
`0.18.1` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1` | |
+| 12.9 | **`0.30.0`**, **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1`,
`0.18.1`, `0.17.1`,
`0.16.0`, `0.15.1`,
`0.14.1`, `0.13.0`,
`0.12.0`, `0.11.2` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1`,
`0.5.9`, `0.5.8.post1`,
`0.5.7`, `0.5.6.post2` | |
| 12.8 | `0.17.1`, `0.16.0`,
`0.15.1`, `0.14.1`,
`0.13.0`, `0.12.0`,
`0.11.2`, `0.10.2` | `0.5.9`, `0.5.8.post1`,
`0.5.7`, `0.5.6.post2`,
`0.5.5.post3` | `0.0.21` |
| 12.6 | `0.15.1`, `0.14.1`,
`0.13.0`, `0.12.0`,
`0.11.2`, `0.10.2` | | `0.0.21` |
@@ -96,7 +96,7 @@ The following table lists the supported accelerated backends and their correspon
| ROCm Version
(Variant) | vLLM | SGLang |
|------------------------------|---------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------|
-| 7.2 | **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1` |
+| 7.2 | **`0.30.0`**, **`0.29.0`**, **`0.27.1`**,
**`0.25.1`**, **`0.24.0`**,
`0.22.1`, `0.21.0`,
`0.20.2`, `0.19.1` | `0.5.18`, `0.5.15.post1`,
`0.5.14`, `0.5.12.post1` |
| 7.1 | `0.17.1` | |
| 7.0 | `0.18.1`, `0.16.0`,
`0.15.1`, `0.14.1`,
`0.13.0`, `0.12.0`,
`0.11.2` | `0.5.9`, `0.5.8.post1`,
`0.5.7`, `0.5.6.post2` |
| 6.4 | `0.16.0`, `0.15.1`,
`0.14.1`, `0.13.0`,
`0.12.0`, `0.11.2`,
`0.10.2` | `0.5.8.post1`, `0.5.7`,
`0.5.6.post2`, `0.5.5.post3` |
diff --git a/pack/cuda/Dockerfile.sglang b/pack/cuda/Dockerfile.sglang
index 930f841e..7f747b6f 100644
--- a/pack/cuda/Dockerfile.sglang
+++ b/pack/cuda/Dockerfile.sglang
@@ -9,7 +9,7 @@ ARG SGLANG_TORCH_CUDA_VERSION=${CUDA_VERSION}
## Keep >= 0.4.6 and in sync with the ROCm and vLLM pins: SGLang's lmc_radix_cache.py imports
## `lmcache.integration.sglang.multi_process_adapter`, added in 0.4.6, and the MP wire protocol
## has no version handshake. The SGLang base ships no lmcache, so it is always built from source.
-ARG SGLANG_LMCACHE_VERSION=0.5.4
+ARG SGLANG_LMCACHE_VERSION=0.5.5
ARG SGLANG_NVIDIA_HPCX_VERSION=2.24.1_cuda13
ARG SGLANG_AWS_EFA_VERSION=1.46.0
@@ -192,6 +192,10 @@ RUN < cuda-bindings<14,>=13.0.3` requirement.
+ ## 0.5.5 generates its gRPC stubs at build time (setup.py _BuildPyWithGrpcStubs);
+ ## provide the pinned codegen tools its build-system requirements name, since
+ ## `--no-isolation` skips installing them.
+ uv pip install grpcio==1.84.0 grpcio-tools==1.84.0
pushd /tmp/lmcache \
&& sed -i "s/\"torch==.*\"/\"torch\"/g" /tmp/lmcache/pyproject.toml \
&& cat /tmp/lmcache/pyproject.toml \
@@ -373,7 +377,7 @@ RUN < cuda-bindings<14,>=13.0.3` requirement.
## Note: separate statements, not an `&&` chain —— `bash -e` ignores a failing non-final
## command of a list, which would leave /workspace empty and fail later at install time.
+ ## 0.5.5 generates its gRPC stubs at build time (setup.py _BuildPyWithGrpcStubs);
+ ## provide the pinned codegen tools its build-system requirements name, since
+ ## `--no-isolation` skips installing them.
+ uv pip install grpcio==1.84.0 grpcio-tools==1.84.0
cd /tmp/lmcache
## Drop the build-time torch pin: we compile against the base's own torch on purpose.
sed -i "s/\"torch==.*\"/\"torch\"/g" /tmp/lmcache/pyproject.toml
cat /tmp/lmcache/pyproject.toml
+ ## torch 2.14's headers require C++20 (#error in torch/all.h), but lmcache 0.5.5
+ ## hardcodes -std=c++17 for its g++ sources (pybind.cpp); nvcc already defaults
+ ## to C++20 under CUDA 13. Harmless on the cu129 base's torch 2.13.
+ sed -i 's/-std=c++17/-std=c++20/g' \
+ /tmp/lmcache/setup_extensions/common_cpp.py \
+ /tmp/lmcache/setup_extensions/build_profiles/cuda.py
## `--skip-dependency-check`: the base does not ship torch's full dependency closure.
python -m build --no-isolation --skip-dependency-check --wheel
tree -hs /tmp/lmcache/dist
@@ -473,7 +532,7 @@ soundfile
mistral_common[audio]
# tokenizer extras
-fastokens==0.2.0
+fastokens==0.3.2
EOT
uv pip install \
-r /tmp/requirements.txt
diff --git a/pack/matrix.yaml b/pack/matrix.yaml
index 6032af33..cba5265b 100644
--- a/pack/matrix.yaml
+++ b/pack/matrix.yaml
@@ -101,8 +101,8 @@ rules:
- "linux/amd64"
args:
- "ROCM_VERSION=7.2.3"
- - "VLLM_VERSION=0.29.0"
- - "VLLM_BASE_IMAGE=vllm/vllm-openai-rocm:v0.29.0"
+ - "VLLM_VERSION=0.30.0"
+ - "VLLM_BASE_IMAGE=vllm/vllm-openai-rocm:v0.30.0"
## AMD ROCm 7.2.1 - SGLang 0.5.18 (rocm720-mi30x)
##
- backend: "rocm"
@@ -119,15 +119,17 @@ rules:
# NVIDIA CUDA
#
- ## NVIDIA CUDA 13.0.1
+ ## NVIDIA CUDA 13.0.3
+ ## (torch is upgraded from the base's 2.13.0 to 2.14.0 —— see the
+ ## "Upgrade Torch" step in pack/cuda/Dockerfile.vllm)
##
- backend: "cuda"
services:
- "vllm"
args:
- - "CUDA_VERSION=13.0.1"
- - "VLLM_VERSION=0.29.0"
- - "VLLM_BASE_IMAGE=vllm/vllm-openai:v0.29.0-ubuntu2404"
+ - "CUDA_VERSION=13.0.3"
+ - "VLLM_VERSION=0.30.0"
+ - "VLLM_BASE_IMAGE=vllm/vllm-openai:v0.30.0-ubuntu2404"
## NVIDIA CUDA 13.0.1 - SGLang 0.5.18
##
- backend: "cuda"
@@ -138,14 +140,18 @@ rules:
- "SGLANG_VERSION=0.5.18"
- "SGLANG_BASE_IMAGE=lmsysorg/sglang:v0.5.18-cu130"
## NVIDIA CUDA 12.9.1
+ ## (stays on the base's torch 2.13.0: 2.14.0 is published for cu130 only,
+ ## so VLLM_TORCH_VERSION is pinned to the base's version and the
+ ## "Upgrade Torch" step in pack/cuda/Dockerfile.vllm is a no-op)
##
- backend: "cuda"
services:
- "vllm"
args:
- "CUDA_VERSION=12.9.1"
- - "VLLM_VERSION=0.29.0"
- - "VLLM_BASE_IMAGE=vllm/vllm-openai:v0.29.0-cu129-ubuntu2404"
+ - "VLLM_VERSION=0.30.0"
+ - "VLLM_BASE_IMAGE=vllm/vllm-openai:v0.30.0-cu129-ubuntu2404"
+ - "VLLM_TORCH_VERSION=2.13.0"
## NVIDIA CUDA 12.9.1 - SGLang 0.5.18
##
- backend: "cuda"
diff --git a/pack/rocm/Dockerfile.sglang b/pack/rocm/Dockerfile.sglang
index 426280b4..6022e1cd 100644
--- a/pack/rocm/Dockerfile.sglang
+++ b/pack/rocm/Dockerfile.sglang
@@ -8,7 +8,7 @@ ARG SGLANG_TORCH_VERSION=2.9.1
ARG SGLANG_TORCH_ROCM_VERSION=${ROCM_VERSION}
## Keep >= 0.4.6 and in sync with the CUDA pin —— see pack/cuda/Dockerfile.sglang.
## Always built from source: upstream's ROCm wheel targets torch 2.11.0+rocm7.2.
-ARG SGLANG_LMCACHE_VERSION=0.5.4
+ARG SGLANG_LMCACHE_VERSION=0.5.5
FROM ${SGLANG_BASE_IMAGE} AS sglang-build
SHELL ["/bin/bash", "-eo", "pipefail", "-c"]
@@ -198,6 +198,10 @@ RUN <