diff --git a/docs/fern/assets/releases-atom.xml b/docs/fern/assets/releases-atom.xml
index 5fef9a36e76a..2d355fcacb95 100644
--- a/docs/fern/assets/releases-atom.xml
+++ b/docs/fern/assets/releases-atom.xml
@@ -9,8 +9,15 @@ Generated by docs/fern/scripts/gen_llms_tables.py from docs/fern/components/rele
NVIDIA Dynamo releases
https://docs.nvidia.com/dynamo/dev/reference/releases
- 2026-07-20T00:00:00Z
+ 2026-08-05T00:00:00Z
NVIDIA Dynamo
+
+ Dynamo v1.3.1 (patch)
+ https://github.com/ai-dynamo/dynamo/releases/tag/v1.3.1
+
+ 2026-08-05T00:00:00Z
+ Patch release. Fixes disaggregated SGLang serving over AWS EFA on GB200: the SGLang EFA runtime moves to NIXL 1.3.2 and all three EFA images to EFA Installer 1.49.0. Backend pins are otherwise unchanged from v1.3.0.
+
Dynamo v1.3.0 (stable)
https://github.com/ai-dynamo/dynamo/releases/tag/v1.3.0
diff --git a/docs/fern/assets/releases.json b/docs/fern/assets/releases.json
index c112e5fd9600..e50024f269ae 100644
--- a/docs/fern/assets/releases.json
+++ b/docs/fern/assets/releases.json
@@ -1,23 +1,43 @@
{
"source": "docs/fern/components/releases.data.ts",
"generator": "docs/fern/scripts/gen_llms_tables.py",
- "updated": "2026-07-20",
+ "updated": "2026-08-05",
"current": {
- "version": "v1.3.0",
- "date": "Jul 20, 2026",
- "dateIso": "2026-07-20",
- "tag": "1.3.0",
- "wheel": "1.3.0.post1"
+ "version": "v1.3.1",
+ "date": "Aug 5, 2026",
+ "dateIso": "2026-08-05",
+ "tag": "1.3.1",
+ "wheel": "1.3.1"
},
"mainTot": {
"sglang": "0.5.16",
"trtllm": "1.3.0rc23",
"vllm": "0.26.0",
"nixlSglang": "1.3.0",
- "nixlTrtllm": "1.0.1",
+ "nixlTrtllm": "1.3.1",
"nixlVllm": "1.3.1"
},
"releases": [
+ {
+ "version": "v1.3.1",
+ "notesHref": "/dynamo/dev/reference/releases/v1-3-0#v131",
+ "date": "Aug 5, 2026",
+ "kind": "patch",
+ "github": "https://github.com/ai-dynamo/dynamo/releases/tag/v1.3.1",
+ "docs": "https://docs.nvidia.com/dynamo",
+ "pins": {
+ "sglang": "0.5.14",
+ "trtllm": "1.3.0rc19",
+ "vllm": "0.23.0",
+ "nixlSglang": "1.3.2",
+ "nixlTrtllm": "1.0.1",
+ "nixlVllm": "1.1.0"
+ },
+ "ucx": "1.20.x",
+ "delta": "Patch release. Fixes disaggregated SGLang serving over AWS EFA on GB200: the SGLang EFA runtime moves to NIXL 1.3.2 and all three EFA images to EFA Installer 1.49.0. Backend pins are otherwise unchanged from v1.3.0.",
+ "dateIso": "2026-08-05",
+ "notesUrl": "https://docs.nvidia.com/dynamo/dev/reference/releases/v1-3-0#v131"
+ },
{
"version": "v1.3.0",
"notesHref": "/dynamo/dev/reference/releases/v1-3-0",
@@ -33,6 +53,7 @@
"nixlTrtllm": "1.0.1",
"nixlVllm": "1.1.0"
},
+ "wheel": "1.3.0.post1",
"ucx": "1.20.x",
"delta": "CUDA 12 container images discontinued; EFA variants go multi-arch as -efa; GA wheels published as 1.3.0.post1 (containers stay :1.3.0); UCX 1.20.x.",
"notesSummary": "Tool-calling and reasoning overhaul, RL rollout serving, the largest Router buildout to date, SLA-driven Planner autoscaling, and production GPU Memory Service on Kubernetes.",
@@ -478,6 +499,24 @@
}
],
"cudaHistory": [
+ {
+ "version": "1.3.1",
+ "backend": "SGLang",
+ "toolkit": "13.0",
+ "minDriver": "580.xx+"
+ },
+ {
+ "version": "1.3.1",
+ "backend": "TensorRT-LLM",
+ "toolkit": "13.1",
+ "minDriver": "580.xx+"
+ },
+ {
+ "version": "1.3.1",
+ "backend": "vLLM",
+ "toolkit": "13.0",
+ "minDriver": "580.xx+"
+ },
{
"version": "1.3.0",
"backend": "SGLang",
@@ -1062,12 +1101,12 @@
"href": "https://catalog.ngc.nvidia.com/orgs/nvidia/ai-dynamo/containers/vllm-runtime/tags",
"tags": [
{
- "label": "1.3.0",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.0"
+ "label": "1.3.1",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.1"
},
{
- "label": "1.3.0-efa",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.0-efa",
+ "label": "1.3.1-efa",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.1-efa",
"variant": "experimental"
}
]
@@ -1081,12 +1120,12 @@
"href": "https://catalog.ngc.nvidia.com/orgs/nvidia/ai-dynamo/containers/sglang-runtime/tags",
"tags": [
{
- "label": "1.3.0",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.0"
+ "label": "1.3.1",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.1"
},
{
- "label": "1.3.0-efa",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.0-efa",
+ "label": "1.3.1-efa",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.1-efa",
"variant": "experimental"
}
]
@@ -1100,12 +1139,12 @@
"href": "https://catalog.ngc.nvidia.com/orgs/nvidia/ai-dynamo/containers/tensorrtllm-runtime/tags",
"tags": [
{
- "label": "1.3.0",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0"
+ "label": "1.3.1",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.1"
},
{
- "label": "1.3.0-efa",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-efa",
+ "label": "1.3.1-efa",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.1-efa",
"variant": "experimental"
}
]
@@ -1119,8 +1158,8 @@
"href": "https://catalog.ngc.nvidia.com/orgs/nvidia/ai-dynamo/containers/dynamo-frontend/tags",
"tags": [
{
- "label": "1.3.0",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/dynamo-frontend:1.3.0"
+ "label": "1.3.1",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/dynamo-frontend:1.3.1"
}
]
},
@@ -1133,8 +1172,8 @@
"href": "https://catalog.ngc.nvidia.com/orgs/nvidia/ai-dynamo/containers/dynamo-planner/tags",
"tags": [
{
- "label": "1.3.0",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/dynamo-planner:1.3.0"
+ "label": "1.3.1",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/dynamo-planner:1.3.1"
}
]
},
@@ -1147,8 +1186,8 @@
"href": "https://catalog.ngc.nvidia.com/orgs/nvidia/ai-dynamo/containers/kubernetes-operator/tags",
"tags": [
{
- "label": "1.3.0",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/kubernetes-operator:1.3.0"
+ "label": "1.3.1",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/kubernetes-operator:1.3.1"
}
]
},
@@ -1162,8 +1201,8 @@
"badge": "Preview",
"tags": [
{
- "label": "1.3.0",
- "clipboard": "nvcr.io/nvidia/ai-dynamo/snapshot-agent:1.3.0"
+ "label": "1.3.1",
+ "clipboard": "nvcr.io/nvidia/ai-dynamo/snapshot-agent:1.3.1"
}
]
},
@@ -1172,11 +1211,11 @@
"name": "ai-dynamo",
"description": "Main package with backend integrations (vLLM, SGLang, TRT-LLM)",
"meta": "Python 3.10–3.12 · Linux (glibc v2.28+)",
- "href": "https://pypi.org/project/ai-dynamo/1.3.0.post1/",
+ "href": "https://pypi.org/project/ai-dynamo/1.3.1/",
"tags": [
{
- "label": "uv pip install ai-dynamo==1.3.0.post1",
- "clipboard": "uv pip install ai-dynamo==1.3.0.post1"
+ "label": "uv pip install ai-dynamo==1.3.1",
+ "clipboard": "uv pip install ai-dynamo==1.3.1"
}
]
},
@@ -1185,11 +1224,11 @@
"name": "ai-dynamo-runtime",
"description": "Core Python bindings for the Dynamo runtime",
"meta": "Python 3.10–3.12 · Linux (glibc v2.28+)",
- "href": "https://pypi.org/project/ai-dynamo-runtime/1.3.0.post1/",
+ "href": "https://pypi.org/project/ai-dynamo-runtime/1.3.1/",
"tags": [
{
- "label": "uv pip install ai-dynamo-runtime==1.3.0.post1",
- "clipboard": "uv pip install ai-dynamo-runtime==1.3.0.post1"
+ "label": "uv pip install ai-dynamo-runtime==1.3.1",
+ "clipboard": "uv pip install ai-dynamo-runtime==1.3.1"
}
]
},
@@ -1198,11 +1237,11 @@
"name": "kvbm",
"description": "KV Block Manager for disaggregated KV cache",
"meta": "Python 3.10–3.12 · Linux (glibc v2.28+)",
- "href": "https://pypi.org/project/kvbm/1.3.0.post1/",
+ "href": "https://pypi.org/project/kvbm/1.3.1/",
"tags": [
{
- "label": "uv pip install kvbm==1.3.0.post1",
- "clipboard": "uv pip install kvbm==1.3.0.post1"
+ "label": "uv pip install kvbm==1.3.1",
+ "clipboard": "uv pip install kvbm==1.3.1"
}
]
},
@@ -1235,11 +1274,11 @@
"name": "dynamo-runtime",
"description": "Core distributed runtime library",
"meta": "MSRV Rust v1.82",
- "href": "https://crates.io/crates/dynamo-runtime/1.3.0",
+ "href": "https://crates.io/crates/dynamo-runtime/1.3.1",
"tags": [
{
- "label": "cargo add dynamo-runtime@1.3.0",
- "clipboard": "cargo add dynamo-runtime@1.3.0"
+ "label": "cargo add dynamo-runtime@1.3.1",
+ "clipboard": "cargo add dynamo-runtime@1.3.1"
}
]
},
@@ -1248,11 +1287,11 @@
"name": "dynamo-llm",
"description": "LLM inference engine",
"meta": "MSRV Rust v1.82",
- "href": "https://crates.io/crates/dynamo-llm/1.3.0",
+ "href": "https://crates.io/crates/dynamo-llm/1.3.1",
"tags": [
{
- "label": "cargo add dynamo-llm@1.3.0",
- "clipboard": "cargo add dynamo-llm@1.3.0"
+ "label": "cargo add dynamo-llm@1.3.1",
+ "clipboard": "cargo add dynamo-llm@1.3.1"
}
]
},
@@ -1301,11 +1340,11 @@
"name": "dynamo-memory",
"description": "Memory management utilities",
"meta": "MSRV Rust v1.82",
- "href": "https://crates.io/crates/dynamo-memory/1.3.0",
+ "href": "https://crates.io/crates/dynamo-memory/1.3.1",
"tags": [
{
- "label": "cargo add dynamo-memory@1.3.0",
- "clipboard": "cargo add dynamo-memory@1.3.0"
+ "label": "cargo add dynamo-memory@1.3.1",
+ "clipboard": "cargo add dynamo-memory@1.3.1"
}
]
},
@@ -1327,11 +1366,11 @@
"name": "dynamo-tokens",
"description": "Tokenizer bindings for LLM inference",
"meta": "MSRV Rust v1.82",
- "href": "https://crates.io/crates/dynamo-tokens/1.3.0",
+ "href": "https://crates.io/crates/dynamo-tokens/1.3.1",
"tags": [
{
- "label": "cargo add dynamo-tokens@1.3.0",
- "clipboard": "cargo add dynamo-tokens@1.3.0"
+ "label": "cargo add dynamo-tokens@1.3.1",
+ "clipboard": "cargo add dynamo-tokens@1.3.1"
}
]
},
@@ -1340,11 +1379,11 @@
"name": "dynamo-tokenizers",
"description": "Tokenizer library for LLM inference",
"meta": "MSRV Rust v1.82",
- "href": "https://crates.io/crates/dynamo-tokenizers/1.3.0",
+ "href": "https://crates.io/crates/dynamo-tokenizers/1.3.1",
"tags": [
{
- "label": "cargo add dynamo-tokenizers@1.3.0",
- "clipboard": "cargo add dynamo-tokenizers@1.3.0"
+ "label": "cargo add dynamo-tokenizers@1.3.1",
+ "clipboard": "cargo add dynamo-tokenizers@1.3.1"
}
]
},
@@ -1353,11 +1392,11 @@
"name": "dynamo-mocker",
"description": "Inference engine simulator for benchmarking",
"meta": "MSRV Rust v1.82",
- "href": "https://crates.io/crates/dynamo-mocker/1.3.0",
+ "href": "https://crates.io/crates/dynamo-mocker/1.3.1",
"tags": [
{
- "label": "cargo add dynamo-mocker@1.3.0",
- "clipboard": "cargo add dynamo-mocker@1.3.0"
+ "label": "cargo add dynamo-mocker@1.3.1",
+ "clipboard": "cargo add dynamo-mocker@1.3.1"
}
]
},
@@ -1366,11 +1405,11 @@
"name": "dynamo-kv-router",
"description": "KV-aware request routing library",
"meta": "MSRV Rust v1.82",
- "href": "https://crates.io/crates/dynamo-kv-router/1.3.0",
+ "href": "https://crates.io/crates/dynamo-kv-router/1.3.1",
"tags": [
{
- "label": "cargo add dynamo-kv-router@1.3.0",
- "clipboard": "cargo add dynamo-kv-router@1.3.0"
+ "label": "cargo add dynamo-kv-router@1.3.1",
+ "clipboard": "cargo add dynamo-kv-router@1.3.1"
}
]
},
@@ -1379,11 +1418,11 @@
"name": "kvbm-logical",
"description": "Logical layer for the KV Block Manager",
"meta": "MSRV Rust v1.82",
- "href": "https://crates.io/crates/kvbm-logical/1.3.0",
+ "href": "https://crates.io/crates/kvbm-logical/1.3.1",
"tags": [
{
- "label": "cargo add kvbm-logical@1.3.0",
- "clipboard": "cargo add kvbm-logical@1.3.0"
+ "label": "cargo add kvbm-logical@1.3.1",
+ "clipboard": "cargo add kvbm-logical@1.3.1"
}
]
}
diff --git a/docs/fern/components/install-selector-data.ts b/docs/fern/components/install-selector-data.ts
index b6a8772f08ac..9c0b9f5c39f1 100644
--- a/docs/fern/components/install-selector-data.ts
+++ b/docs/fern/components/install-selector-data.ts
@@ -59,8 +59,14 @@ function dockerCommand(image: string, tag: string): string {
return `docker run --gpus all --network host --ipc host --rm -it nvcr.io/nvidia/ai-dynamo/${image}:${tag}`;
}
-function stableWheelCommand(backend: Backend, version: string): string {
- const wheel = version === CURRENT_VERSION.slice(1) ? CURRENT_WHEEL : version;
+function stableWheelCommand(
+ backend: Backend,
+ version: string,
+ wheelOverride?: string,
+): string {
+ const wheel =
+ wheelOverride ??
+ (version === CURRENT_VERSION.slice(1) ? CURRENT_WHEEL : version);
const prerelease = backend.id === "sglang" ? "--prerelease=allow " : "";
return `uv pip install ${prerelease}"ai-dynamo[${backend.extra}]==${wheel}"`;
}
@@ -84,7 +90,7 @@ function stableEntries(backend: Backend): InstallEntry[] {
commands: {
container: dockerCommand(`${backend.image}-runtime`, version),
...(backend.extra
- ? { wheel: stableWheelCommand(backend, version) }
+ ? { wheel: stableWheelCommand(backend, version, release.wheel) }
: {}),
},
};
diff --git a/docs/fern/components/releases.data.ts b/docs/fern/components/releases.data.ts
index 312019719700..6525dc556e36 100644
--- a/docs/fern/components/releases.data.ts
+++ b/docs/fern/components/releases.data.ts
@@ -50,6 +50,9 @@ export interface Release {
/** Docs-native release notes page (absolute site path); GitHub link used when absent. */
notesHref?: string;
pins?: BackendPins;
+ /** PyPI wheel version when it differs from the container tag (e.g. a .post
+ * rebuild). Falls back to the container version when absent. */
+ wheel?: string;
/** UCX version shipped with the release's NIXL builds — from the release's
* Key Dependencies table; omitted where the source never stated one
* (v1.0.0 and patch releases). */
@@ -62,10 +65,10 @@ export interface Release {
partial?: boolean;
}
-export const CURRENT_VERSION = "v1.3.0";
-export const CURRENT_DATE = "Jul 20, 2026";
-export const CURRENT_TAG = "1.3.0";
-export const CURRENT_WHEEL = "1.3.0.post1";
+export const CURRENT_VERSION = "v1.3.1";
+export const CURRENT_DATE = "Aug 5, 2026";
+export const CURRENT_TAG = "1.3.1";
+export const CURRENT_WHEEL = "1.3.1";
export const MAIN_TOT: BackendPins = {
sglang: "0.5.16",
@@ -79,6 +82,18 @@ export const MAIN_TOT: BackendPins = {
const GH = "https://github.com/ai-dynamo/dynamo/releases/tag/";
export const RELEASES: Release[] = [
+ {
+ version: "v1.3.1",
+ notesHref: "/dynamo/dev/reference/releases/v1-3-0#v131",
+ date: "Aug 5, 2026",
+ kind: "patch",
+ github: `${GH}v1.3.1`,
+ docs: "https://docs.nvidia.com/dynamo",
+ pins: { sglang: "0.5.14", trtllm: "1.3.0rc19", vllm: "0.23.0", nixlSglang: "1.3.2", nixlTrtllm: "1.0.1", nixlVllm: "1.1.0" },
+ ucx: "1.20.x",
+ delta:
+ "Patch release. Fixes disaggregated SGLang serving over AWS EFA on GB200: the SGLang EFA runtime moves to NIXL 1.3.2 and all three EFA images to EFA Installer 1.49.0. Backend pins are otherwise unchanged from v1.3.0.",
+ },
{
version: "v1.3.0",
notesHref: "/dynamo/dev/reference/releases/v1-3-0",
@@ -87,6 +102,7 @@ export const RELEASES: Release[] = [
github: `${GH}v1.3.0`,
docs: "https://docs.nvidia.com/dynamo",
pins: { sglang: "0.5.14", trtllm: "1.3.0rc19", vllm: "0.23.0", nixlSglang: "1.3.0", nixlTrtllm: "1.0.1", nixlVllm: "1.1.0" },
+ wheel: "1.3.0.post1",
ucx: "1.20.x",
delta:
"CUDA 12 container images discontinued; EFA variants go multi-arch as -efa; GA wheels published as 1.3.0.post1 (containers stay :1.3.0); UCX 1.20.x.",
@@ -330,6 +346,9 @@ export interface CudaRow {
}
export const CUDA_HISTORY: CudaRow[] = [
+ { version: "1.3.1", backend: "SGLang", toolkit: "13.0", minDriver: "580.xx+" },
+ { version: "1.3.1", backend: "TensorRT-LLM", toolkit: "13.1", minDriver: "580.xx+" },
+ { version: "1.3.1", backend: "vLLM", toolkit: "13.0", minDriver: "580.xx+" },
{ version: "1.3.0", backend: "SGLang", toolkit: "13.0", minDriver: "580.xx+" },
{ version: "1.3.0", backend: "TensorRT-LLM", toolkit: "13.1", minDriver: "580.xx+" },
{ version: "1.3.0", backend: "vLLM", toolkit: "13.0", minDriver: "580.xx+" },
@@ -559,8 +578,8 @@ export const ARTIFACTS: Artifact[] = [
meta: "vLLM v0.23.0 · CUDA 13.0 · AMD64/ARM64",
href: `${NGC_C}/vllm-runtime/tags`,
tags: [
- { label: "1.3.0", clipboard: "nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.0" },
- { label: "1.3.0-efa", clipboard: "nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.0-efa", variant: "experimental" },
+ { label: "1.3.1", clipboard: "nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.1" },
+ { label: "1.3.1-efa", clipboard: "nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.1-efa", variant: "experimental" },
],
},
{
@@ -571,8 +590,8 @@ export const ARTIFACTS: Artifact[] = [
meta: "SGLang v0.5.14 · CUDA 13.0 · AMD64/ARM64",
href: `${NGC_C}/sglang-runtime/tags`,
tags: [
- { label: "1.3.0", clipboard: "nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.0" },
- { label: "1.3.0-efa", clipboard: "nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.0-efa", variant: "experimental" },
+ { label: "1.3.1", clipboard: "nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.1" },
+ { label: "1.3.1-efa", clipboard: "nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.1-efa", variant: "experimental" },
],
},
{
@@ -583,8 +602,8 @@ export const ARTIFACTS: Artifact[] = [
meta: "TRT-LLM v1.3.0rc19 · CUDA 13.1 · AMD64/ARM64",
href: `${NGC_C}/tensorrtllm-runtime/tags`,
tags: [
- { label: "1.3.0", clipboard: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0" },
- { label: "1.3.0-efa", clipboard: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-efa", variant: "experimental" },
+ { label: "1.3.1", clipboard: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.1" },
+ { label: "1.3.1-efa", clipboard: "nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.1-efa", variant: "experimental" },
],
},
{
@@ -594,7 +613,7 @@ export const ARTIFACTS: Artifact[] = [
description: "OpenAI-compatible API gateway with Endpoint Prediction Protocol (EPP)",
meta: "AMD64/ARM64",
href: `${NGC_C}/dynamo-frontend/tags`,
- tags: [{ label: "1.3.0", clipboard: "nvcr.io/nvidia/ai-dynamo/dynamo-frontend:1.3.0" }],
+ tags: [{ label: "1.3.1", clipboard: "nvcr.io/nvidia/ai-dynamo/dynamo-frontend:1.3.1" }],
},
{
category: "container",
@@ -603,7 +622,7 @@ export const ARTIFACTS: Artifact[] = [
description: "Standalone Planner used by Profiler jobs and Planner pods",
meta: "AMD64/ARM64",
href: `${NGC_C}/dynamo-planner/tags`,
- tags: [{ label: "1.3.0", clipboard: "nvcr.io/nvidia/ai-dynamo/dynamo-planner:1.3.0" }],
+ tags: [{ label: "1.3.1", clipboard: "nvcr.io/nvidia/ai-dynamo/dynamo-planner:1.3.1" }],
},
{
category: "container",
@@ -612,7 +631,7 @@ export const ARTIFACTS: Artifact[] = [
description: "Operator that manages Dynamo deployments and CRDs",
meta: "AMD64/ARM64",
href: `${NGC_C}/kubernetes-operator/tags`,
- tags: [{ label: "1.3.0", clipboard: "nvcr.io/nvidia/ai-dynamo/kubernetes-operator:1.3.0" }],
+ tags: [{ label: "1.3.1", clipboard: "nvcr.io/nvidia/ai-dynamo/kubernetes-operator:1.3.1" }],
},
{
category: "container",
@@ -622,24 +641,24 @@ export const ARTIFACTS: Artifact[] = [
meta: "AMD64/ARM64",
href: `${NGC_C}/snapshot-agent/tags`,
badge: "Preview",
- tags: [{ label: "1.3.0", clipboard: "nvcr.io/nvidia/ai-dynamo/snapshot-agent:1.3.0" }],
+ tags: [{ label: "1.3.1", clipboard: "nvcr.io/nvidia/ai-dynamo/snapshot-agent:1.3.1" }],
},
{
category: "wheel",
name: "ai-dynamo",
description: "Main package with backend integrations (vLLM, SGLang, TRT-LLM)",
meta: "Python 3.10–3.12 · Linux (glibc v2.28+)",
- href: "https://pypi.org/project/ai-dynamo/1.3.0.post1/",
- tags: [{ label: "uv pip install ai-dynamo==1.3.0.post1", clipboard: "uv pip install ai-dynamo==1.3.0.post1" }],
+ href: "https://pypi.org/project/ai-dynamo/1.3.1/",
+ tags: [{ label: "uv pip install ai-dynamo==1.3.1", clipboard: "uv pip install ai-dynamo==1.3.1" }],
},
{
category: "wheel",
name: "ai-dynamo-runtime",
description: "Core Python bindings for the Dynamo runtime",
meta: "Python 3.10–3.12 · Linux (glibc v2.28+)",
- href: "https://pypi.org/project/ai-dynamo-runtime/1.3.0.post1/",
+ href: "https://pypi.org/project/ai-dynamo-runtime/1.3.1/",
tags: [
- { label: "uv pip install ai-dynamo-runtime==1.3.0.post1", clipboard: "uv pip install ai-dynamo-runtime==1.3.0.post1" },
+ { label: "uv pip install ai-dynamo-runtime==1.3.1", clipboard: "uv pip install ai-dynamo-runtime==1.3.1" },
],
},
{
@@ -647,8 +666,8 @@ export const ARTIFACTS: Artifact[] = [
name: "kvbm",
description: "KV Block Manager for disaggregated KV cache",
meta: "Python 3.10–3.12 · Linux (glibc v2.28+)",
- href: "https://pypi.org/project/kvbm/1.3.0.post1/",
- tags: [{ label: "uv pip install kvbm==1.3.0.post1", clipboard: "uv pip install kvbm==1.3.0.post1" }],
+ href: "https://pypi.org/project/kvbm/1.3.1/",
+ tags: [{ label: "uv pip install kvbm==1.3.1", clipboard: "uv pip install kvbm==1.3.1" }],
},
{
category: "helm",
@@ -680,16 +699,16 @@ export const ARTIFACTS: Artifact[] = [
name: "dynamo-runtime",
description: "Core distributed runtime library",
meta: "MSRV Rust v1.82",
- href: "https://crates.io/crates/dynamo-runtime/1.3.0",
- tags: [{ label: "cargo add dynamo-runtime@1.3.0", clipboard: "cargo add dynamo-runtime@1.3.0" }],
+ href: "https://crates.io/crates/dynamo-runtime/1.3.1",
+ tags: [{ label: "cargo add dynamo-runtime@1.3.1", clipboard: "cargo add dynamo-runtime@1.3.1" }],
},
{
category: "crate",
name: "dynamo-llm",
description: "LLM inference engine",
meta: "MSRV Rust v1.82",
- href: "https://crates.io/crates/dynamo-llm/1.3.0",
- tags: [{ label: "cargo add dynamo-llm@1.3.0", clipboard: "cargo add dynamo-llm@1.3.0" }],
+ href: "https://crates.io/crates/dynamo-llm/1.3.1",
+ tags: [{ label: "cargo add dynamo-llm@1.3.1", clipboard: "cargo add dynamo-llm@1.3.1" }],
},
{
category: "crate",
@@ -721,8 +740,8 @@ export const ARTIFACTS: Artifact[] = [
name: "dynamo-memory",
description: "Memory management utilities",
meta: "MSRV Rust v1.82",
- href: "https://crates.io/crates/dynamo-memory/1.3.0",
- tags: [{ label: "cargo add dynamo-memory@1.3.0", clipboard: "cargo add dynamo-memory@1.3.0" }],
+ href: "https://crates.io/crates/dynamo-memory/1.3.1",
+ tags: [{ label: "cargo add dynamo-memory@1.3.1", clipboard: "cargo add dynamo-memory@1.3.1" }],
},
{
category: "crate",
@@ -737,40 +756,40 @@ export const ARTIFACTS: Artifact[] = [
name: "dynamo-tokens",
description: "Tokenizer bindings for LLM inference",
meta: "MSRV Rust v1.82",
- href: "https://crates.io/crates/dynamo-tokens/1.3.0",
- tags: [{ label: "cargo add dynamo-tokens@1.3.0", clipboard: "cargo add dynamo-tokens@1.3.0" }],
+ href: "https://crates.io/crates/dynamo-tokens/1.3.1",
+ tags: [{ label: "cargo add dynamo-tokens@1.3.1", clipboard: "cargo add dynamo-tokens@1.3.1" }],
},
{
category: "crate",
name: "dynamo-tokenizers",
description: "Tokenizer library for LLM inference",
meta: "MSRV Rust v1.82",
- href: "https://crates.io/crates/dynamo-tokenizers/1.3.0",
- tags: [{ label: "cargo add dynamo-tokenizers@1.3.0", clipboard: "cargo add dynamo-tokenizers@1.3.0" }],
+ href: "https://crates.io/crates/dynamo-tokenizers/1.3.1",
+ tags: [{ label: "cargo add dynamo-tokenizers@1.3.1", clipboard: "cargo add dynamo-tokenizers@1.3.1" }],
},
{
category: "crate",
name: "dynamo-mocker",
description: "Inference engine simulator for benchmarking",
meta: "MSRV Rust v1.82",
- href: "https://crates.io/crates/dynamo-mocker/1.3.0",
- tags: [{ label: "cargo add dynamo-mocker@1.3.0", clipboard: "cargo add dynamo-mocker@1.3.0" }],
+ href: "https://crates.io/crates/dynamo-mocker/1.3.1",
+ tags: [{ label: "cargo add dynamo-mocker@1.3.1", clipboard: "cargo add dynamo-mocker@1.3.1" }],
},
{
category: "crate",
name: "dynamo-kv-router",
description: "KV-aware request routing library",
meta: "MSRV Rust v1.82",
- href: "https://crates.io/crates/dynamo-kv-router/1.3.0",
- tags: [{ label: "cargo add dynamo-kv-router@1.3.0", clipboard: "cargo add dynamo-kv-router@1.3.0" }],
+ href: "https://crates.io/crates/dynamo-kv-router/1.3.1",
+ tags: [{ label: "cargo add dynamo-kv-router@1.3.1", clipboard: "cargo add dynamo-kv-router@1.3.1" }],
},
{
category: "crate",
name: "kvbm-logical",
description: "Logical layer for the KV Block Manager",
meta: "MSRV Rust v1.82",
- href: "https://crates.io/crates/kvbm-logical/1.3.0",
- tags: [{ label: "cargo add kvbm-logical@1.3.0", clipboard: "cargo add kvbm-logical@1.3.0" }],
+ href: "https://crates.io/crates/kvbm-logical/1.3.1",
+ tags: [{ label: "cargo add kvbm-logical@1.3.1", clipboard: "cargo add kvbm-logical@1.3.1" }],
},
];
diff --git a/docs/fern/pages/reference/general/compatibility.mdx b/docs/fern/pages/reference/general/compatibility.mdx
index 1ba662769a42..a2e1b866a01a 100644
--- a/docs/fern/pages/reference/general/compatibility.mdx
+++ b/docs/fern/pages/reference/general/compatibility.mdx
@@ -33,6 +33,9 @@ Every released line — stable releases and their patches, grouped by minor line
| Dynamo | Backend | CUDA Toolkit | Min Driver | Note |
| --- | --- | --- | --- | --- |
+| 1.3.1 | SGLang | 13.0 | 580.xx+ | - |
+| 1.3.1 | TensorRT-LLM | 13.1 | 580.xx+ | - |
+| 1.3.1 | vLLM | 13.0 | 580.xx+ | - |
| 1.3.0 | SGLang | 13.0 | 580.xx+ | - |
| 1.3.0 | TensorRT-LLM | 13.1 | 580.xx+ | - |
| 1.3.0 | vLLM | 13.0 | 580.xx+ | - |
@@ -304,13 +307,14 @@ Pairwise feature-by-feature compatibility within each backend. Each cell reports
{/* llms-tables:begin — generated by scripts/gen_llms_tables.py, do not edit */}
-Current stable release: v1.3.0 (container tag `1.3.0`, wheel version `1.3.0.post1`).
+Current stable release: v1.3.1 (container tag `1.3.1`, wheel version `1.3.1`).
**Backend engine pins per Dynamo release**
| Dynamo | Type | SGLang | TensorRT-LLM | vLLM | NIXL (SGL / TRT / vLLM) | UCX |
| --- | --- | --- | --- | --- | --- | --- |
-| main (ToT) | development head | 0.5.16 | 1.3.0rc23 | 0.26.0 | 1.3.0 / 1.0.1 / 1.3.1 | - |
+| main (ToT) | development head | 0.5.16 | 1.3.0rc23 | 0.26.0 | 1.3.0 / 1.3.1 / 1.3.1 | - |
+| v1.3.1 | patch | 0.5.14 | 1.3.0rc19 | 0.23.0 | 1.3.2 / 1.0.1 / 1.1.0 | 1.20.x |
| v1.3.0 | stable | 0.5.14 | 1.3.0rc19 | 0.23.0 | 1.3.0 / 1.0.1 / 1.1.0 | 1.20.x |
| v1.3.0-dev.1 | platform-preview | 0.5.12.post1 | 1.3.0rc17 | 0.22.0 | 1.0.1 / 0.10.1 / 1.1.0 | - |
| v1.2.1 | patch | 0.5.11 | 1.3.0rc14 | 0.20.1 | 1.0.1 / 0.10.1 / 0.10.1 | - |
@@ -343,6 +347,9 @@ Current stable release: v1.3.0 (container tag `1.3.0`, wheel version `1.3.0.post
| Dynamo | Backend | CUDA Toolkit | Min Driver | Note |
| --- | --- | --- | --- | --- |
+| 1.3.1 | SGLang | 13.0 | 580.xx+ | - |
+| 1.3.1 | TensorRT-LLM | 13.1 | 580.xx+ | - |
+| 1.3.1 | vLLM | 13.0 | 580.xx+ | - |
| 1.3.0 | SGLang | 13.0 | 580.xx+ | - |
| 1.3.0 | TensorRT-LLM | 13.1 | 580.xx+ | - |
| 1.3.0 | vLLM | 13.0 | 580.xx+ | - |
@@ -408,7 +415,7 @@ Current stable release: v1.3.0 (container tag `1.3.0`, wheel version `1.3.0.post
- Early access v1.1.0-dev.* images follow the same CUDA matrix as v1.0.2. The v1.2.0-deepseek-v4-dev.3 vLLM container is CUDA 13.0 multi-arch; the SGLang containers split by arch (CUDA 12.9 on amd64, CUDA 13.0 on arm64).
- Experimental CUDA 13 images are not published for all versions.
-**Feature support by backend (v1.3.0)**
+**Feature support by backend (v1.3.1)**
| Feature | SGLang | TensorRT-LLM | vLLM |
| --- | --- | --- | --- |
diff --git a/docs/fern/pages/reference/general/release-artifacts.mdx b/docs/fern/pages/reference/general/release-artifacts.mdx
index 4d319e950b92..a004fcfa633b 100644
--- a/docs/fern/pages/reference/general/release-artifacts.mdx
+++ b/docs/fern/pages/reference/general/release-artifacts.mdx
@@ -90,36 +90,36 @@ For the full release history — every release newest-first with its notes — s
{/* llms-tables:begin — generated by scripts/gen_llms_tables.py, do not edit */}
-Current stable release: v1.3.0 (container tag `1.3.0`, wheel version `1.3.0.post1`).
+Current stable release: v1.3.1 (container tag `1.3.1`, wheel version `1.3.1`).
-**Artifact inventory (v1.3.0)**
+**Artifact inventory (v1.3.1)**
| Category | Name | Description | Meta | Tags / install |
| --- | --- | --- | --- | --- |
-| container | vllm-runtime | vLLM backend runtime | vLLM v0.23.0 · CUDA 13.0 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.0`; `nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.0-efa` |
-| container | sglang-runtime | SGLang backend runtime | SGLang v0.5.14 · CUDA 13.0 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.0`; `nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.0-efa` |
-| container | tensorrtllm-runtime | TensorRT-LLM backend runtime | TRT-LLM v1.3.0rc19 · CUDA 13.1 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0`; `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-efa` |
-| container | dynamo-frontend | OpenAI-compatible API gateway with Endpoint Prediction Protocol (EPP) | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/dynamo-frontend:1.3.0` |
-| container | dynamo-planner | Standalone Planner used by Profiler jobs and Planner pods | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/dynamo-planner:1.3.0` |
-| container | kubernetes-operator | Operator that manages Dynamo deployments and CRDs | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/kubernetes-operator:1.3.0` |
-| container | snapshot-agent (Preview) | Fast GPU worker recovery via CRIU | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/snapshot-agent:1.3.0` |
-| wheel | ai-dynamo | Main package with backend integrations (vLLM, SGLang, TRT-LLM) | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install ai-dynamo==1.3.0.post1` |
-| wheel | ai-dynamo-runtime | Core Python bindings for the Dynamo runtime | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install ai-dynamo-runtime==1.3.0.post1` |
-| wheel | kvbm | KV Block Manager for disaggregated KV cache | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install kvbm==1.3.0.post1` |
+| container | vllm-runtime | vLLM backend runtime | vLLM v0.23.0 · CUDA 13.0 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.1`; `nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.1-efa` |
+| container | sglang-runtime | SGLang backend runtime | SGLang v0.5.14 · CUDA 13.0 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.1`; `nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.1-efa` |
+| container | tensorrtllm-runtime | TensorRT-LLM backend runtime | TRT-LLM v1.3.0rc19 · CUDA 13.1 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.1`; `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.1-efa` |
+| container | dynamo-frontend | OpenAI-compatible API gateway with Endpoint Prediction Protocol (EPP) | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/dynamo-frontend:1.3.1` |
+| container | dynamo-planner | Standalone Planner used by Profiler jobs and Planner pods | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/dynamo-planner:1.3.1` |
+| container | kubernetes-operator | Operator that manages Dynamo deployments and CRDs | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/kubernetes-operator:1.3.1` |
+| container | snapshot-agent (Preview) | Fast GPU worker recovery via CRIU | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/snapshot-agent:1.3.1` |
+| wheel | ai-dynamo | Main package with backend integrations (vLLM, SGLang, TRT-LLM) | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install ai-dynamo==1.3.1` |
+| wheel | ai-dynamo-runtime | Core Python bindings for the Dynamo runtime | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install ai-dynamo-runtime==1.3.1` |
+| wheel | kvbm | KV Block Manager for disaggregated KV cache | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install kvbm==1.3.1` |
| helm | dynamo-platform | Platform services (etcd, NATS) and the Dynamo Operator for a Dynamo cluster | - | `helm install dynamo-platform https://helm.ngc.nvidia.com/nvidia/ai-dynamo/charts/dynamo-platform-1.3.0.tgz` |
| helm | snapshot | Snapshot DaemonSet for fast GPU worker recovery | - | `helm install snapshot https://helm.ngc.nvidia.com/nvidia/ai-dynamo/charts/snapshot-1.3.0.tgz` |
-| crate | dynamo-runtime | Core distributed runtime library | MSRV Rust v1.82 | `cargo add dynamo-runtime@1.3.0` |
-| crate | dynamo-llm | LLM inference engine | MSRV Rust v1.82 | `cargo add dynamo-llm@1.3.0` |
+| crate | dynamo-runtime | Core distributed runtime library | MSRV Rust v1.82 | `cargo add dynamo-runtime@1.3.1` |
+| crate | dynamo-llm | LLM inference engine | MSRV Rust v1.82 | `cargo add dynamo-llm@1.3.1` |
| crate | dynamo-protocols | Async OpenAI-compatible API client | MSRV Rust v1.82 | `cargo add dynamo-protocols@1.3.0` |
| crate | dynamo-async-openai (Deprecated) | Legacy OpenAI client; use dynamo-protocols | MSRV Rust v1.82 · final release | `cargo add dynamo-async-openai@1.0.2` |
| crate | dynamo-parsers | Protocol parsers (SSE, JSON streaming) | MSRV Rust v1.82 | `cargo add dynamo-parsers@1.3.0` |
-| crate | dynamo-memory | Memory management utilities | MSRV Rust v1.82 | `cargo add dynamo-memory@1.3.0` |
+| crate | dynamo-memory | Memory management utilities | MSRV Rust v1.82 | `cargo add dynamo-memory@1.3.1` |
| crate | dynamo-config | Configuration management | MSRV Rust v1.82 | `cargo add dynamo-config@1.3.0` |
-| crate | dynamo-tokens | Tokenizer bindings for LLM inference | MSRV Rust v1.82 | `cargo add dynamo-tokens@1.3.0` |
-| crate | dynamo-tokenizers | Tokenizer library for LLM inference | MSRV Rust v1.82 | `cargo add dynamo-tokenizers@1.3.0` |
-| crate | dynamo-mocker | Inference engine simulator for benchmarking | MSRV Rust v1.82 | `cargo add dynamo-mocker@1.3.0` |
-| crate | dynamo-kv-router | KV-aware request routing library | MSRV Rust v1.82 | `cargo add dynamo-kv-router@1.3.0` |
-| crate | kvbm-logical | Logical layer for the KV Block Manager | MSRV Rust v1.82 | `cargo add kvbm-logical@1.3.0` |
+| crate | dynamo-tokens | Tokenizer bindings for LLM inference | MSRV Rust v1.82 | `cargo add dynamo-tokens@1.3.1` |
+| crate | dynamo-tokenizers | Tokenizer library for LLM inference | MSRV Rust v1.82 | `cargo add dynamo-tokenizers@1.3.1` |
+| crate | dynamo-mocker | Inference engine simulator for benchmarking | MSRV Rust v1.82 | `cargo add dynamo-mocker@1.3.1` |
+| crate | dynamo-kv-router | KV-aware request routing library | MSRV Rust v1.82 | `cargo add dynamo-kv-router@1.3.1` |
+| crate | kvbm-logical | Logical layer for the KV Block Manager | MSRV Rust v1.82 | `cargo add kvbm-logical@1.3.1` |
**Known artifact issues**
diff --git a/docs/fern/pages/reference/general/releases-machine-readable.mdx b/docs/fern/pages/reference/general/releases-machine-readable.mdx
index 196210b1a2a5..74755274a6a9 100644
--- a/docs/fern/pages/reference/general/releases-machine-readable.mdx
+++ b/docs/fern/pages/reference/general/releases-machine-readable.mdx
@@ -9,13 +9,14 @@ This page is a plain-markdown rendering of [`components/releases.data.ts`](https
{/* llms-tables:begin — generated by scripts/gen_llms_tables.py, do not edit */}
-Current stable release: v1.3.0 (Jul 20, 2026; container tag `1.3.0`, wheel version `1.3.0.post1`).
+Current stable release: v1.3.1 (Aug 5, 2026; container tag `1.3.1`, wheel version `1.3.1`).
## Releases
| Version | Kind | Date | SGLang | TensorRT-LLM | vLLM | NIXL (SGL / TRT / vLLM) | UCX | Notes | Delta |
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
-| main (ToT) | development head | - | 0.5.16 | 1.3.0rc23 | 0.26.0 | 1.3.0 / 1.0.1 / 1.3.1 | - | - | - |
+| main (ToT) | development head | - | 0.5.16 | 1.3.0rc23 | 0.26.0 | 1.3.0 / 1.3.1 / 1.3.1 | - | - | - |
+| v1.3.1 | patch | Aug 5, 2026 | 0.5.14 | 1.3.0rc19 | 0.23.0 | 1.3.2 / 1.0.1 / 1.1.0 | 1.20.x | [release notes](https://docs.nvidia.com/dynamo/dev/reference/releases/v1-3-0#v131) | Patch release. Fixes disaggregated SGLang serving over AWS EFA on GB200: the SGLang EFA runtime moves to NIXL 1.3.2 and all three EFA images to EFA Installer 1.49.0. Backend pins are otherwise unchanged from v1.3.0. |
| v1.3.0 | stable | Jul 20, 2026 | 0.5.14 | 1.3.0rc19 | 0.23.0 | 1.3.0 / 1.0.1 / 1.1.0 | 1.20.x | [release notes](https://docs.nvidia.com/dynamo/dev/reference/releases/v1-3-0) | CUDA 12 container images discontinued; EFA variants go multi-arch as -efa; GA wheels published as 1.3.0.post1 (containers stay :1.3.0); UCX 1.20.x. |
| v1.3.0-dev.1 | platform-preview | Jun 9, 2026 | 0.5.12.post1 | 1.3.0rc17 | 0.22.0 | 1.0.1 / 0.10.1 / 1.1.0 | - | [release notes](https://github.com/ai-dynamo/dynamo/releases/tag/v1.3.0-dev.1) | Full-platform preview of v1.3.0: complete runtime matrix, wheels on pypi.nvidia.com, crates, and Helm charts. Superseded by v1.3.0 GA. |
| v1.2.1 | patch | Jun 13, 2026 | 0.5.11 | 1.3.0rc14 | 0.20.1 | 1.0.1 / 0.10.1 / 0.10.1 | - | [release notes](https://docs.nvidia.com/dynamo/dev/reference/releases/v1-2-0) | Patch release. Same backend pins as v1.2.0. |
@@ -55,6 +56,9 @@ Release highlights (stable releases):
| Dynamo | Backend | CUDA Toolkit | Min Driver | Note |
| --- | --- | --- | --- | --- |
+| 1.3.1 | SGLang | 13.0 | 580.xx+ | - |
+| 1.3.1 | TensorRT-LLM | 13.1 | 580.xx+ | - |
+| 1.3.1 | vLLM | 13.0 | 580.xx+ | - |
| 1.3.0 | SGLang | 13.0 | 580.xx+ | - |
| 1.3.0 | TensorRT-LLM | 13.1 | 580.xx+ | - |
| 1.3.0 | vLLM | 13.0 | 580.xx+ | - |
@@ -120,7 +124,7 @@ Release highlights (stable releases):
- Early access v1.1.0-dev.* images follow the same CUDA matrix as v1.0.2. The v1.2.0-deepseek-v4-dev.3 vLLM container is CUDA 13.0 multi-arch; the SGLang containers split by arch (CUDA 12.9 on amd64, CUDA 13.0 on arm64).
- Experimental CUDA 13 images are not published for all versions.
-## Feature support by backend (v1.3.0)
+## Feature support by backend (v1.3.1)
| Feature | SGLang | TensorRT-LLM | vLLM |
| --- | --- | --- | --- |
@@ -140,34 +144,34 @@ Release highlights (stable releases):
| Shadow Engine Failover | Experimental (No KV-cache reuse or hardware fault tolerance) | Experimental (No KV-cache reuse or hardware fault tolerance) | Supported with caveat (Software-process failover only; no KV-cache reuse or hardware fault tolerance) |
| Dynamo Snapshot | Supported with caveat (Single-GPU supported; multi-GPU and multinode remain in progress) | Experimental (Single-GPU aggregated text-worker path only) | Supported with caveat (Single-GPU supported; multi-GPU is highly experimental and multinode remains in progress) |
-## Artifact inventory (v1.3.0)
+## Artifact inventory (v1.3.1)
| Category | Name | Description | Meta | Tags / install |
| --- | --- | --- | --- | --- |
-| container | vllm-runtime | vLLM backend runtime | vLLM v0.23.0 · CUDA 13.0 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.0`; `nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.0-efa` |
-| container | sglang-runtime | SGLang backend runtime | SGLang v0.5.14 · CUDA 13.0 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.0`; `nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.0-efa` |
-| container | tensorrtllm-runtime | TensorRT-LLM backend runtime | TRT-LLM v1.3.0rc19 · CUDA 13.1 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0`; `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.0-efa` |
-| container | dynamo-frontend | OpenAI-compatible API gateway with Endpoint Prediction Protocol (EPP) | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/dynamo-frontend:1.3.0` |
-| container | dynamo-planner | Standalone Planner used by Profiler jobs and Planner pods | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/dynamo-planner:1.3.0` |
-| container | kubernetes-operator | Operator that manages Dynamo deployments and CRDs | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/kubernetes-operator:1.3.0` |
-| container | snapshot-agent (Preview) | Fast GPU worker recovery via CRIU | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/snapshot-agent:1.3.0` |
-| wheel | ai-dynamo | Main package with backend integrations (vLLM, SGLang, TRT-LLM) | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install ai-dynamo==1.3.0.post1` |
-| wheel | ai-dynamo-runtime | Core Python bindings for the Dynamo runtime | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install ai-dynamo-runtime==1.3.0.post1` |
-| wheel | kvbm | KV Block Manager for disaggregated KV cache | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install kvbm==1.3.0.post1` |
+| container | vllm-runtime | vLLM backend runtime | vLLM v0.23.0 · CUDA 13.0 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.1`; `nvcr.io/nvidia/ai-dynamo/vllm-runtime:1.3.1-efa` |
+| container | sglang-runtime | SGLang backend runtime | SGLang v0.5.14 · CUDA 13.0 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.1`; `nvcr.io/nvidia/ai-dynamo/sglang-runtime:1.3.1-efa` |
+| container | tensorrtllm-runtime | TensorRT-LLM backend runtime | TRT-LLM v1.3.0rc19 · CUDA 13.1 · AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.1`; `nvcr.io/nvidia/ai-dynamo/tensorrtllm-runtime:1.3.1-efa` |
+| container | dynamo-frontend | OpenAI-compatible API gateway with Endpoint Prediction Protocol (EPP) | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/dynamo-frontend:1.3.1` |
+| container | dynamo-planner | Standalone Planner used by Profiler jobs and Planner pods | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/dynamo-planner:1.3.1` |
+| container | kubernetes-operator | Operator that manages Dynamo deployments and CRDs | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/kubernetes-operator:1.3.1` |
+| container | snapshot-agent (Preview) | Fast GPU worker recovery via CRIU | AMD64/ARM64 | `nvcr.io/nvidia/ai-dynamo/snapshot-agent:1.3.1` |
+| wheel | ai-dynamo | Main package with backend integrations (vLLM, SGLang, TRT-LLM) | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install ai-dynamo==1.3.1` |
+| wheel | ai-dynamo-runtime | Core Python bindings for the Dynamo runtime | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install ai-dynamo-runtime==1.3.1` |
+| wheel | kvbm | KV Block Manager for disaggregated KV cache | Python 3.10–3.12 · Linux (glibc v2.28+) | `uv pip install kvbm==1.3.1` |
| helm | dynamo-platform | Platform services (etcd, NATS) and the Dynamo Operator for a Dynamo cluster | - | `helm install dynamo-platform https://helm.ngc.nvidia.com/nvidia/ai-dynamo/charts/dynamo-platform-1.3.0.tgz` |
| helm | snapshot | Snapshot DaemonSet for fast GPU worker recovery | - | `helm install snapshot https://helm.ngc.nvidia.com/nvidia/ai-dynamo/charts/snapshot-1.3.0.tgz` |
-| crate | dynamo-runtime | Core distributed runtime library | MSRV Rust v1.82 | `cargo add dynamo-runtime@1.3.0` |
-| crate | dynamo-llm | LLM inference engine | MSRV Rust v1.82 | `cargo add dynamo-llm@1.3.0` |
+| crate | dynamo-runtime | Core distributed runtime library | MSRV Rust v1.82 | `cargo add dynamo-runtime@1.3.1` |
+| crate | dynamo-llm | LLM inference engine | MSRV Rust v1.82 | `cargo add dynamo-llm@1.3.1` |
| crate | dynamo-protocols | Async OpenAI-compatible API client | MSRV Rust v1.82 | `cargo add dynamo-protocols@1.3.0` |
| crate | dynamo-async-openai (Deprecated) | Legacy OpenAI client; use dynamo-protocols | MSRV Rust v1.82 · final release | `cargo add dynamo-async-openai@1.0.2` |
| crate | dynamo-parsers | Protocol parsers (SSE, JSON streaming) | MSRV Rust v1.82 | `cargo add dynamo-parsers@1.3.0` |
-| crate | dynamo-memory | Memory management utilities | MSRV Rust v1.82 | `cargo add dynamo-memory@1.3.0` |
+| crate | dynamo-memory | Memory management utilities | MSRV Rust v1.82 | `cargo add dynamo-memory@1.3.1` |
| crate | dynamo-config | Configuration management | MSRV Rust v1.82 | `cargo add dynamo-config@1.3.0` |
-| crate | dynamo-tokens | Tokenizer bindings for LLM inference | MSRV Rust v1.82 | `cargo add dynamo-tokens@1.3.0` |
-| crate | dynamo-tokenizers | Tokenizer library for LLM inference | MSRV Rust v1.82 | `cargo add dynamo-tokenizers@1.3.0` |
-| crate | dynamo-mocker | Inference engine simulator for benchmarking | MSRV Rust v1.82 | `cargo add dynamo-mocker@1.3.0` |
-| crate | dynamo-kv-router | KV-aware request routing library | MSRV Rust v1.82 | `cargo add dynamo-kv-router@1.3.0` |
-| crate | kvbm-logical | Logical layer for the KV Block Manager | MSRV Rust v1.82 | `cargo add kvbm-logical@1.3.0` |
+| crate | dynamo-tokens | Tokenizer bindings for LLM inference | MSRV Rust v1.82 | `cargo add dynamo-tokens@1.3.1` |
+| crate | dynamo-tokenizers | Tokenizer library for LLM inference | MSRV Rust v1.82 | `cargo add dynamo-tokenizers@1.3.1` |
+| crate | dynamo-mocker | Inference engine simulator for benchmarking | MSRV Rust v1.82 | `cargo add dynamo-mocker@1.3.1` |
+| crate | dynamo-kv-router | KV-aware request routing library | MSRV Rust v1.82 | `cargo add dynamo-kv-router@1.3.1` |
+| crate | kvbm-logical | Logical layer for the KV Block Manager | MSRV Rust v1.82 | `cargo add kvbm-logical@1.3.1` |
## Known artifact issues
diff --git a/docs/fern/pages/reference/general/releases/dynamo-v1-3-0.mdx b/docs/fern/pages/reference/general/releases/dynamo-v1-3-0.mdx
index e7d286677521..0c9e121ff7c4 100644
--- a/docs/fern/pages/reference/general/releases/dynamo-v1-3-0.mdx
+++ b/docs/fern/pages/reference/general/releases/dynamo-v1-3-0.mdx
@@ -2,7 +2,7 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
title: Dynamo v1.3.0
-subtitle: Release notes for Dynamo v1.3.0 (GA Jul 20, 2026)
+subtitle: Release notes for Dynamo v1.3.0 (GA Jul 20, 2026), including patch release v1.3.1
---
import { ReferenceStyles } from "@/components/ReferenceStyles";
@@ -447,3 +447,35 @@ Between v1.2.1 and v1.3.0 the project merged 930 PRs from 125 contributors. Than
If you would like to get involved, please see our [Contribution Guide](https://docs.nvidia.com/dynamo/dev/contributing/contribution-guide).
+
+## Patch releases
+
+
+
+### v1.3.1 — Aug 5, 2026
+
+#### Summary
+
+Dynamo v1.3.1 is a patch release on top of v1.3.0, addressing **disaggregated SGLang serving over AWS EFA on GB200**. The SGLang EFA runtime moves to the released **NIXL 1.3.2** wheel, and all three EFA runtime images move to **EFA Installer 1.49.0**.
+
+**Base Branch**: `release/1.3.1`
+
+#### Bug Fixes
+
+- **EFA / SGLang Disaggregated Serving on GB200:** Fixed a GB200-specific KV-transfer stall in which disaggregated SGLang inference hung and returned an empty response over AWS EFA ([#12106](https://github.com/ai-dynamo/dynamo/pull/12106)). The SGLang EFA runtime now installs the released `nixl==1.3.2` wheel in place of the NIXL built into the upstream framework image, and all three EFA images move to EFA Installer 1.49.0.
+
+#### Key Dependencies
+
+The following dependencies changed in this release:
+
+| Dependency | Version |
+| --- | --- |
+| NIXL (SGLang EFA runtime) | 1.3.2 (released wheel) |
+| EFA Installer | 1.49.0 |
+| libfabric (EFA runtime images) | 2.4.0amzn5.0 (EFA installer stock) |
+
+The EFA Installer and libfabric changes apply to all three `-efa` runtime images. The NIXL wheel change applies to `sglang-runtime` only. Backend runtime versions are unchanged from v1.3.0.
+
+#### Known Issues
+
+See [Known Issues](/dynamo/dev/reference/releases/known-issues#v131) for the v1.3.1 list.
diff --git a/docs/fern/pages/reference/general/releases/known-issues.mdx b/docs/fern/pages/reference/general/releases/known-issues.mdx
index afe3fae2cccd..493950523e53 100644
--- a/docs/fern/pages/reference/general/releases/known-issues.mdx
+++ b/docs/fern/pages/reference/general/releases/known-issues.mdx
@@ -12,9 +12,19 @@ import { RELEASE_STATS } from "@/components/releases.data";
Known issues are mirrored verbatim from each release's GitHub release notes. The current release is listed first, with older releases collapsed below; artifact-specific issues sit at the bottom of the page.
+
+
+## v1.3.1 — current release
+
+SGLang **Intermittent Stall in Disaggregated Serving over EFA:** Requests return an empty HTTP 200 response with zero completion tokens after approximately 300 seconds. The condition is intermittent and more likely on newly started decode workers, including after a restart. It affects SGLang disaggregated serving on the NIXL LIBFABRIC backend over EFA; other transports are unaffected. The defect is in the EFA provider in libfabric. The data path and the network are healthy and EFA counters show no drops or errors, so the failure surfaces as a silent stall with no error reported. **Targeted fix:** an AWS EFA release in the coming weeks. It is not included in this release.
+
+SGLang **Partial EFA Device Allocation Can Mismatch GPU PCIe Topology:** Workers that request fewer than all EFA devices on a node may be allocated devices that are not on the same PCIe switch as their GPUs, and disaggregated KV transfer fails. Affected requests return an empty HTTP 200 response with zero completion tokens after approximately 10 to 20 seconds, and the decode worker logs `Lost connection with prefill instance`. This is a limitation of EFA device allocation on Kubernetes and is not specific to Dynamo: GPUs and EFA devices are allocated by two independent device plugins, and upstream Kubernetes provides no mechanism for one plugin to observe another's allocations. The impact is greatest where each GPU is paired one-to-one with an EFA NIC, as on GB200, since a mismatched allocation leaves a GPU with no local network device. **Mitigations,** in order of preference: request all EFA devices on the node, which constrains scheduling to one worker per node; or use EFA DRA together with NVIDIA DRA, which allows requesting a GPU and EFA devices on the same PCIe switch.
+
+Mocker **`dynamo-mocker` `aic-forward-pass` Feature Fails to Compile from crates.io:** The published `dynamo-mocker` 1.3.0 and 1.3.1 crates do not compile with `aic-forward-pass` enabled, or with `--all-features`. Their `aiconfigurator-core` requirement resolves to `0.10.0`, which added required `cp_size` and `perf_db_sources` fields that the mocker source does not supply. Consumers building with default features are unaffected, as are the published wheels and containers: the wheel builder compiles the feature from local workspace source at the pinned commit and does not read the crates.io metadata. No published `aiconfigurator-core` version satisfies the feature, since `0.9.0` predates the pinned commit and fails on missing APIs. **Workaround:** Build `dynamo-mocker` from release-branch source. **Targeted fix:** A corrective `dynamo-mocker` release once a matching `aiconfigurator-core` is published.
+
-## v1.3.0 — current release
+## v1.3.0
vLLM **GMS + Expert Parallel Crash at Model Load:** When running the GPU Memory Service with Expert Parallel, `Tensor.item()` is called on a meta tensor in `ExpertMapManager` and the worker crashes at model load. The root cause is an upstream vLLM defect (fix in flight as vllm-project/vllm#43928, not yet in v0.23.0 or v0.24.0). **Workaround:** Avoid GMS with Expert Parallel (MoE) on vLLM until the upstream vLLM fix ships.
@@ -22,15 +32,15 @@ Known issues are mirrored verbatim from each release's GitHub release notes. The
vLLM **Disagg Decode + EAGLE3 Speculative Decoding Crashes Silently:** In a disaggregated deployment, EAGLE3 speculative decoding crashes silently under large KV-transfer volume (an upstream vLLM defect). The crash reproduces even with speculative decoding enabled on both the prefill and decode workers using an identical `--speculative_config`. **Workaround:** None reliable in v1.3.0 — EAGLE3 speculative decoding with disaggregated serving is unsupported. **Targeted fix:** v1.4.0.
-vLLM **KVBM PdConnector Disaggregated E2E Broken (NIXL VRAM_SEG):** All four disaggregated KVBM/LMCache configurations fail the NIXL `VRAM_SEG` handshake (aggregated paths pass). The KVBM PD connector subclasses vLLM's `MultiConnector`, which in vLLM v0.23.0 inherits `SupportsHMA` and wrongly assumes the PD connector is HMA-capable, tripping the handshake at `MultiConnector.__init__`. **Workaround:** Pass `--disable-hybrid-kv-cache-manager` on both prefill and decode workers. **Targeted fix:** v1.3.1.
+vLLM **KVBM PdConnector Disaggregated E2E Broken (NIXL VRAM_SEG):** All four disaggregated KVBM/LMCache configurations fail the NIXL `VRAM_SEG` handshake (aggregated paths pass). The KVBM PD connector subclasses vLLM's `MultiConnector`, which in vLLM v0.23.0 inherits `SupportsHMA` and wrongly assumes the PD connector is HMA-capable, tripping the handshake at `MultiConnector.__init__`. **Workaround:** Pass `--disable-hybrid-kv-cache-manager` on both prefill and decode workers. **Targeted fix:** v1.4.0.
TRT-LLM **Qwen2-VL-7B Multimodal Inference Fails:** Qwen2-VL-7B multimodal inference fails with `AssertionError: Number of mm_embeds (2) does not match expected total (3897)`. This is an upstream TensorRT-LLM regression introduced across the rc14→rc19 bump (transformers 4.57.3→5.5.4); it affects Qwen2-VL-7B only — Qwen3-VL is unaffected. **Workaround:** Use Qwen3-VL. **Targeted fix:** v1.4.0, contingent on the upstream TensorRT-LLM fix.
TRT-LLM **Crashed TensorRT-LLM Worker Keeps Receiving Traffic and Is Never Restarted:** After a TensorRT-LLM engine-core OOM or SIGKILL, `check_health()` still reports healthy, so `TrtllmEngineMonitor` never trips and the worker keeps receiving requests it can no longer serve while the pod is never restarted. Correct idle-gap fatal-state detection requires an upstream TensorRT-LLM change; this is pre-existing behavior, not a v1.3.0 regression. **Workaround:** Manually restart the affected worker. **Targeted fix:** pending upstream TensorRT-LLM.
-TRT-LLM **Wide-EP Decode Crashes on All Ranks:** Wide Expert-Parallel decode crashes on all WideEP ranks due to a `run_moe()` argument-count mismatch (reproduced since rc1/rc2). The root cause is an upstream TensorRT-LLM ABI skew (a 37-vs-44 argument mismatch present in the rc18–rc20 bases, fixed upstream in rc21); v1.3.0 stays on the rc19 base rather than taking the destabilizing rc21 bump. **Workaround:** Rebuild against a TensorRT-LLM rc21 base, where the upstream `run_moe()` fix is present — v1.3.0 ships the rc21-compatibility changes ([#11769](https://github.com/ai-dynamo/dynamo/pull/11769), [#11799](https://github.com/ai-dynamo/dynamo/pull/11799)) so the Dynamo TensorRT-LLM worker builds against rc21. **Targeted fix:** v1.3.1.
+TRT-LLM **Wide-EP Decode Crashes on All Ranks:** Wide Expert-Parallel decode crashes on all WideEP ranks due to a `run_moe()` argument-count mismatch (reproduced since rc1/rc2). The root cause is an upstream TensorRT-LLM ABI skew (a 37-vs-44 argument mismatch present in the rc18–rc20 bases, fixed upstream in rc21); v1.3.0 stays on the rc19 base rather than taking the destabilizing rc21 bump. **Workaround:** Rebuild against a TensorRT-LLM rc21 base, where the upstream `run_moe()` fix is present — v1.3.0 ships the rc21-compatibility changes ([#11769](https://github.com/ai-dynamo/dynamo/pull/11769), [#11799](https://github.com/ai-dynamo/dynamo/pull/11799)) so the Dynamo TensorRT-LLM worker builds against rc21. **Targeted fix:** v1.4.0.
-SGLang **Disaggregated SGLang Inference over EFA Stalls and Returns Empty Response:** On AWS EFA clusters (GB200 and H100 / P5), the first disaggregated end-to-end SGLang request stalls for roughly 300 seconds and then returns HTTP 200 with empty content and zero completion tokens, because the KV cache is never transferred. The root cause is a NIXL `FI_MORE` multi-rail write-batching deadlock in the KV transfer over the EFA LIBFABRIC backend. **Workaround:** Avoid EFA-based disaggregated SGLang serving; use aggregated serving or a non-EFA transport instead. **Targeted fix:** v1.3.1.
+SGLang **Disaggregated SGLang Inference over EFA Stalls and Returns Empty Response:** On AWS EFA clusters (GB200 and H100 / P5), the first disaggregated end-to-end SGLang request stalls for roughly 300 seconds and then returns HTTP 200 with empty content and zero completion tokens, because the KV cache is never transferred. The root cause is a NIXL `FI_MORE` multi-rail write-batching deadlock in the KV transfer over the EFA LIBFABRIC backend. **Workaround:** Avoid EFA-based disaggregated SGLang serving; use aggregated serving or a non-EFA transport instead. **Targeted fix:** v1.3.1 resolves this on GB200. H100 / P5 remain affected, pending an AWS EFA release.
Planner **Planner SLA and Load-Based Autoscaling Does Not Scale Deployments:** SLA and load-based scaling is an advanced, opt-in path; default throughput-based scaling and core disaggregated serving are unaffected. On that path, decode-side KV-rate scaling never fires on TensorRT-LLM (workers do not report `total_kv_blocks` / `kv_cache_block_size`), disagg-prefill scale-up never triggers on SGLang (the prefill-token signal omits the chunked-prefill backlog), the MTP accept-length discount never reaches replica decisions (accept length pins at 1.0), and GlobalPlanner scale operations fail while reading the deprecated DGD `spec.services` field. **Workaround:** Use throughput-based Planner scaling. **Targeted fix:** v1.4.0.