Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 0 additions & 5 deletions docs/docs.json
Original file line number Diff line number Diff line change
Expand Up @@ -149,10 +149,6 @@
"source": "/advanced_features/pd_disaggregation.html",
"destination": "/docs/advanced_features/pd_disaggregation"
},
{
"source": "/advanced_features/piecewise_cuda_graph.html",
"destination": "/docs/advanced_features/piecewise_cuda_graph"
},
{
"source": "/advanced_features/pipeline_parallelism.html",
"destination": "/docs/advanced_features/pipeline_parallelism"
Expand Down Expand Up @@ -964,7 +960,6 @@
"docs/advanced_features/dp_for_multi_modal_encoder",
"docs/advanced_features/cuda_graph_for_multi_modal_encoder",
"docs/advanced_features/breakable_cuda_graph",
"docs/advanced_features/piecewise_cuda_graph",
"docs/advanced_features/sgl_model_gateway",
"docs/advanced_features/llm-d",
"docs/advanced_features/deterministic_inference",
Expand Down
2 changes: 1 addition & 1 deletion docs/docs/advanced_features/breakable_cuda_graph.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -35,7 +35,7 @@ For production use, you can mark specific functions as "non-graphable" using the
```python
from sglang.srt.model_executor.runner_backend_utils.breakable_cuda_graph import eager_on_graph

@eager_on_graph(enable=True)
@eager_on_graph
def my_dynamic_op(x):
# This op is incompatible with CUDA graph capture
return some_dynamic_operation(x)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -69,7 +69,7 @@ it; naming a backend locks the choice and skips that rule:
SGLANG_VIT_ENABLE_CUDA_GRAPH=1 \
python3 -m sglang.launch_server \
--model Qwen/Qwen3-VL-8B-Instruct \
--cuda-graph-backend-prefill tc_piecewise \
--cuda-graph-backend-prefill breakable \
--cuda-graph-max-bs-prefill 4096 \
--cuda-graph-tc-compiler eager
```
Expand Down
289 changes: 0 additions & 289 deletions docs/docs/advanced_features/piecewise_cuda_graph.mdx

This file was deleted.

12 changes: 3 additions & 9 deletions docs/docs/advanced_features/server_arguments.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -2408,21 +2408,21 @@ Combining `--enable-response-store` with `--disaggregation-mode=prefill` or `dec
</tr>
<tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>`--cuda-graph-config`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>Canonical per-phase CUDA graph settings as JSON, e.g. <code>{`{"decode":{"backend":"full","max_bs":256},"prefill":{"backend":"tc_piecewise","tc_compiler":"eager"}}`}</code>. JSON wins over the per-phase <code>--cuda-graph-*</code> convenience flags and over the legacy flags. Allowed backends: <code>full</code>, <code>breakable</code>, <code>tc_piecewise</code>, <code>disabled</code> (<code>full</code> is decode-only).</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>Canonical per-phase CUDA graph settings as JSON, e.g. <code>{`{"decode":{"backend":"full","max_bs":256},"prefill":{"backend":"breakable"}}`}</code>. JSON wins over the per-phase <code>--cuda-graph-*</code> convenience flags and over the legacy flags. Allowed backends: <code>full</code>, <code>breakable</code>, <code>disabled</code>.</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>`None`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>Type: JSON (dict-of-dicts)</td>
</tr>
<tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>`--cuda-graph-backend-decode`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>Backend for the decode phase. Folds into <code>cuda_graph_config[decode].backend</code>.</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>`None`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}><code>full</code>, <code>breakable</code>, <code>tc_piecewise</code>, <code>disabled</code></td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}><code>full</code>, <code>breakable</code>, <code>disabled</code></td>
</tr>
<tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>`--cuda-graph-backend-prefill`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>Backend for the prefill phase. Folds into <code>cuda_graph_config[prefill].backend</code>.</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>`None`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}><code>breakable</code>, <code>tc_piecewise</code>, <code>disabled</code></td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}><code>breakable</code>, <code>disabled</code></td>
</tr>
<tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>`--cuda-graph-max-bs-decode`</td>
Expand Down Expand Up @@ -2454,12 +2454,6 @@ Combining `--enable-response-store` with `--disaggregation-mode=prefill` or `dec
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>`None`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>List[int]</td>
</tr>
<tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>`--cuda-graph-tc-compiler`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>Compiler used by the <code>tc_piecewise</code> backend (only the prefill phase consumes it today).</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>`None`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}><code>eager</code>, <code>inductor</code></td>
</tr>
<tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>`--disable-cuda-graph-padding`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>Disable cuda graph when padding is needed. Still uses cuda graph when padding is not needed.</td>
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -81,7 +81,7 @@ python3 -m sglang.launch_server \
--swa-full-tokens-ratio 0.3 \
--disaggregation-mode prefill --disaggregation-transfer-backend ascend \
--disaggregation-bootstrap-port 8996 \
--disable-piecewise-cuda-graph \
--cuda-graph-backend-prefill disabled \
--dp-size 2 --enable-dp-attention --enable-dp-lm-head \
--moe-a2a-backend deepep --deepep-mode normal
```
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2061,7 +2061,7 @@ If the value is int8, you must also set the environment variable:DEEP_NORMAL_MOD
<tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>`--cuda-graph-backend-prefill`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>`None`</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>`disabled`, `tc_piecewise`<br/> (`tc_piecewise` currently supports Llama-3.1-8B-Instruct and Qwen2.5-7B-Instruct)</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>`disabled`<br/> (the former `tc_piecewise` prefill backend has been removed)</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>A2/A3 Series</td>
</tr>
<tr>
Expand Down
10 changes: 0 additions & 10 deletions docs/docs/hardware-platforms/plugin.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -556,11 +556,6 @@ python -c "from sglang.srt.platforms import current_platform; print(current_plat
<td><code>False</code></td>
<td>Whether device graph capture is supported (plain CUDA graph)</td>
</tr>
<tr>
<td><code>support_piecewise_cuda_graph()</code></td>
<td><code>False</code></td>
<td>Whether piecewise CUDA graph (torch.compile backend) is supported</td>
</tr>
<tr>
<td><code>supports_fp8()</code></td>
<td><code>False</code></td>
Expand Down Expand Up @@ -620,11 +615,6 @@ python -c "from sglang.srt.platforms import current_platform; print(current_plat
<td><code>raise NotImplementedError</code></td>
<td>Return hardware-specific quantization config for the specific quantization scheme, raise an error if not supported or return None to use the default config.</td>
</tr>
<tr>
<td><code>get_piecewise_backend_cls()</code></td>
<td><code>raise NotImplementedError</code></td>
<td>Piecewise compilation backend class</td>
</tr>
<tr>
<td><code>get_compile_backend(mode)</code></td>
<td><code>"inductor"</code></td>
Expand Down
36 changes: 5 additions & 31 deletions docs/docs/hardware-platforms/xpu.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -177,7 +177,6 @@ SGLang enables XPU graph capture to reduce per-step kernel-launch overhead.
| Phase | Backend | Mechanism | Default |
|---|---|---|---|
| Decode | `full` | One `torch.xpu.XPUGraph` per batch size, captured on startup | **Off** (opt-in) |
| Prefill | `tc_piecewise` | `torch.compile` + XPU graph, one graph segment per token-length bucket | **Off** (opt-in) |
| Prefill | `breakable` | Segmented `torch.xpu.XPUGraph` capture/replay (no `torch.compile`); eager break points at attention / MoE boundaries | **Off** (opt-in) |

### Enable Decode Graph
Expand All @@ -192,26 +191,7 @@ python -m sglang.launch_server --model-path <MODEL> --device xpu \
### Enable Prefill Graph

Prefill graph capture is **opt-in** on XPU and must be enabled explicitly.
Two backends are available: `tc_piecewise` and `breakable`.

#### tc_piecewise

Uses `torch.compile` plus an XPU graph, one graph segment per token-length
bucket:

```bash
python -m sglang.launch_server --model-path <MODEL> --device xpu \
--cuda-graph-backend-prefill tc_piecewise
```

By default the prefill subgraphs are compiled with `eager` mode. Switch to
`inductor` for higher-quality generated code at the cost of longer startup:

```bash
python -m sglang.launch_server --model-path <MODEL> --device xpu \
--cuda-graph-backend-prefill tc_piecewise \
--cuda-graph-tc-compiler inductor
```
Use the `breakable` backend. The former `tc_piecewise` backend has been removed.

#### breakable

Expand All @@ -227,7 +207,7 @@ You can also configure both phases together with a single `--cuda-graph-config`

```bash
python -m sglang.launch_server --model-path <MODEL> --device xpu \
--cuda-graph-config '{"decode":{"backend":"full"},"prefill":{"backend":"tc_piecewise","tc_compiler":"eager"}}'
--cuda-graph-config '{"decode":{"backend":"full"},"prefill":{"backend":"breakable"}}'
```

### Enable torch.compile for Decode
Expand All @@ -242,11 +222,6 @@ python -m sglang.launch_server --model-path <MODEL> --device xpu \
--enable-torch-compile
```

> **Note:** `--enable-torch-compile` is mutually exclusive with the prefill
> `tc_piecewise` graph (the compatibility rules auto-disable it). Use them
> separately or lock the prefill backend explicitly via `--cuda-graph-config`
> if you need both.
### Disable XPU Graph

Both phases are disabled by default. To explicitly disable them anyway:
Expand Down Expand Up @@ -274,7 +249,7 @@ To specify explicit token-length buckets:
```bash
python -m sglang.launch_server \
--model-path <MODEL> --device xpu \
--cuda-graph-backend-prefill tc_piecewise \
--cuda-graph-backend-prefill breakable \
--cuda-graph-bs-prefill 64 128 256 512
```

Expand All @@ -291,11 +266,10 @@ python -m sglang.launch_server \
| Argument | XPU allowed values | Default | Description |
|---|---|---|---|
| `--cuda-graph-backend-decode` | `full`, `disabled` | `disabled` | Backend for the decode phase. Only `full` is supported on XPU. Set to `full` to enable. |
| `--cuda-graph-backend-prefill` | `tc_piecewise`, `breakable`, `disabled` | `disabled`* | Backend for the prefill phase. Set to `tc_piecewise` or `breakable` explicitly to enable. |
| `--cuda-graph-tc-compiler` | `eager`, `inductor` | `eager` | Compiler for `tc_piecewise` prefill subgraphs. `inductor` produces more optimized code but has longer startup. |
| `--cuda-graph-backend-prefill` | `breakable`, `disabled` | `disabled`* | Backend for the prefill phase. Set to `breakable` explicitly to enable. |
| `--cuda-graph-bs-prefill` | list of ints | auto | Explicit token-length buckets to capture for prefill. |
| `--cuda-graph-bs-decode` | list of ints | auto | Explicit batch sizes to capture for decode. |
| `--cuda-graph-config` | JSON string | — | One-shot JSON config for both phases, e.g. `'{"decode":{"backend":"full"},"prefill":{"backend":"tc_piecewise","tc_compiler":"eager"}}'`. Overrides all per-phase flags. |
| `--cuda-graph-config` | JSON string | — | One-shot JSON config for both phases, e.g. `'{"decode":{"backend":"full"},"prefill":{"backend":"breakable"}}'`. Overrides all per-phase flags. |
| `--disable-decode-cuda-graph` | — | `False` | Shorthand for `--cuda-graph-backend-decode=disabled`. |
| `--disable-prefill-cuda-graph` | — | `False` | Shorthand for `--cuda-graph-backend-prefill=disabled`. |
| `--enable-torch-compile` | — | `False` | Apply `torch.compile` on top of the decode XPU graph for further kernel optimization. |
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -17,11 +17,6 @@

from sglang.kernels.jit.utils import is_arch_support_pdl
from sglang.srt.batch_invariant_ops import is_batch_invariant_mode_enabled
from sglang.srt.model_executor.cuda_graph_config import (
Backend,
Phase,
check_cuda_graph_backend,
)
from sglang.srt.utils import (
cdiv,
cpu_has_amx_support,
Expand Down Expand Up @@ -211,9 +206,7 @@ def _get_sm_count(device: torch.device) -> int:

def calc_rows_per_block(M: int, device: torch.device) -> int:
# Use a constant value when the row count must not affect kernel numerics.
if is_batch_invariant_mode_enabled() or check_cuda_graph_backend(
Phase.PREFILL, Backend.TC_PIECEWISE
):
if is_batch_invariant_mode_enabled():
return MAX_ROWS_PER_BLOCK
sm_count = _get_sm_count(device)
rows_per_block = next_power_of_2(cdiv(M, 2 * sm_count))
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2101,16 +2101,15 @@ def _make_breakable_attention_forward(forward_method):
disabled this is a transparent pass-through to the original method.
"""

def _forward_boxing_tuples(*args, **kwargs):
@eager_on_graph
def _eager_attention(*args, **kwargs):
out = forward_method(*args, **kwargs)
return _BCGBoxedTupleOutput(out) if isinstance(out, tuple) else out

bcg_forward = eager_on_graph(True)(_forward_boxing_tuples)

@functools.wraps(forward_method)
def forward(self, *args, **kwargs):
if is_in_breakable_cuda_graph():
out = bcg_forward(self, *args, **kwargs)
out = _eager_attention(self, *args, **kwargs)
return out.astuple() if isinstance(out, _BCGBoxedTupleOutput) else out
return forward_method(self, *args, **kwargs)

Expand Down
14 changes: 8 additions & 6 deletions python/sglang/multimodal_gen/runtime/models/dits/minimax_h3.py
Original file line number Diff line number Diff line change
Expand Up @@ -775,7 +775,9 @@ def _minimax_h3_attention_core_impl(
return out


_minimax_h3_attention_core_bcg = eager_on_graph(True)(_minimax_h3_attention_core_impl)
@eager_on_graph
def _eager_attention_core(*args, **kwargs):
return _minimax_h3_attention_core_impl(*args, **kwargs)


class MiniMaxH3Attention(nn.Module):
Expand Down Expand Up @@ -1173,7 +1175,7 @@ def forward(
gate_compress = gate_flat.view(total, self.num_heads, self.head_dim)

attention_core = (
_minimax_h3_attention_core_bcg
_eager_attention_core
if self.bcg_breakpoint
else _minimax_h3_attention_core_impl
)
Expand Down Expand Up @@ -2395,8 +2397,8 @@ def build_rope_cache(
self.release_mps_non_layer_weights("rope")
return result

@eager_on_graph(True)
def _embed(
@eager_on_graph
def _eager_embed(
self,
*,
x: torch.Tensor,
Expand Down Expand Up @@ -2428,7 +2430,7 @@ def _embed(
elif torch.is_tensor(refined_prompt_embeds_length):
# BCG turns this request-varying host constant into a scalar input
# so different live lengths can replay one padded-text signature.
# _embed is an eager graph break, so this value is read outside
# _eager_embed is an eager graph break, so this value is read outside
# captured CUDA graphs.
text_len = int(refined_prompt_embeds_length.item())
else:
Expand Down Expand Up @@ -2692,7 +2694,7 @@ def forward(self, **kwargs: Any) -> tuple[torch.Tensor, torch.Tensor]:
audio_pos = audio_pos.to(device)
text_pos = text_pos.to(device)

decoder_input, t_emb = self._embed(
decoder_input, t_emb = self._eager_embed(
x=x,
audio_x=audio_x,
text_embeddings_selected=text_selected,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -140,7 +140,7 @@ def forward(
beta = self.linear_attention.beta(x)
gate_hidden, _ = self.linear_attention.output_gate.down(x)
attention_core = (
_hybrid_attention_core_bcg
_eager_hybrid_attention_core
if attention.bcg_breakpoint
else _minimax_h3_hybrid_attention_core_impl
)
Expand Down Expand Up @@ -563,9 +563,9 @@ def _vdn_return_to_rows(
return merged[0], linear_rows


_hybrid_attention_core_bcg = eager_on_graph(True)(
_minimax_h3_hybrid_attention_core_impl
)
@eager_on_graph
def _eager_hybrid_attention_core(*args, **kwargs):
return _minimax_h3_hybrid_attention_core_impl(*args, **kwargs)


def prepare_hybrid_attention_metadata(
Expand Down
Loading
Loading