SemiAnalysisAI · Oseltamivir · Apr 30, 2026 · Apr 25, 2026 · Apr 25, 2026 · Apr 25, 2026
@@ -7666,3 +7666,115 @@ dsv4-fp4-gb200-dynamo-vllm:
         tp: 16
         ep: 16
         dp-attn: true
+
+dsv4-fp4-gb200-dynamo-sglang:
+  image: lmsysorg/sglang:deepseek-v4-grace-blackwell
+  model: deepseek-ai/DeepSeek-V4-Pro
+  model-prefix: dsv4
+  runner: gb200
+  precision: fp4
+  framework: dynamo-sglang
+  multinode: true
+  disagg: true
+  seq-len-configs:
+  # 1k/1k — hand-rolled. NVIDIA/srt-slurm has no DSV4 sglang disagg
+  # recipe yet; topologies match the dsv4-fp4-gb200-dynamo-vllm sibling
+  # so framework-level numbers are directly comparable. Per-worker
+  # tunings cross-reference benchmarks/single_node/dsv4_fp4_b200.sh and
+  # NVIDIA/srt-slurm@sa-submission-q2-2026 recipes/gb200-fp4/1k1k/*.yaml
+  # (DSR1 sglang disagg structure).
+  - isl: 1024
+    osl: 1024
+    search-space:
+    # Low-concurrency / interactivity: 1 prefill (DP=8) + 1 decode (TP=8). 4 nodes.
+    - conc-list: [1, 4, 8, 16, 32, 64]
+      prefill:
+        num-worker: 1
+        tp: 8
+        ep: 8
+        dp-attn: true
+        additional-settings:
+        - "CONFIG_FILE=recipes/sglang/deepseek-v4/1k1k/disagg-gb200-1p1d-dep8-tep8.yaml"
+      decode:
+        num-worker: 1
+        tp: 8
+        ep: 1
+        dp-attn: false
+    # Mid throughput: 1 prefill (DP=8) + 1 wide decode (DP=16). 6 nodes.
+    - conc-list: [128, 256, 1024, 2048, 4096]
+      prefill:
+        num-worker: 1
+        tp: 8
+        ep: 8
+        dp-attn: true
+        additional-settings:
+        - "CONFIG_FILE=recipes/sglang/deepseek-v4/1k1k/disagg-gb200-1p1d-dep8-dep16.yaml"
+      decode:
+        num-worker: 1
+        tp: 16
+        ep: 16
+        dp-attn: true
+    # High throughput: 3 prefills (DP=8) + 1 wide decode (DP=16). 10 nodes.
+    # 4096 overlap with the 1p1d block gives a topology-crossover A/B.
+    - conc-list: [4096, 8192]
+      prefill:
+        num-worker: 3
+        tp: 8
+        ep: 8
+        dp-attn: true
+        additional-settings:
+        - "CONFIG_FILE=recipes/sglang/deepseek-v4/1k1k/disagg-gb200-3p1d-dep8-dep16.yaml"
+      decode:
+        num-worker: 1
+        tp: 16
+        ep: 16
+        dp-attn: true
+
+  # 8k/1k block kept commented out — same rationale as the dsv4-fp4-
+  # gb200-dynamo-vllm sibling: keep `sweep-enabled` runtime bounded.
+  # Uncomment to re-enable (recipes are already in place).
+  # - isl: 8192
+  #   osl: 1024
+  #   search-space:
+  #   # Low-concurrency: 1 prefill (DP=8) + 1 decode (TP=8). 4 nodes.
+  #   - conc-list: [1, 4, 8, 16, 32, 64]
+  #     prefill:
+  #       num-worker: 1
+  #       tp: 8
+  #       ep: 8
+  #       dp-attn: true
+  #       additional-settings:
+  #       - "CONFIG_FILE=recipes/sglang/deepseek-v4/8k1k/disagg-gb200-1p1d-dep8-tep8.yaml"
+  #     decode:
+  #       num-worker: 1
+  #       tp: 8
+  #       ep: 1
+  #       dp-attn: false
+  #   # Mid: 3 prefills (DP=8) + 1 wide decode (DP=16). 10 nodes.
+  #   - conc-list: [512, 1024]
+  #     prefill:
+  #       num-worker: 3
+  #       tp: 8
+  #       ep: 8
+  #       dp-attn: true
+  #       additional-settings:
+  #       - "CONFIG_FILE=recipes/sglang/deepseek-v4/8k1k/disagg-gb200-3p1d-dep8-dep16.yaml"
+  #     decode:
+  #       num-worker: 1
+  #       tp: 16
+  #       ep: 16
+  #       dp-attn: true
+  #   # Max throughput: 7 prefills (DP=8) + 1 wide decode (DP=16). 18 nodes.
+  #   - conc-list: [4096, 8192]
+  #     prefill:
+  #       num-worker: 7
+  #       tp: 8
+  #       ep: 8
+  #       dp-attn: true
+  #       additional-settings:
+  #       - "CONFIG_FILE=recipes/sglang/deepseek-v4/8k1k/disagg-gb200-7p1d-dep8-dep16.yaml"
+  #     decode:
+  #       num-worker: 1
+  #       tp: 16
+  #       ep: 16
+  #       dp-attn: true
diff --git a/...ks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/1k1k/disagg-gb200-1p1d-dep8-dep16.yaml b/...ks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/1k1k/disagg-gb200-1p1d-dep8-dep16.yaml
@@ -0,0 +1,110 @@
+name: "dsv4-sglang-disagg-gb200-1p1d-dep8-dep16"
+
+# Hand-rolled — see ./disagg-gb200-1p1d-dep8-tep8.yaml header for the
+# upstream-reference list (PR #69 GB200 agg, PR #75 GB300 disagg).
+# Topology mirrors the dsv4-fp4-gb200-dynamo-vllm sibling.
+#
+# Topology: 1 prefill (DP=8 EP=8) + 1 decode (DP=16 EP=16). 6 nodes.
+# Single prefill is enough for 1k prompts up to ~conc 4096 (per-rank
+# prefill TFlops at 1k ISL is high; matches the vLLM sibling sizing).
+
+model:
+  path: "deepseek-v4-pro"
+  container: "lmsysorg/sglang:deepseek-v4-grace-blackwell"
+  precision: "fp4"
+
+dynamo:
+  version: 0.8.1
+
+slurm:
+  time_limit: "8:00:00"
+
+health_check:
+  max_attempts: 1440
+  interval_seconds: 10
+
+resources:
+  gpu_type: "gb200"
+  gpus_per_node: 4
+  prefill_nodes: 2
+  decode_nodes: 4
+  prefill_workers: 1
+  decode_workers: 1
+  gpus_per_prefill: 8
+  gpus_per_decode: 16
+
+frontend:
+  type: dynamo
+  enable_multiple_frontends: false
+
+backend:
+  type: sglang
+  connector: null
+
+  prefill_environment:
+    PYTHONUNBUFFERED: "1"
+    SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0"
+    NCCL_MNNVL_ENABLE: "1"
+    NCCL_CUMEM_ENABLE: "1"
+    SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000"
+    SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000"
+    SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000"
+
+  decode_environment:
+    PYTHONUNBUFFERED: "1"
+    SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0"
+    NCCL_MNNVL_ENABLE: "1"
+    NCCL_CUMEM_ENABLE: "1"
+    SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000"
+    SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000"
+    SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000"
+
+  sglang_config:
+    prefill:
+      served-model-name: "deepseek-ai/DeepSeek-V4-Pro"
+      trust-remote-code: true
+      tensor-parallel-size: 8
+      dp-size: 8
+      ep-size: 8
+      enable-dp-attention: true
+      moe-a2a-backend: "deepep"
+      moe-runner-backend: "flashinfer_mxfp4"
+      chunked-prefill-size: 4096
+      disable-flashinfer-autotune: true
+      disable-radix-cache: true
+      mem-fraction-static: 0.82
+      context-length: 3072
+      max-running-requests: 16
+      stream-interval: 50
+      decode-log-interval: 1000
+      disaggregation-mode: "prefill"
+      disaggregation-transfer-backend: nixl
+
+    decode:
+      served-model-name: "deepseek-ai/DeepSeek-V4-Pro"
+      trust-remote-code: true
+      tensor-parallel-size: 16
+      dp-size: 16
+      ep-size: 16
+      enable-dp-attention: true
+      moe-a2a-backend: "deepep"
+      moe-runner-backend: "flashinfer_mxfp4"
+      chunked-prefill-size: 4096
+      disable-flashinfer-autotune: true
+      disable-radix-cache: true
+      mem-fraction-static: 0.82
+      context-length: 3072
+      max-running-requests: 512
+      cuda-graph-max-bs: 512
+      stream-interval: 50
+      decode-log-interval: 1000
+      disaggregation-mode: "decode"
+      disaggregation-transfer-backend: nixl
+
+benchmark:
+  type: "sa-bench"
+  isl: 1024
+  osl: 1024
+  concurrencies: "128x256x1024x2048x4096"
+  req_rate: "inf"
+  use_chat_template: false
diff --git a/...rks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/1k1k/disagg-gb200-1p1d-dep8-tep8.yaml b/...rks/multi_node/srt-slurm-recipes/sglang/deepseek-v4/1k1k/disagg-gb200-1p1d-dep8-tep8.yaml
@@ -0,0 +1,115 @@
+name: "dsv4-sglang-disagg-gb200-1p1d-dep8-tep8"
+
+# Hand-rolled — no GB200 DSV4 sglang disagg recipe exists upstream. The
+# closest references on NVIDIA/srt-slurm are:
+#   * PR #69 (recipes/gb200-fp4/1k1k-dsv4/agg-2n-low-latency.yaml) —
+#     GB200 DSV4 sglang AGGREGATED: per-worker flag set + env vars.
+#   * PR #75 (recipes/gb300-fp4/1k1k-dsv4/disagg-1p1d-tp4-mxfp4.yaml) —
+#     GB300 DSV4 sglang DISAGG: confirms nixl + flashinfer_mxfp4 +
+#     chunked-prefill-size=4096 + disable-flashinfer-autotune.
+# Topology mirrors the dsv4-fp4-gb200-dynamo-vllm sibling so cross-
+# framework numbers stay directly comparable.
+#
+# Topology: 1 prefill (DP=8 EP=8) + 1 decode (TP=8, no DP-attn). 4 nodes.
+# Targets very low concurrency (1-64) where TP-sharded decode gives the
+# best per-user latency.
+
+model:
+  path: "deepseek-v4-pro"
+  container: "lmsysorg/sglang:deepseek-v4-grace-blackwell"
+  precision: "fp4"
+
+dynamo:
+  version: 0.8.1
+
+slurm:
+  time_limit: "8:00:00"
+
+health_check:
+  max_attempts: 1440
+  interval_seconds: 10
+
+resources:
+  gpu_type: "gb200"
+  gpus_per_node: 4
+  prefill_nodes: 2
+  decode_nodes: 2
+  prefill_workers: 1
+  decode_workers: 1
+  gpus_per_prefill: 8
+  gpus_per_decode: 8
+
+frontend:
+  type: dynamo
+  enable_multiple_frontends: false
+
+backend:
+  type: sglang
+  connector: null
+
+  # Env var set mirrored from PR #69 (the GB200 DSV4 aggregated baseline
+  # that's actually been run upstream) plus the disaggregation timeout
+  # triple — heartbeat 100k matches the DSR1 sglang disagg convention.
+  prefill_environment:
+    PYTHONUNBUFFERED: "1"
+    SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0"
+    NCCL_MNNVL_ENABLE: "1"
+    NCCL_CUMEM_ENABLE: "1"
+    SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000"
+    SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000"
+    SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000"
+
+  decode_environment:
+    PYTHONUNBUFFERED: "1"
+    SGLANG_JIT_DEEPGEMM_PRECOMPILE: "0"
+    NCCL_MNNVL_ENABLE: "1"
+    NCCL_CUMEM_ENABLE: "1"
+    SGLANG_DISAGGREGATION_HEARTBEAT_MAX_FAILURE: "100000"
+    SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT: "100000"
+    SGLANG_DISAGGREGATION_WAITING_TIMEOUT: "100000"
+
+  sglang_config:
+    prefill:
+      served-model-name: "deepseek-ai/DeepSeek-V4-Pro"
+      trust-remote-code: true
+      tensor-parallel-size: 8
+      dp-size: 8
+      ep-size: 8
+      enable-dp-attention: true
+      moe-a2a-backend: "deepep"
+      moe-runner-backend: "flashinfer_mxfp4"
+      chunked-prefill-size: 4096
+      disable-flashinfer-autotune: true
+      disable-radix-cache: true
+      mem-fraction-static: 0.82
+      context-length: 3072
+      max-running-requests: 16
+      stream-interval: 50
+      decode-log-interval: 1000
+      disaggregation-mode: "prefill"
+      disaggregation-transfer-backend: nixl
+
+    decode:
+      served-model-name: "deepseek-ai/DeepSeek-V4-Pro"
+      trust-remote-code: true
+      tensor-parallel-size: 8
+      moe-runner-backend: "flashinfer_mxfp4"
+      chunked-prefill-size: 4096
+      disable-flashinfer-autotune: true
+      disable-radix-cache: true
+      mem-fraction-static: 0.82
+      context-length: 3072
+      max-running-requests: 64
+      cuda-graph-max-bs: 64
+      stream-interval: 50
+      decode-log-interval: 1000
+      disaggregation-mode: "decode"
+      disaggregation-transfer-backend: nixl
+
+benchmark:
+  type: "sa-bench"
+  isl: 1024
+  osl: 1024
+  concurrencies: "1x4x8x16x32x64"
+  req_rate: "inf"
+  use_chat_template: false