Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -61,7 +61,7 @@ base:
SGLANG_HICACHE_DEBUG_LOG: '1'
SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384'
SGLANG_MOE_NVFP4_DISPATCH: '1'
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1'
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0'
SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8'
PIP_BREAK_SYSTEM_PACKAGES: '1'
args:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -51,7 +51,7 @@ roles:
SGLANG_HICACHE_DEBUG_LOG: '1'
SGLANG_HICACHE_DEBUG_SAMPLE_RATE: '16384'
SGLANG_MOE_NVFP4_DISPATCH: '1'
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1'
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0'
SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8'
PIP_BREAK_SYSTEM_PACKAGES: '1'
SGLANG_DG_CACHE_DIR: /deepgemm_cache
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,17 +7,17 @@ schema: 2
base:
model:
path: glm-5.2-fp4
container: dynamo-sglang
container: lmsysorg/sglang:v0.5.18-cu130@sha256:db37df8aded8fa169f7264404f3c893672f8046609a4a43d15bedcb4f63d692a
precision: fp4
identity:
model:
repo: nvidia/GLM-5.2-NVFP4
revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa
container:
image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642
image: lmsysorg/sglang:v0.5.18-cu130@sha256:db37df8aded8fa169f7264404f3c893672f8046609a4a43d15bedcb4f63d692a
frameworks:
dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd
sglang: nightly-dev-cu13-20260805-211ee642
sglang: 0.5.18
resources:
gpu_type: gb200
gpus_per_node: 4
Expand Down Expand Up @@ -124,7 +124,7 @@ base:
SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1'
SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512'
SGLANG_MOE_NVFP4_DISPATCH: '1'
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1'
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0'
SGLANG_ENABLE_THINKING: '1'
SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8'
SGLANG_REASONING_EFFORT: max
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -3,17 +3,17 @@ base:
name: gb200-fp4-glm5.2-agentx
model:
path: glm-5.2-fp4
container: dynamo-sglang
container: lmsysorg/sglang:v0.5.18-cu130@sha256:db37df8aded8fa169f7264404f3c893672f8046609a4a43d15bedcb4f63d692a
precision: fp4
identity:
model:
repo: nvidia/GLM-5.2-NVFP4
revision: aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa
container:
image: lmsysorg/sglang:v0.5.17-cu130
image: lmsysorg/sglang:v0.5.18-cu130@sha256:db37df8aded8fa169f7264404f3c893672f8046609a4a43d15bedcb4f63d692a
frameworks:
dynamo: 71eb001e17fa73c742f0afe1a6ed96836cb135fd
sglang: 0.5.17
sglang: 0.5.18
resources:
gpu_type: gb200
gpus_per_node: 4
Expand Down Expand Up @@ -130,7 +130,7 @@ base:
SGLANG_DISABLE_TP_MEMORY_INBALANCE_CHECK: '1'
SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK: '512'
SGLANG_MOE_NVFP4_DISPATCH: '1'
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '1'
SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE: '0'
SGLANG_ENABLE_THINKING: '1'
SGLANG_CLIP_MAX_NEW_TOKENS_ESTIMATION: '8'
SGLANG_REASONING_EFFORT: max
Expand Down
4 changes: 2 additions & 2 deletions inferencex-e2e/configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7591,7 +7591,7 @@ glm5.2-fp4-gb200-dynamo-sglang-agentic-agg:
# Shared deployment GPU power telemetry.
- "CONFIG_FILE=recipes/glm5.2/sglang/gb200-fp4/agentx/agg.yaml"
glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg:
image: lmsysorg/sglang:v0.5.17-cu130
image: lmsysorg/sglang:v0.5.18-cu130@sha256:db37df8aded8fa169f7264404f3c893672f8046609a4a43d15bedcb4f63d692a
model: nvidia/GLM-5.2-NVFP4
model-prefix: glm5.2
runner: cluster:gb200-nv
Expand Down Expand Up @@ -7625,7 +7625,7 @@ glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg:
# GLM-5.2 NVFP4 GB200 AgentX disaggregated variants use the committed
# thinking-on golden acceptance length for two speculative tokens.
glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp:
image: lmsysorg/sglang:nightly-dev-cu13-20260805-211ee642
image: lmsysorg/sglang:v0.5.18-cu130@sha256:db37df8aded8fa169f7264404f3c893672f8046609a4a43d15bedcb4f63d692a
model: nvidia/GLM-5.2-NVFP4
model-prefix: glm5.2
runner: cluster:gb200-nv
Expand Down
12 changes: 12 additions & 0 deletions inferencex-e2e/perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9370,3 +9370,15 @@
- "Run decoder SWA bounded replay with prefill graphs disabled on every point."
- "Throughput keeps the committed thinking-on golden AL 3.51 selected by the srt connector; evals use real verification."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3696

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

why are there 2?

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

first is txt version patch to NIXL
second is upgrade SGLang image that alr includes NICL 1.4.0

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

removed first one

- config-keys:
- glm5.2-fp4-gb200-dynamo-sglang-agentic-agg
- glm5.2-fp4-gb200-dynamo-sglang-agentic-disagg
- glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp
- glm5.2-fp4-gb200-dynamo-sglang-agentic-mtp-agg
scenario-type:
- agentic-coding
description:
- "Keep the GLM-5.2 NextN/MTP draft at shipped precision on GB200."
- "Use the digest-pinned upstream SGLang v0.5.18 CUDA 13 image with NIXL 1.4.0 for the two disaggregated recipes."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3401
Loading