From 434a4d79c425e5eb3d2a34a1cfc76f89ce79ad7b Mon Sep 17 00:00:00 2001 From: ganyi Date: Sun, 27 Sep 2026 23:01:12 -0500 Subject: [PATCH 1/2] fix(ci-mesh): align Kimi-K3 AgentX cells with the recipe (#2382) Sync the K3 P/D nightly cells with recipes/Agentic-Kimi-K3.md as of d8cdd4e2: - ATOM_USE_FLYDSL_FP8_PREFILL_ATTN=1 on every band - PrefillDelayer (ATOM_PREFILL_DECODE_INTERVAL=4, ATOM_PREFILL_DELAYER_MAX_QUEUE_MS=5000) from CONC 16 up - `cudagraph: auto` captures the dense range from bs=1 instead of 2 Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/benchmark/models_atomesh.yaml | 18 +++++++++++++++++- .github/scripts/atomesh/pd_server_atom.sh | 10 +++++----- 2 files changed, 22 insertions(+), 6 deletions(-) diff --git a/.github/benchmark/models_atomesh.yaml b/.github/benchmark/models_atomesh.yaml index ce65f84811..48d5430666 100644 --- a/.github/benchmark/models_atomesh.yaml +++ b/.github/benchmark/models_atomesh.yaml @@ -479,6 +479,8 @@ models: # move the published numbers. ATOM_USE_FLYDSL_GATHER_KV_B_PROJ: "1" ATOM_STATE_CHECKPOINT_DEMAND: "0" + # AgentX runs FlyDSL FP8 prefill attention on every band. + ATOM_USE_FLYDSL_FP8_PREFILL_ATTN: "1" # AgentX band default: ReplaySSM off on CONC 1/2/4, on for the mid band # (CONC 8-16), off again from CONC 32. Cases that declare their own env # block must restate the key, because a YAML merge key loses to an @@ -526,7 +528,7 @@ models: workers: 1 tp: 8 # AgentX pins the compile path and captures the dense query-token - # range 2..2*CONC*(1+spec) that `auto` derives per concurrency. + # range 1..2*CONC*(1+spec) that `auto` derives per concurrency. cudagraph: auto cudagraph_mode: FULL compilation_level: 3 @@ -589,9 +591,17 @@ models: PREFILL_KV_TRANSFER_CONFIG: >- {"kv_connector":"multi","connectors":[{"kv_connector":"mooncake","kv_role":"kv_producer","proxy_ip":"${ROLE_IP}","handshake_port":6301},{"kv_connector":"lmcache_offload","kv_role":"offload"}]} + # AgentX turns the PrefillDelayer on from CONC 16 up: 4 decode passes + # after each prefill, force-released once a prefill has queued 5 s. - <<: *kimi_k3_agentic_dcp8_lmcache name: kimi-k3-mxfp4-1p1d-tp8-dcp8-dspark3-agentic-lmcache-1m-c16 concurrency: [16] + env: + common: + ATOM_ENABLE_REPLAYSSM: "1" + ATOM_PREFILL_DECODE_INTERVAL: "4" + ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: "5000" + prefill: *kimi_k3_agentic_lmcache128_env - &kimi_k3_agentic_no_dspark <<: *kimi_k3_agentic_dcp8_lmcache @@ -606,6 +616,8 @@ models: env: common: ATOM_ENABLE_REPLAYSSM: "0" + ATOM_PREFILL_DECODE_INTERVAL: "4" + ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: "5000" # No CPU state tier here: the throughput band carries no draft model # and no ReplaySSM, so the whole per-rank CPU budget goes to the # paged KV instead of a state pool nothing replays from. @@ -640,6 +652,8 @@ models: common: ATOM_ENABLE_REPLAYSSM: "0" AITER_REUSE_IDENTICAL_COMM_GROUPS: "1" + ATOM_PREFILL_DECODE_INTERVAL: "4" + ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: "5000" prefill: <<: *kimi_k3_agentic_lmcache128_env LMCACHE_MAX_LOCAL_CPU_SIZE: "192" @@ -684,6 +698,8 @@ models: common: ATOM_ENABLE_REPLAYSSM: "0" AITER_REUSE_IDENTICAL_COMM_GROUPS: "1" + ATOM_PREFILL_DECODE_INTERVAL: "4" + ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: "5000" # The aggregated server reads env.decode. These are the C64 prefill's # LMCache settings, spelled out rather than merged from its anchor: # that anchor also carries the mooncake half of the `multi` diff --git a/.github/scripts/atomesh/pd_server_atom.sh b/.github/scripts/atomesh/pd_server_atom.sh index 7572fb415f..5f375469e8 100644 --- a/.github/scripts/atomesh/pd_server_atom.sh +++ b/.github/scripts/atomesh/pd_server_atom.sh @@ -407,7 +407,7 @@ if [[ "${DECODE_ENABLE_DP}" == "true" ]]; then fi # AgentX captures every query-token count the engine can produce, i.e. the dense -# range [2, graph_max] with graph_max = seqs * (1 + spec_tokens), where seqs +# range [1, graph_max] with graph_max = seqs * (1 + spec_tokens), where seqs # defaults to 2 * CONC. Concurrencies whose in-flight window is wider than # 2 * CONC pin seqs explicitly via cudagraph_max_num_seqs. auto_cudagraph_capture_sizes() { @@ -422,11 +422,11 @@ auto_cudagraph_capture_sizes() { seqs=$(( 2 * conc )) fi graph_max=$(( seqs * (1 + spec) )) - if (( graph_max < 2 )); then - graph_max=2 + if (( graph_max < 1 )); then + graph_max=1 fi - echo "[${role}] cudagraph auto range 2..${graph_max} (seqs=${seqs} spec=${spec})" >&2 - echo "[$(seq -s, 2 "${graph_max}")]" + echo "[${role}] cudagraph auto range 1..${graph_max} (seqs=${seqs} spec=${spec})" >&2 + echo "[$(seq -s, 1 "${graph_max}")]" } build_cudagraph_args() { From 7cd958f7d1f84583164c4bca2f405fc4fb9ec579 Mon Sep 17 00:00:00 2001 From: ganyi Date: Sun, 27 Sep 2026 23:05:56 -0500 Subject: [PATCH 2/2] fix(ci-mesh): drop PrefillDelayer env from K3 P/D cells Its benefit was measured on the aggregated recipe server; on the P/D split the decode instance never installs it and the prefill instance has no decode batch to protect, so keep the cells unchanged there. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/benchmark/models_atomesh.yaml | 14 -------------- 1 file changed, 14 deletions(-) diff --git a/.github/benchmark/models_atomesh.yaml b/.github/benchmark/models_atomesh.yaml index 48d5430666..687afa4cc2 100644 --- a/.github/benchmark/models_atomesh.yaml +++ b/.github/benchmark/models_atomesh.yaml @@ -591,17 +591,9 @@ models: PREFILL_KV_TRANSFER_CONFIG: >- {"kv_connector":"multi","connectors":[{"kv_connector":"mooncake","kv_role":"kv_producer","proxy_ip":"${ROLE_IP}","handshake_port":6301},{"kv_connector":"lmcache_offload","kv_role":"offload"}]} - # AgentX turns the PrefillDelayer on from CONC 16 up: 4 decode passes - # after each prefill, force-released once a prefill has queued 5 s. - <<: *kimi_k3_agentic_dcp8_lmcache name: kimi-k3-mxfp4-1p1d-tp8-dcp8-dspark3-agentic-lmcache-1m-c16 concurrency: [16] - env: - common: - ATOM_ENABLE_REPLAYSSM: "1" - ATOM_PREFILL_DECODE_INTERVAL: "4" - ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: "5000" - prefill: *kimi_k3_agentic_lmcache128_env - &kimi_k3_agentic_no_dspark <<: *kimi_k3_agentic_dcp8_lmcache @@ -616,8 +608,6 @@ models: env: common: ATOM_ENABLE_REPLAYSSM: "0" - ATOM_PREFILL_DECODE_INTERVAL: "4" - ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: "5000" # No CPU state tier here: the throughput band carries no draft model # and no ReplaySSM, so the whole per-rank CPU budget goes to the # paged KV instead of a state pool nothing replays from. @@ -652,8 +642,6 @@ models: common: ATOM_ENABLE_REPLAYSSM: "0" AITER_REUSE_IDENTICAL_COMM_GROUPS: "1" - ATOM_PREFILL_DECODE_INTERVAL: "4" - ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: "5000" prefill: <<: *kimi_k3_agentic_lmcache128_env LMCACHE_MAX_LOCAL_CPU_SIZE: "192" @@ -698,8 +686,6 @@ models: common: ATOM_ENABLE_REPLAYSSM: "0" AITER_REUSE_IDENTICAL_COMM_GROUPS: "1" - ATOM_PREFILL_DECODE_INTERVAL: "4" - ATOM_PREFILL_DELAYER_MAX_QUEUE_MS: "5000" # The aggregated server reads env.decode. These are the C64 prefill's # LMCache settings, spelled out rather than merged from its anchor: # that anchor also carries the mooncake half of the `multi`